)]}'
{"/COMMIT_MSG":[{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"ce90ddca130feb4e6973b067f74b6f3382ed5452","unresolved":true,"context_lines":[{"line_number":22,"context_line":""},{"line_number":23,"context_line":"Closes-Bug: #2136811"},{"line_number":24,"context_line":"Closes-Bug: #2153425"},{"line_number":25,"context_line":""},{"line_number":26,"context_line":"Change-Id: I19800408198051ad8dfe9ca04d1a02757a7725c7"},{"line_number":27,"context_line":"Signed-off-by: Masanori Ueno \u003cms-ueno@kddi.com\u003e"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":1,"id":"6d2630b1_efcdf002","line":25,"updated":"2026-06-04 08:42:18.000000000","message":"a release note would be appreciated as it\u0027s a bit of a behavioural change.","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"95f003118be3542fb54f66b47d38c892298f02aa","unresolved":false,"context_lines":[{"line_number":22,"context_line":""},{"line_number":23,"context_line":"Closes-Bug: #2136811"},{"line_number":24,"context_line":"Closes-Bug: #2153425"},{"line_number":25,"context_line":""},{"line_number":26,"context_line":"Change-Id: I19800408198051ad8dfe9ca04d1a02757a7725c7"},{"line_number":27,"context_line":"Signed-off-by: Masanori Ueno \u003cms-ueno@kddi.com\u003e"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":1,"id":"57a62111_baede78d","line":25,"in_reply_to":"6440c1b4_8f2a3b75","updated":"2026-06-11 10:27:50.000000000","message":"Done","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"0516073b0d0ddfee8d81261c077bd3e318baafdc","unresolved":true,"context_lines":[{"line_number":22,"context_line":""},{"line_number":23,"context_line":"Closes-Bug: #2136811"},{"line_number":24,"context_line":"Closes-Bug: #2153425"},{"line_number":25,"context_line":""},{"line_number":26,"context_line":"Change-Id: I19800408198051ad8dfe9ca04d1a02757a7725c7"},{"line_number":27,"context_line":"Signed-off-by: Masanori Ueno \u003cms-ueno@kddi.com\u003e"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":1,"id":"6440c1b4_8f2a3b75","line":25,"in_reply_to":"6d2630b1_efcdf002","updated":"2026-06-09 06:38:38.000000000","message":"Thank you for your review. I\u0027ve added the release notes.","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"728b93b363cff8fcc2ec3fa0e747bd18a3b233b6","unresolved":true,"context_lines":[{"line_number":4,"context_line":"Commit:     Masanori Ueno \u003cms-ueno@kddi.com\u003e"},{"line_number":5,"context_line":"CommitDate: 2026-06-08 23:45:59 +0900"},{"line_number":6,"context_line":""},{"line_number":7,"context_line":"NUMA live-migration: ensure allocation_ratio is respected"},{"line_number":8,"context_line":""},{"line_number":9,"context_line":"In Bug #2153425, when `_call_livem_checks_on_host()` is invoked from"},{"line_number":10,"context_line":"`LiveMigrationTask._find_destination()`, the `limits` produced by the"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":2,"id":"c34a87f1_d2c36fe2","line":7,"range":{"start_line":7,"start_character":28,"end_line":7,"end_character":44},"updated":"2026-06-08 14:58:21.000000000","message":"this is kind of misleading as its not really about allcoation ratos more so about limits in general.\n\nthere is a boradaer issue that sylvain was already looking into related to numa aware vsiwchws whic results for a similar underlying issue where the instnace claim on teh compute node is not correctly considering the limits.","commit_id":"5596af305569d5006ff96c943b3ad0678128e640"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"95f003118be3542fb54f66b47d38c892298f02aa","unresolved":false,"context_lines":[{"line_number":4,"context_line":"Commit:     Masanori Ueno \u003cms-ueno@kddi.com\u003e"},{"line_number":5,"context_line":"CommitDate: 2026-06-08 23:45:59 +0900"},{"line_number":6,"context_line":""},{"line_number":7,"context_line":"NUMA live-migration: ensure allocation_ratio is respected"},{"line_number":8,"context_line":""},{"line_number":9,"context_line":"In Bug #2153425, when `_call_livem_checks_on_host()` is invoked from"},{"line_number":10,"context_line":"`LiveMigrationTask._find_destination()`, the `limits` produced by the"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":2,"id":"dff59f98_cef5d6dc","line":7,"range":{"start_line":7,"start_character":28,"end_line":7,"end_character":44},"in_reply_to":"347152c9_2482c927","updated":"2026-06-11 10:27:50.000000000","message":"Done","commit_id":"5596af305569d5006ff96c943b3ad0678128e640"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"0516073b0d0ddfee8d81261c077bd3e318baafdc","unresolved":true,"context_lines":[{"line_number":4,"context_line":"Commit:     Masanori Ueno \u003cms-ueno@kddi.com\u003e"},{"line_number":5,"context_line":"CommitDate: 2026-06-08 23:45:59 +0900"},{"line_number":6,"context_line":""},{"line_number":7,"context_line":"NUMA live-migration: ensure allocation_ratio is respected"},{"line_number":8,"context_line":""},{"line_number":9,"context_line":"In Bug #2153425, when `_call_livem_checks_on_host()` is invoked from"},{"line_number":10,"context_line":"`LiveMigrationTask._find_destination()`, the `limits` produced by the"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":2,"id":"347152c9_2482c927","line":7,"range":{"start_line":7,"start_character":28,"end_line":7,"end_character":44},"in_reply_to":"c34a87f1_d2c36fe2","updated":"2026-06-09 06:38:38.000000000","message":"Thank you for your review. I’ve gone ahead and revised the commit message. I’d appreciate it if you could take a look.","commit_id":"5596af305569d5006ff96c943b3ad0678128e640"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"728b93b363cff8fcc2ec3fa0e747bd18a3b233b6","unresolved":true,"context_lines":[{"line_number":11,"context_line":"NUMA topology filter are not set on `self.limits`, resulting in limits"},{"line_number":12,"context_line":"not being propagated to subsequent processing."},{"line_number":13,"context_line":"As a result, CPU limit checks in `nova/virt/hardware.py` were not"},{"line_number":14,"context_line":"executed."},{"line_number":15,"context_line":""},{"line_number":16,"context_line":"Fix this by setting `self.limits` before `_call_livem_checks_on_host()`"},{"line_number":17,"context_line":"is called in `_find_destination()`"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":2,"id":"3f082c64_f208bfad","line":14,"updated":"2026-06-08 14:58:21.000000000","message":"while we do allow disk and cpu oversubscriton for numa instance we need to be carful as we do not allow memory oversubsciption for numa instances.","commit_id":"5596af305569d5006ff96c943b3ad0678128e640"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"95f003118be3542fb54f66b47d38c892298f02aa","unresolved":false,"context_lines":[{"line_number":11,"context_line":"NUMA topology filter are not set on `self.limits`, resulting in limits"},{"line_number":12,"context_line":"not being propagated to subsequent processing."},{"line_number":13,"context_line":"As a result, CPU limit checks in `nova/virt/hardware.py` were not"},{"line_number":14,"context_line":"executed."},{"line_number":15,"context_line":""},{"line_number":16,"context_line":"Fix this by setting `self.limits` before `_call_livem_checks_on_host()`"},{"line_number":17,"context_line":"is called in `_find_destination()`"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":2,"id":"a64a7471_23591f3b","line":14,"in_reply_to":"3f082c64_f208bfad","updated":"2026-06-11 10:27:50.000000000","message":"Done","commit_id":"5596af305569d5006ff96c943b3ad0678128e640"}],"/PATCHSET_LEVEL":[{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"ce90ddca130feb4e6973b067f74b6f3382ed5452","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":1,"id":"132f2534_990d0e9e","updated":"2026-06-04 08:42:18.000000000","message":"excellent efforts, thanks for having worked on a solution. I just have a design concern, which is to fix the root cause instead of recreating a fake object.","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":28329,"name":"yangjianfeng","display_name":"JeffYang","email":"yjf1970231893@gmail.com","username":"yangjianfeng"},"change_message_id":"261bbdb338ac402b01ddc571a502c19a892c2a16","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"3c83d8ed_2bfe465c","updated":"2026-06-09 07:00:52.000000000","message":"How about force live-migration? Does it has the same issue?","commit_id":"53dc5cf00a7239680c31733a32fd578004692f8b"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"95f003118be3542fb54f66b47d38c892298f02aa","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"e94c3e7f_1dbce0e2","updated":"2026-06-11 10:27:50.000000000","message":"Nothing critical, basically close to approve it, but the relnote should be changed. Please also rebase this patch on the latest revision of https://review.opendev.org/c/openstack/nova/+/988777 (PS3)\n\n-1 for signaling a relnote change, but as said, very close to be approved, very good work.","commit_id":"53dc5cf00a7239680c31733a32fd578004692f8b"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"f97491b5326163537f1081ee0826d184aaa6382f","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"72c9b01e_177afdce","updated":"2026-06-09 04:18:24.000000000","message":"recheck","commit_id":"53dc5cf00a7239680c31733a32fd578004692f8b"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"30a7ae1da2c0bfa4a9d8b99e543db0e940d3ee16","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"bf8f4486_0566a4ec","updated":"2026-06-08 23:55:16.000000000","message":"recheck","commit_id":"53dc5cf00a7239680c31733a32fd578004692f8b"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"271d2a8aa401a52060fa6aa5847f5c699fea7329","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":4,"id":"e1403bde_d05cc94f","in_reply_to":"3c83d8ed_2bfe465c","updated":"2026-06-09 10:08:19.000000000","message":"Yes. force live-migration has the same issue, and I understand that this patch will not fix it. \nThis is because `self.query_client.select_destinations()` is not called during a force live-migration, so there is no logic to retrieve the limits.\n\nHowever, since limits are generated by the scheduler, I believe this is unavoidable in force live-migration, where the scheduler is bypassed.","commit_id":"53dc5cf00a7239680c31733a32fd578004692f8b"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"4ad8c66825d122748a7fa189bed4f57e237939e5","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"e30a32bb_1cdc6c95","in_reply_to":"e94c3e7f_1dbce0e2","updated":"2026-06-11 14:15:09.000000000","message":"Thank you for your detailed review! \nI\u0027ve rebased this patch on the latest revision of the test patch, and updated the release notes.","commit_id":"53dc5cf00a7239680c31733a32fd578004692f8b"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"c37763e07bea33b701f97bea441c0bdf391b6aef","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":5,"id":"787796b1_3d71e6d8","updated":"2026-06-15 12:24:32.000000000","message":"all good now, thanks for the hard work !","commit_id":"e09757f5e0a2d03170fd3895246ad72b6d8b5c07"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"9f0c17f8903f513ce789800d674539c3835faddb","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":5,"id":"8d164f04_db437c50","updated":"2026-06-11 22:26:32.000000000","message":"recheck","commit_id":"e09757f5e0a2d03170fd3895246ad72b6d8b5c07"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"037c23cd518815080e221db953622a83fb83e9ce","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":6,"id":"8d4dead5_bc063a96","updated":"2026-06-19 10:37:20.000000000","message":"we should do more testing of the other move ops after this is merged to validate are we missign the limits in other places but i think we can move forward with this","commit_id":"817d800dcba3ca171d1d1a54192390bb9925c14a"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"c5213ede99d23ef64ddbeb7ef26cd2574a0e0aa4","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":7,"id":"b5d58009_c4bbbed5","updated":"2026-06-23 22:38:53.000000000","message":"recheck","commit_id":"f4709ec59f0c16358ee94a76e8e1addff5a5339e"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"9c17d531bde07ee29261dd1c87d5f2812feac476","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":7,"id":"196291cf_35559e66","updated":"2026-06-23 13:47:57.000000000","message":"trivial rebase for parent changes","commit_id":"f4709ec59f0c16358ee94a76e8e1addff5a5339e"}],"nova/compute/claims.py":[{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"ce90ddca130feb4e6973b067f74b6f3382ed5452","unresolved":true,"context_lines":[{"line_number":150,"context_line":"            if ("},{"line_number":151,"context_line":"                not limit and"},{"line_number":152,"context_line":"                \u0027cpu_allocation_ratio\u0027 in compute_node and"},{"line_number":153,"context_line":"                \u0027ram_allocation_ratio\u0027 in compute_node"},{"line_number":154,"context_line":"            ):"},{"line_number":155,"context_line":"                limit \u003d objects.NUMATopologyLimits("},{"line_number":156,"context_line":"                    cpu_allocation_ratio\u003dcompute_node.cpu_allocation_ratio,"}],"source_content_type":"text/x-python","patch_set":1,"id":"c52396a3_e5723b66","line":153,"updated":"2026-06-04 08:42:18.000000000","message":"the last two terms of the conditions will always be True, compute_node *always* have cpu and ram allocation ratios set by default.","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"0516073b0d0ddfee8d81261c077bd3e318baafdc","unresolved":true,"context_lines":[{"line_number":150,"context_line":"            if ("},{"line_number":151,"context_line":"                not limit and"},{"line_number":152,"context_line":"                \u0027cpu_allocation_ratio\u0027 in compute_node and"},{"line_number":153,"context_line":"                \u0027ram_allocation_ratio\u0027 in compute_node"},{"line_number":154,"context_line":"            ):"},{"line_number":155,"context_line":"                limit \u003d objects.NUMATopologyLimits("},{"line_number":156,"context_line":"                    cpu_allocation_ratio\u003dcompute_node.cpu_allocation_ratio,"}],"source_content_type":"text/x-python","patch_set":1,"id":"4420108a_4313d76d","line":153,"in_reply_to":"c52396a3_e5723b66","updated":"2026-06-09 06:38:38.000000000","message":"I have modified the code so that it no longer generates fake limits but instead uses the limits passed from the scheduler.","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"ce90ddca130feb4e6973b067f74b6f3382ed5452","unresolved":true,"context_lines":[{"line_number":155,"context_line":"                limit \u003d objects.NUMATopologyLimits("},{"line_number":156,"context_line":"                    cpu_allocation_ratio\u003dcompute_node.cpu_allocation_ratio,"},{"line_number":157,"context_line":"                    ram_allocation_ratio\u003dcompute_node.ram_allocation_ratio"},{"line_number":158,"context_line":"                )"},{"line_number":159,"context_line":""},{"line_number":160,"context_line":"            instance_topology \u003d hardware.numa_fit_instance_to_host("},{"line_number":161,"context_line":"                host_topology,"}],"source_content_type":"text/x-python","patch_set":1,"id":"cf5d4654_06e045a2","line":158,"updated":"2026-06-04 08:42:18.000000000","message":"recreating fake limits is a simple solution but can hit some problems : \n* the NUMATopologyLimits object created from the scheduler can have network_metadata field set, you don\u0027t rehydrate it here\n* the ratio values can be different from the compute nova.conf option value if there are some config overrides on the scheduler side\n\nI\u0027d rather want the root cause to be fixed, which is on the conductor service. Let me explain shortly : \n* when asking the scheduler to find a destination when live-migrating, we return the limits there https://github.com/openstack/nova/blob/278c6e305c3da085fd1c1e338e95ce4d63631272/nova/conductor/tasks/live_migrate.py#L100\n* by default self.limits is True https://github.com/openstack/nova/blob/278c6e305c3da085fd1c1e338e95ce4d63631272/nova/conductor/tasks/live_migrate.py#L70\n* inside _find_destination() we call _call_livem_checks_on_host() which checks the limits https://github.com/openstack/nova/blob/278c6e305c3da085fd1c1e338e95ce4d63631272/nova/conductor/tasks/live_migrate.py#L380\n\n... but since _call_livem_checks_on_host() is called *within* _find_destinations() we *always checks the limits against None !\n\nlong story short, you just need to set self.limits \u003d selection.limits before that call (or use selection.limits instead of self.limits in that method) so then the limits would be sent over RPC to the computes and then the live migration claim here could get for free instead of resynthesizing it.","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"0516073b0d0ddfee8d81261c077bd3e318baafdc","unresolved":true,"context_lines":[{"line_number":155,"context_line":"                limit \u003d objects.NUMATopologyLimits("},{"line_number":156,"context_line":"                    cpu_allocation_ratio\u003dcompute_node.cpu_allocation_ratio,"},{"line_number":157,"context_line":"                    ram_allocation_ratio\u003dcompute_node.ram_allocation_ratio"},{"line_number":158,"context_line":"                )"},{"line_number":159,"context_line":""},{"line_number":160,"context_line":"            instance_topology \u003d hardware.numa_fit_instance_to_host("},{"line_number":161,"context_line":"                host_topology,"}],"source_content_type":"text/x-python","patch_set":1,"id":"0672c8f1_c34ee84b","line":158,"in_reply_to":"7ac60b3c_82fe5112","updated":"2026-06-09 06:38:38.000000000","message":"I have modified the code so that `self.limits \u003d selection.limits` is set before calling `_call_livem_checks_on_host`. I have also added logic to reset the value to `None` if the host selection fails.","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"e0f1a2e2dd33974f5e66eeb0dffb3ced3a4d90cc","unresolved":true,"context_lines":[{"line_number":155,"context_line":"                limit \u003d objects.NUMATopologyLimits("},{"line_number":156,"context_line":"                    cpu_allocation_ratio\u003dcompute_node.cpu_allocation_ratio,"},{"line_number":157,"context_line":"                    ram_allocation_ratio\u003dcompute_node.ram_allocation_ratio"},{"line_number":158,"context_line":"                )"},{"line_number":159,"context_line":""},{"line_number":160,"context_line":"            instance_topology \u003d hardware.numa_fit_instance_to_host("},{"line_number":161,"context_line":"                host_topology,"}],"source_content_type":"text/x-python","patch_set":1,"id":"7ac60b3c_82fe5112","line":158,"in_reply_to":"cf5d4654_06e045a2","updated":"2026-06-04 12:01:59.000000000","message":"Thank you very much for your thorough and detailed review！\n\nI have actually uploaded another patch for this issue, which addresses the point you raised in your review. I would greatly appreciate it if you could take a look and let me know your thoughts.\nhttps://review.opendev.org/c/openstack/nova/+/990648","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"}],"nova/tests/functional/regressions/test_bug_2153425.py":[{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"ce90ddca130feb4e6973b067f74b6f3382ed5452","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":1,"id":"eb231460_b774085a","updated":"2026-06-04 08:42:18.000000000","message":"we usually provide two gerrit changes for a regression bug fix : \n* we first provide the regression test that shows the error (eg. something like https://review.opendev.org/c/openstack/nova/+/988777) \n* then in the bugfix itself, we just amend the test to modify the asserts to say it works now.","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"95f003118be3542fb54f66b47d38c892298f02aa","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":1,"id":"5ca2ceaf_2686c99f","in_reply_to":"47de9a2f_744ec327","updated":"2026-06-11 10:27:50.000000000","message":"Done","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"0516073b0d0ddfee8d81261c077bd3e318baafdc","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":1,"id":"47de9a2f_744ec327","in_reply_to":"5c5da456_9be1adca","updated":"2026-06-09 06:38:38.000000000","message":"I treated https://review.opendev.org/c/openstack/nova/+/988777 as a regression test patch and made this patch a bug fix for it.","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"e0f1a2e2dd33974f5e66eeb0dffb3ced3a4d90cc","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":1,"id":"5c5da456_9be1adca","in_reply_to":"eb231460_b774085a","updated":"2026-06-04 12:01:59.000000000","message":"Thank you very much for your feedback. I will rework this change into a regression test patch and update it\nhttps://review.opendev.org/c/openstack/nova/+/988777","commit_id":"4f6b10439e1cb32b6322c1f62e5f33a59d75d9c8"}],"releasenotes/notes/bug-2153425-numa-live-migration-limit-cba1523e91f1f8a9.yaml":[{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"95f003118be3542fb54f66b47d38c892298f02aa","unresolved":true,"context_lines":[{"line_number":1,"context_line":"---"},{"line_number":2,"context_line":"fixes:"},{"line_number":3,"context_line":"  - |"},{"line_number":4,"context_line":"    Fixed the ``bug #2153425``_ where ``cpu_allocation_ratio`` was not considered"},{"line_number":5,"context_line":"    during live migration on NUMA-capable compute hosts, leading to CPU usage"},{"line_number":6,"context_line":"    per NUMA cell exceeding the configured limit on the destination host after"},{"line_number":7,"context_line":"    migration."}],"source_content_type":"text/x-yaml","patch_set":4,"id":"2fc4c08c_6b2bec43","line":4,"range":{"start_line":4,"start_character":14,"end_line":4,"end_character":30},"updated":"2026-06-11 10:27:50.000000000","message":"link target doesn\u0027t work as you can see here https://94e1717ddb0b25fac440-452dd4b4c84f55a29b71ee3516016f2d.ssl.cf5.rackcdn.com/openstack/b0e5dbbdd6114d08b94a061532610b34/docs/unreleased.html\n\nYou can take examples on the other reno files we have like https://review.opendev.org/c/openstack/nova/+/983672/6/releasenotes/notes/bug-2134375-580c10cfefc279fd.yaml\n\nKeep the release note minimal and just mention that scheduler limits were not honored during live-migration, leading to allocation ratios not being enforced on compute claim (as just one of the issues)","commit_id":"53dc5cf00a7239680c31733a32fd578004692f8b"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"4ad8c66825d122748a7fa189bed4f57e237939e5","unresolved":true,"context_lines":[{"line_number":1,"context_line":"---"},{"line_number":2,"context_line":"fixes:"},{"line_number":3,"context_line":"  - |"},{"line_number":4,"context_line":"    Fixed the ``bug #2153425``_ where ``cpu_allocation_ratio`` was not considered"},{"line_number":5,"context_line":"    during live migration on NUMA-capable compute hosts, leading to CPU usage"},{"line_number":6,"context_line":"    per NUMA cell exceeding the configured limit on the destination host after"},{"line_number":7,"context_line":"    migration."}],"source_content_type":"text/x-yaml","patch_set":4,"id":"7b783bee_54820adb","line":4,"range":{"start_line":4,"start_character":14,"end_line":4,"end_character":30},"in_reply_to":"2fc4c08c_6b2bec43","updated":"2026-06-11 14:15:09.000000000","message":"Thank you very much for your feedback. Sorry, that was a typo on my side. I\u0027ve made the corrections.\n\nI\u0027ve also tried to keep the release notes brief.","commit_id":"53dc5cf00a7239680c31733a32fd578004692f8b"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"037c23cd518815080e221db953622a83fb83e9ce","unresolved":false,"context_lines":[{"line_number":1,"context_line":"---"},{"line_number":2,"context_line":"fixes:"},{"line_number":3,"context_line":"  - |"},{"line_number":4,"context_line":"    Fixed the ``bug #2153425``_ where ``cpu_allocation_ratio`` was not considered"},{"line_number":5,"context_line":"    during live migration on NUMA-capable compute hosts, leading to CPU usage"},{"line_number":6,"context_line":"    per NUMA cell exceeding the configured limit on the destination host after"},{"line_number":7,"context_line":"    migration."}],"source_content_type":"text/x-yaml","patch_set":4,"id":"b431d792_b3ae5a72","line":4,"range":{"start_line":4,"start_character":14,"end_line":4,"end_character":30},"in_reply_to":"7b783bee_54820adb","updated":"2026-06-19 10:37:20.000000000","message":"Done","commit_id":"53dc5cf00a7239680c31733a32fd578004692f8b"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"c2bfe3ce724bd0725fc65a5f4c204734c8df2398","unresolved":true,"context_lines":[{"line_number":3,"context_line":"  - |"},{"line_number":4,"context_line":"    `Bug #2153425`_: Fix scheduler limits not being honored during"},{"line_number":5,"context_line":"    live migration compute claims, causing allocation ratios to be"},{"line_number":6,"context_line":"    ignored."},{"line_number":7,"context_line":""},{"line_number":8,"context_line":"    .. _bug #2153425: https://launchpad.net/bugs/2153425"}],"source_content_type":"text/x-yaml","patch_set":5,"id":"5dc250ab_baa58796","line":6,"updated":"2026-06-15 13:10:00.000000000","message":"cpu allocation ratio is enfoced globally not per numa node.\nWe coudl supprot it per numa node but that was orgianly put out of scope fo the orginal numa feature so this is technialy a new feature not a bug\n\nthe minim requirement for a numa flavor is to enable numa aware memoy tracking via\nhw:mem_page_size in the flavor or hw_mem_page_size in the image unless the host is using file backed memroy.\n\nthe reason that vms were not being blanaced is taht\nthe guest did not requist pinned cpus or enable numa aware memory allcoation\n\nit just enabled enabled hw:numa_nodes\u003d1  which as i said is only valid if you use file backed memory.\n\nand even then that does not enabel numa local cpu allocation ratios as that has never been supported.\n\nwe can support it but we didn\u0027t in the past as to do that properly we would also need to change how we report CPU to placement\n\nthe only oversubscrption check we for shared cpus is to make sure a vm dose not oversubscribe against itslef\n\nwhat that means is if you have a single numa node vm we will not allow you to both on a host that has fewer vcpu on a given numa node then the vms reqeusts.\n\nim not agaisnt extendeing the numa supprot with vcpu numa awareness but that is a new feature rather hten a bug","commit_id":"e09757f5e0a2d03170fd3895246ad72b6d8b5c07"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"9316c69a76dabfa72cda48b4bf04126371860320","unresolved":true,"context_lines":[{"line_number":3,"context_line":"  - |"},{"line_number":4,"context_line":"    `Bug #2153425`_: Fix scheduler limits not being honored during"},{"line_number":5,"context_line":"    live migration compute claims, causing allocation ratios to be"},{"line_number":6,"context_line":"    ignored."},{"line_number":7,"context_line":""},{"line_number":8,"context_line":"    .. _bug #2153425: https://launchpad.net/bugs/2153425"}],"source_content_type":"text/x-yaml","patch_set":5,"id":"f7f0dbbb_9563413e","line":6,"in_reply_to":"050f8268_ceeeee18","updated":"2026-06-18 13:29:45.000000000","message":"To be clear, this patch does not guarentee full NUMA-aware CPU\noversubscription for all move operation\n(cold migrate, resize, evacuate, unshelve) as there are multiple\ndistict codepaths that would need testing.\n\nThere is legacy code that accounts for allocation ratios deep in the\n_numa_fit_instance_cell funcation, but my orginal objecton was trigged\nby the fact that relying on hw:numa_nodes\u003d1 without proper memory\nconstraints is dangerous and not recommended for production.\n\nThat is the main reason i was unhappy with the repoducer as we\ngenerally consider bugs that only happen when hw:numa_nodes\u003d1 is set\nwithout hw:mem_page_size as invalid. the only expction to that rule\nis file backed memroy so regardless of if you are using htat i guess\nthe real qustion we have to answer is is your expctation that the behavior\non live migrate and create should be consitent holds.\n\nTo answer that we need ot looks at how teh compute manger selects a numa node\nfor an instance as part of the instance claim.\n\nlets start with some context.\n\nOriginally, limits were used only by filters to select a host. When\nNUMA was added, we extended these limits. The compute node currently\nreconstructs NUMA requirements from the flavor and image on move operations\nand uses the instance.numa toplogy on spawns. the intent was to not depend on the\nscheuler popuplating an numa constatints and passign down the limits object for\nthe numa enforcement to work as we didnt want to rely on operators enableing\nthe numa toplogy filter. note we condier using numa without it to be unsupproted\nbut we did not want the behavior of the numa code on the compute to depend\non the config the schduler has.\n\nWhen we are on the compute node in the instance claim, we use the same\nfunction to generate the NUMA constraints:\nhttps://github.com/openstack/nova/blob/master/nova/compute/claims.py#L192-L194\nfor move operations, although for boot we just use the copy from the\ninstance object:\n\nhttps://github.com/openstack/nova/blob/master/nova/compute/claims.py#L77-L79\n\nso for spwan we are calling the numa_get_constraints function earlier and storign the obejct in the instnace but for moves we generate it cleanly on the compute.\nThe reason we build it from scratch for move operations is we could be\ndoing a resize, so the instance\u0027s current topology may not be correct.\n\nin your case, since you are not using asymmetric CPU requests, and are just setting\nhw:numa_nodes we will take the auto toplogy plah so this is done\nby first building an InstanceNumaTopology object:\nFor a 1-node request, we end up with a single InstanceNUMACell subobject. \n\nhttps://github.com/openstack/nova/blob/master/nova/virt/hardware.py#L1757-L1787\nwith just the requests from the flavor.vcpu and flavor.memory_mb, splitting\nthe cores and RAM evenly between all NUMA nodes (in your case, just 1).\n\nOnce we have that object, we copy some info that applies to all the NUMA\ncells into them:\nhttps://github.com/openstack/nova/blob/master/nova/virt/hardware.py#L2479-L2485\nand return back the object from numa_get_constraints.\n\nregarless of how we get the numa toplogy object when we are doing the instnace claim on the host we call _test_numa_topology https://github.com/openstack/nova/blob/master/nova/compute/claims.py#L77-L79\nthat is the entryp point that take a partical constucted InstanceNumaTopology object\nthat only has the constraits and invoke hardware.numa_fit_instance_to_host\n\ninternlly hardware.numa_fit_instance_to_host works by iterating \niterate over a sorted set of host NUMA nodes using itertools.permutations and\ntake the first one that fits for each guest numa node (see:\nhttps://github.com/openstack/nova/blob/master/nova/virt/hardware.py#L2718-L2719).\n\"Fits\" is determined by _numa_fit_instance_cell (see:\nhttps://github.com/openstack/nova/blob/master/nova/virt/hardware.py#L1054):\n\n1. CPU COUNT: We first verify the requested CPUs do not exceed the\n   host NUMA node core count (see:\n   https://github.com/openstack/nova/blob/master/nova/virt/hardware.py#L1174-L1186).\n2. ALLOCATION RATIOS: Contrary to my initial assessment, we DO\n   consider allocation ratios at the final stage of the fit logic\n   (see:\n   https://github.com/openstack/nova/blob/d5b82e107e46a8e29483c22f9713335f2e028939/nova/virt/hardware.py#L1213-L1237).\n\n\nThe NUMA balancing feature was added 4 years ago (see:\nhttps://github.com/openstack/nova/commit/d13412648d011994a146dac1e7214ead3b82b31b)\nuses CPU usage to bias selection but does not consult allocation\nratios. It was intended to pack/spread based on pinned CPUs and\nhugepages, not to enforce general oversubscription ratios.\nthat is being done only via the _numa_fit_instance_cell the intent of the\nnuma balancing freature was to pack or spread based on usage/aviablity\nand allow capcity checkign to be done by the exiting _numa_fit_instance_cell code\npaths. since that capscity code path happens to allication ratio aware\nif we fixed all place where were should be passing the limits object to the compute\nnode to actully do it then yes we coudl supprot numa level  allcaotion ration enformcne with the only caveat being that the same allication ration will be enfoce for each numa node.\n\nLooping back to my orginal objection, a flavor with hw:numa_nodes\u003d1 is not safe without file-backed memory.  Without setting hw:mem_page_size, it is possible to overcommit memory on a NUMA node because its entrilly possible to have free memory on other numa ndoes and as a result not have enough reserved space for the qemu process per guest overhead, leading the kernel OOM killer to reap VMs\nbecause it operates on a per-NUMA node basis not globally when the kernel need memory.\n\nStandard configurations like reserved_host_memory_mb (see:\nhttps://docs.openstack.org/nova/latest/configuration/config.html#DEFAULT.reserved_host_memory_mb)\ndo not reserve memory in these NUMA-specific paths. To safely reserve\nmemory for QEMU overhead, you must use reserved_huge_pages (see:\nhttps://docs.openstack.org/nova/latest/configuration/config.html#DEFAULT.reserved_huge_pages) to reseve i memoy at the sytems\ndefualt page size which is typiclly 4k pages on linux/x86.\n(e.g., node:0,size:4,count:512) and ensure the VM uses\nhw:mem_page_size\u003dsmall for NUMA-aware memory protection.\n\nAlternatively, set the memory allocation ratio \u003c1.0 so that _numa_fit_instance_cell\nwould give you a per numa node buffer so if you set \n```\n[DEFAULT]\nram_allocation_ratio \u003d 0.85\n```\nhttps://docs.openstack.org/nova/latest/configuration/config.html#DEFAULT.ram_allocation_ratio\nthat can work but its not safe to set it to 1.0 or higher for host with numa instnaces. Do not rely on swap for NUMA instances, as the kernel will not aggressively swap memory if a different NUMA node has free RAM, the OOM killer\nwill still trigger if the local node is exhausted and that will result in vms beign killed.\n\ni woudl perfer if we updated the repoducer to either use file backed memory or request hw:mem_page_size\u003dsmall\n\nbut we can proceed with this fix in general as a bug fix wehre the bug is that we shoudl be passing limits regardlles of if the numa toplogiy filter is enabled or not for all move operations.\n\nthis is needed to also fix numa aware vswtich and some other issues.\n\nthis was sort of unintlly regresses a long time ago so this change is a particl fix of the wider issue https://bugs.launchpad.net/nova/+bug/2145135","commit_id":"e09757f5e0a2d03170fd3895246ad72b6d8b5c07"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"c3878b45c5821b61ff42d04325ebbf1789acc4f3","unresolved":true,"context_lines":[{"line_number":3,"context_line":"  - |"},{"line_number":4,"context_line":"    `Bug #2153425`_: Fix scheduler limits not being honored during"},{"line_number":5,"context_line":"    live migration compute claims, causing allocation ratios to be"},{"line_number":6,"context_line":"    ignored."},{"line_number":7,"context_line":""},{"line_number":8,"context_line":"    .. _bug #2153425: https://launchpad.net/bugs/2153425"}],"source_content_type":"text/x-yaml","patch_set":5,"id":"e4ea12bb_35c7a38f","line":6,"in_reply_to":"5dc250ab_baa58796","updated":"2026-06-15 13:37:38.000000000","message":"Thanks for the clarification.\n\nFirst, even when we specify hw:mem_page_size in addition to hw:numa_nodes\u003d1, we can still reproduce the behavior in our environment where CPU allocation ratios are not effectively enforced on a per-NUMA-node basis during live migration. If it would be helpful, we believe we could also reproduce this with a regression test and share the details.\n\nSecond, regarding the point that this should be considered a new feature rather than a bug, what seems unusual to us is that the behavior differs between instance creation and live migration.\n\nIn our environment, when a server is initially created, the scheduler appears to take per-NUMA CPU allocation ratios into account, and the instance is placed as expected. However, during live migration, the destination host selection does not appear to apply the same consideration. From our perspective, this inconsistency is what makes the behavior look more like a bug than a missing feature.\n\nIf per-NUMA CPU allocation ratios have never been intended to be supported at any stage of scheduling, we would expect the behavior to be consistent between initial scheduling and live migration. The fact that they appear to behave differently is what we find surprising.","commit_id":"e09757f5e0a2d03170fd3895246ad72b6d8b5c07"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"037c23cd518815080e221db953622a83fb83e9ce","unresolved":false,"context_lines":[{"line_number":3,"context_line":"  - |"},{"line_number":4,"context_line":"    `Bug #2153425`_: Fix scheduler limits not being honored during"},{"line_number":5,"context_line":"    live migration compute claims, causing allocation ratios to be"},{"line_number":6,"context_line":"    ignored."},{"line_number":7,"context_line":""},{"line_number":8,"context_line":"    .. _bug #2153425: https://launchpad.net/bugs/2153425"}],"source_content_type":"text/x-yaml","patch_set":5,"id":"cd47265d_a62da855","line":6,"in_reply_to":"c5c8f5a8_fb0992cc","updated":"2026-06-19 10:37:20.000000000","message":"Done","commit_id":"e09757f5e0a2d03170fd3895246ad72b6d8b5c07"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"88b5f02a115be924b879f3a406282bc0492eb8c4","unresolved":true,"context_lines":[{"line_number":3,"context_line":"  - |"},{"line_number":4,"context_line":"    `Bug #2153425`_: Fix scheduler limits not being honored during"},{"line_number":5,"context_line":"    live migration compute claims, causing allocation ratios to be"},{"line_number":6,"context_line":"    ignored."},{"line_number":7,"context_line":""},{"line_number":8,"context_line":"    .. _bug #2153425: https://launchpad.net/bugs/2153425"}],"source_content_type":"text/x-yaml","patch_set":5,"id":"050f8268_ceeeee18","line":6,"in_reply_to":"e4ea12bb_35c7a38f","updated":"2026-06-18 07:44:19.000000000","message":"That said, I\u0027m fine with either a bug-fix release or a new feature release. @sbauza@redhat.com, if you have any thoughts on this, I\u0027d appreciate your input.","commit_id":"e09757f5e0a2d03170fd3895246ad72b6d8b5c07"},{"author":{"_account_id":38918,"name":"Masanori Ueno","display_name":"Masanori Ueno","email":"ms-ueno@kddi.com","username":"masan4444"},"change_message_id":"d52ad52c6543a2b29ff6ce8a583cdc151dbfafd7","unresolved":true,"context_lines":[{"line_number":3,"context_line":"  - |"},{"line_number":4,"context_line":"    `Bug #2153425`_: Fix scheduler limits not being honored during"},{"line_number":5,"context_line":"    live migration compute claims, causing allocation ratios to be"},{"line_number":6,"context_line":"    ignored."},{"line_number":7,"context_line":""},{"line_number":8,"context_line":"    .. _bug #2153425: https://launchpad.net/bugs/2153425"}],"source_content_type":"text/x-yaml","patch_set":5,"id":"c5c8f5a8_fb0992cc","line":6,"in_reply_to":"f7f0dbbb_9563413e","updated":"2026-06-18 14:21:13.000000000","message":"Thank you very much for the very detailed review and for taking the time to walk through the NUMA claim flow and the related code paths.\n\nI understand your concern about the reproducer relying only on hw:numa_nodes\u003d1 without proper memory constraints. To make the regression test represent a supported and safer NUMA configuration, I will update the test case to include hw:mem_page_size\u003dsmall.\n\nThis should avoid the ambiguity around NUMA memory overcommit behavior and keep the regression test focused on the actual issue being fixed: ensuring NUMA constraints and limits are correctly handled during move operations.\n\nThanks again for the thorough analysis and guidance.","commit_id":"e09757f5e0a2d03170fd3895246ad72b6d8b5c07"}]}
