)]}'
{"/COMMIT_MSG":[{"author":{"_account_id":38081,"name":"Anthony Galica","display_name":"agalica","email":"anthony.galica@hitachivantara.com","username":"agalica","status":"Hitachi Vantara"},"change_message_id":"5eafe5c6bf8c6d15fd8ef5baaac4c1231cb849e1","unresolved":true,"context_lines":[{"line_number":7,"context_line":"Remove disabled services from state_map"},{"line_number":8,"context_line":""},{"line_number":9,"context_line":"It appears that Cinder is still pushing jobs to hosts that an admin has"},{"line_number":10,"context_line":"set a disabled. This may cause some troubles in production."},{"line_number":11,"context_line":""},{"line_number":12,"context_line":"Closes-Bug: https://bugs.launchpad.net/bugs/2158889"},{"line_number":13,"context_line":"Signed-off-by: Thomas Goirand \u003czigo@debian.org\u003e"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":4,"id":"7efbaaa5_db38bf15","line":10,"updated":"2026-08-07 15:06:20.000000000","message":"\"set a disabled\" -\u003e \"set as disabled\"","commit_id":"994ca5f827e6a161a4f2b8f8885c2e860814869a"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"5d44e2e08d6b7a21e7a5a1acd681f66a5df25b32","unresolved":false,"context_lines":[{"line_number":7,"context_line":"Remove disabled services from state_map"},{"line_number":8,"context_line":""},{"line_number":9,"context_line":"It appears that Cinder is still pushing jobs to hosts that an admin has"},{"line_number":10,"context_line":"set a disabled. This may cause some troubles in production."},{"line_number":11,"context_line":""},{"line_number":12,"context_line":"Closes-Bug: https://bugs.launchpad.net/bugs/2158889"},{"line_number":13,"context_line":"Signed-off-by: Thomas Goirand \u003czigo@debian.org\u003e"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":4,"id":"80936c1b_cedf5b10","line":10,"in_reply_to":"7efbaaa5_db38bf15","updated":"2026-08-10 10:21:30.000000000","message":"Done","commit_id":"994ca5f827e6a161a4f2b8f8885c2e860814869a"},{"author":{"_account_id":15197,"name":"Pierre Riteau","email":"pierre@stackhpc.com","username":"priteau","status":"StackHPC"},"change_message_id":"a3601a301be5870e071c37f3d76e4eb7809f6fd0","unresolved":true,"context_lines":[{"line_number":9,"context_line":"It appears that Cinder is still pushing jobs to hosts that an admin has"},{"line_number":10,"context_line":"set as disabled. This may cause some troubles in production."},{"line_number":11,"context_line":""},{"line_number":12,"context_line":"Closes-Bug: https://bugs.launchpad.net/bugs/2158889"},{"line_number":13,"context_line":"Signed-off-by: Thomas Goirand \u003czigo@debian.org\u003e"},{"line_number":14,"context_line":"Change-Id: I7c98819116d0fec282477edff7c8a4eecd9fbe2f"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":5,"id":"45676fd7_16c1b004","line":12,"range":{"start_line":12,"start_character":0,"end_line":12,"end_character":51},"updated":"2026-09-01 14:48:40.000000000","message":"I think you need to use `Closes-Bug: #2158889` for the automatic linking with Launchpad.","commit_id":"4f9fa83dcd7d33cd2359cc7218439be73990eb13"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"fc3520f88b0add9a2de8d92683f99ec1aa21d9fb","unresolved":false,"context_lines":[{"line_number":9,"context_line":"It appears that Cinder is still pushing jobs to hosts that an admin has"},{"line_number":10,"context_line":"set as disabled. This may cause some troubles in production."},{"line_number":11,"context_line":""},{"line_number":12,"context_line":"Closes-Bug: https://bugs.launchpad.net/bugs/2158889"},{"line_number":13,"context_line":"Signed-off-by: Thomas Goirand \u003czigo@debian.org\u003e"},{"line_number":14,"context_line":"Change-Id: I7c98819116d0fec282477edff7c8a4eecd9fbe2f"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":5,"id":"26cad5da_e3d037bd","line":12,"range":{"start_line":12,"start_character":0,"end_line":12,"end_character":51},"in_reply_to":"45676fd7_16c1b004","updated":"2026-09-01 19:15:39.000000000","message":"Done","commit_id":"4f9fa83dcd7d33cd2359cc7218439be73990eb13"}],"/PATCHSET_LEVEL":[{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"980be55354b4199b839f7631324fd666039adc31","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":1,"id":"f8c8d8d3_caff9353","updated":"2026-05-11 09:37:07.000000000","message":"Hi. My patch is probably wrong because of what\u0027s been pointed out: objects.ServiceList.get_all() is supposed to return only active and enabled services. However, we do have a serious bug here, because that\u0027s unfortunately not working. In production, we did see Cinder targeting volume services that were disabled. It\u0027d be nice if this could be investigated by the team.","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":32666,"name":"Damian Dąbrowski","email":"damian@dabrowski.cloud","username":"ddabrowski"},"change_message_id":"f702a8ab00790ea99f3352ccf975b1e1b5384244","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":1,"id":"00eebdd5_6d6f59b9","updated":"2026-05-12 12:56:30.000000000","message":"I tried to reproduce this issue on my test env and I couldn\u0027t.\n\nVolume creation requests were never delegated to the disabled cinder-volume service.\n\nPotentially, some special conditions need to be met in order to reproduce the problem.\n\n\nHowever, I did face an issue with disabling cinder-volume services.\nEven though API call was returning 200, it wasn\u0027t really disabling the service.\nEventually I had to disable the service directly in cinder.services table.","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":10058,"name":"Erlon R. Cruz","email":"erlon.rodrigues.cruz@canonical.com","username":"sombrafam"},"change_message_id":"733e73c421447d2f3e44b33d3ac2862db38b7e03","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":1,"id":"c1d562df_76f314cb","updated":"2026-03-25 12:44:05.000000000","message":"They change looks okay, but I added an improvements suggestion.","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"c4d12f914d3f722ff34b0ea93c9deee4eb3ca9e3","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":1,"id":"86277c7d_22b84052","in_reply_to":"00eebdd5_6d6f59b9","updated":"2026-07-01 09:39:39.000000000","message":"Hi Damian,\n\nIt took me some time to understand what was going on. For this, I vibe coded a quick test:\n\nfrom cinder.common import config\nfrom cinder import context\nfrom cinder import objects\nfrom cinder.db.sqlalchemy import api as db_api\n\nCONF \u003d config.CONF\n                                                                                                                                                                                              \n\ndef test_cinder(ctxt):\n    volume_services \u003d objects.ServiceList.get_all(\n        ctxt,\n        {\n            \"topic\": \"cinder-volume\",\n            \"disabled\": False,\n        },\n    )\n\n    for s in volume_services:\n        print(\n            s.id,\n            s.host,\n            s.binary,\n            s.topic,\n            s.disabled,\n            s.updated_at\n        )\n\ndef main():\n    CONF([], default_config_files\u003d[\"/etc/cinder/cinder.conf\"])\n    db_api.get_engine()\n    objects.register_all()\n    ctxt \u003d context.get_admin_context()\n\n    test_cinder(ctxt)\n\nif __name__ \u003d\u003d \"__main__\":\n    main()\n\nThis shows that objects.ServiceList.get_all() always return hosts, even if they are disabled. But there\u0027s a subtitle thing here: that\u0027s the case ONLY if one sets the cluster\u003d thing AND if there\u0027s other hosts in the cluster that haven\u0027t been disabled. This behavior comes from _clustered_bool_field_filter() that\u0027s doing this.\n\nWhile _clustered_bool_field_filter() cloud be \"fixed\", this would also break the scheduler, so it shouldn\u0027t be done at this level. IMO, my original patch is the correct one, unless we add a new parameter to keep the old behavior by default. Something like:\n\ndef _clustered_bool_field_filter(query, field_name, filter_value, cluster_filer\u003dTrue):\n\nso that callers that do not set cluster_filer would keep the same behavior. This would work, but that would make my original patch a way more complicated.\n\nPlease share your thoughts here, so I can move forward to fix this bug.\n\nAs I believe I\u0027ve discovered a real bug here, I\u0027ll open it on launchpad.","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"8daffe896f65d650e88719fb9309660107316251","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":1,"id":"16f5cd41_e752f8c5","in_reply_to":"86277c7d_22b84052","updated":"2026-07-03 07:51:15.000000000","message":"Done","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"9daab3d3d0c3632383f77bab5df959f03c388f77","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":2,"id":"148c7494_d831fdd3","updated":"2026-07-03 07:21:48.000000000","message":"recheck","commit_id":"b74a4f327329a29bbeac465537f876e67295e18f"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"4a1d8cb30c75c439a981a2434fa34f746f93d739","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":2,"id":"3e6a9a28_60a08444","updated":"2026-07-02 12:36:19.000000000","message":"recheck","commit_id":"b74a4f327329a29bbeac465537f876e67295e18f"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"b8b8dba0fd03f7da2e82f82505bc86ed6729c07b","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":2,"id":"6c5a9e8e_4df4ebc0","updated":"2026-07-02 07:05:21.000000000","message":"recheck","commit_id":"b74a4f327329a29bbeac465537f876e67295e18f"},{"author":{"_account_id":11904,"name":"Sean McGinnis","email":"sean.mcginnis@gmail.com","username":"SeanM"},"change_message_id":"359f5e8da4760311feb5fab33587cb482caffa6c","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":3,"id":"038f596e_931d26e7","updated":"2026-08-05 14:37:43.000000000","message":"Minor comment inline, but makes sense.","commit_id":"c5f7b0127ea674327cdc1d5bf1906a72c3b3c3e8"},{"author":{"_account_id":5314,"name":"Brian Rosmaita","email":"rosmaita.fossdev@gmail.com","username":"brian-rosmaita"},"change_message_id":"5f96d8434eae6f6d49527109f62cc329fd77e6e8","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":3,"id":"75d382f3_a8440eba","updated":"2026-08-05 14:34:39.000000000","message":"Raising priority, this is an operator pain point we should address.","commit_id":"c5f7b0127ea674327cdc1d5bf1906a72c3b3c3e8"},{"author":{"_account_id":36171,"name":"jayaanand borra","display_name":"jayaanand borra","email":"jayaanand.borra@netapp.com","username":"jayaanan","status":"netapp"},"change_message_id":"f502d70fa14f87cf38cc6ffee0dbc3e64ffaaba4","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":3,"id":"4dd4dbc8_246f4639","updated":"2026-07-15 11:00:06.000000000","message":"Thank you! Brian...i missed this point, stale entries are removed at lines 735-744","commit_id":"c5f7b0127ea674327cdc1d5bf1906a72c3b3c3e8"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"aec1586564588f450db2bf64bebce66575a547aa","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":3,"id":"1f9095dd_ad7f7f5d","updated":"2026-07-04 20:52:26.000000000","message":"recheck","commit_id":"c5f7b0127ea674327cdc1d5bf1906a72c3b3c3e8"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"c774be1481fefc3ac0bde1458f0a8fa4532c8397","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":3,"id":"7dd27967_abbee69d","updated":"2026-07-03 10:32:39.000000000","message":"recheck","commit_id":"c5f7b0127ea674327cdc1d5bf1906a72c3b3c3e8"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"fd57d4ee68e4100f244033658cc2da9a6d7a72d7","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"5c182ccc_992a4126","updated":"2026-08-06 07:40:32.000000000","message":"Hi Sean, I agreed, so I\u0027ve lower the log level to debug. Please +2 again this patch!","commit_id":"994ca5f827e6a161a4f2b8f8885c2e860814869a"},{"author":{"_account_id":38081,"name":"Anthony Galica","display_name":"agalica","email":"anthony.galica@hitachivantara.com","username":"agalica","status":"Hitachi Vantara"},"change_message_id":"5eafe5c6bf8c6d15fd8ef5baaac4c1231cb849e1","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"73bfb35e_b2acc0b0","updated":"2026-08-07 15:06:20.000000000","message":"This looks like a good patch to me, but I don\u0027t feel fully qualified to workflow it at this time.\n\nI have one nitpick comment about the commit message, but probably shouldn\u0027t hold up a workflow.","commit_id":"994ca5f827e6a161a4f2b8f8885c2e860814869a"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"1947f9554e977d271b7650bb1da8f04fd51d04da","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"1d879656_ee249e1a","updated":"2026-08-06 13:33:11.000000000","message":"recheck","commit_id":"994ca5f827e6a161a4f2b8f8885c2e860814869a"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"5d44e2e08d6b7a21e7a5a1acd681f66a5df25b32","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"d35e584f_f0459f46","in_reply_to":"39b3c33a_d2cebe1c","updated":"2026-08-10 10:21:30.000000000","message":"Done","commit_id":"994ca5f827e6a161a4f2b8f8885c2e860814869a"},{"author":{"_account_id":38081,"name":"Anthony Galica","display_name":"agalica","email":"anthony.galica@hitachivantara.com","username":"agalica","status":"Hitachi Vantara"},"change_message_id":"acd67c6637bc63f0ed56e3f4423a21d82da45132","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":4,"id":"39b3c33a_d2cebe1c","in_reply_to":"73bfb35e_b2acc0b0","updated":"2026-08-07 15:07:44.000000000","message":"Also appears to be some Zuul failures (not sure if transient or not).","commit_id":"994ca5f827e6a161a4f2b8f8885c2e860814869a"},{"author":{"_account_id":10058,"name":"Erlon R. Cruz","email":"erlon.rodrigues.cruz@canonical.com","username":"sombrafam"},"change_message_id":"877e90ada7cd3d1e849902511c8a198f42e4de76","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":5,"id":"aaafbf93_d06857ff","updated":"2026-08-11 18:09:09.000000000","message":"I have some concerns with this patch that I still have to dig deeper, so I\u0027m setting it to -1 to make sure I get to the bottom of it and it doesn\u0027t get merged, until I complete my investigation. \n\nMy preliminary findings is that, the disabled volume service continues to \u0027receive\u0027 volume operations because it reads events from an rpc topic, which is the same for all backends in a cluster. Even if you set the service as disabled, if the remote service continues to consume events from from that queue, so skipping this update might not be enough to fix the problem.","commit_id":"4f9fa83dcd7d33cd2359cc7218439be73990eb13"},{"author":{"_account_id":10058,"name":"Erlon R. Cruz","email":"erlon.rodrigues.cruz@canonical.com","username":"sombrafam"},"change_message_id":"c3d76ce5ad5345f442fb9f667c601bc29e28cd6e","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":5,"id":"d63cbdb3_1bb8d86e","in_reply_to":"3c949ae0_8a31f1f3","updated":"2026-09-16 14:30:26.000000000","message":"My proposed solution for an admin to service the host, is not to stop the cluster, is just to stop one of the hosts of the cluster. In pratical terms, that will have the same effect as a stopping the service from listening to the RPC queue.\n\nIf that\u0027s not the case, can you explain why it wouldn\u0027t have the same end result?","commit_id":"4f9fa83dcd7d33cd2359cc7218439be73990eb13"},{"author":{"_account_id":27615,"name":"Rajat Dhasmana","email":"rajatdhasmana@gmail.com","username":"whoami-rajat"},"change_message_id":"c4fbd9e9932921a28e1711d1ea4a291dde8cc8e9","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":5,"id":"3c949ae0_8a31f1f3","in_reply_to":"3d904a63_e8bb109d","updated":"2026-09-12 18:32:29.000000000","message":"I spend quite a bit of time on this and have the following points:\n1. I looked into the idea of \"rejecting disable service request\" and I feel it\u0027s too restrictive. If an admin wants to fix a malfunctioning service, having them disable the cluster and fix it kind of defeats the purpose of high availability\n2. Filtering out the service in scheduler (current solution) can be a \"patching\" mechanism but doesn\u0027t solve the core issue where the disabled service still listens to the cluster queue\n3. Eventually i landed on the solution to stop the host consumer from accepting messages from the queue[1]\nOn every disable requests, it stops the rabbitmq consumer (via the RPC server object) so that the disabled host doesn\u0027t pick any requests accidently in any scenario and after enabling the host, rpc server starts the rabbitmq consumer again.\nHaving limited time and resources, I couldn\u0027t deploy an A/A deployment and test it so would appreciate if anyone can help with that.\n\n[1] https://review.opendev.org/c/openstack/cinder/+/1005206","commit_id":"4f9fa83dcd7d33cd2359cc7218439be73990eb13"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"866991104dadc6c76f5fdb35dd8ed7a30413ced4","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":5,"id":"ab3d2855_66baa433","in_reply_to":"aaafbf93_d06857ff","updated":"2026-08-12 08:24:09.000000000","message":"Hi,\n\nThat\u0027s not how it works. Every Cinder service listen to a topic that matches its hostname by default, so it really is up to the control plane to target a specific service. There\u0027s no such thing as a common topic that all services would listen to, and then a random service would pick it if it is alive.\n\nSo what you are describing is *not* happening, and my patch is still valid. Please remove your -1 and let\u0027s fix the situation.","commit_id":"4f9fa83dcd7d33cd2359cc7218439be73990eb13"},{"author":{"_account_id":10058,"name":"Erlon R. Cruz","email":"erlon.rodrigues.cruz@canonical.com","username":"sombrafam"},"change_message_id":"0f3df3d95d3af5e5e104b221cd24cf50560e3f21","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":5,"id":"3d904a63_e8bb109d","in_reply_to":"ab3d2855_66baa433","updated":"2026-09-09 20:44:11.000000000","message":"Hey Thomas,\n\nLet me correct something I said in a mislead way. RPC topics are not the same for all backends in the cluster, but the hosts that belongs to a cluster share rpc queues with other members in the cluster[1]. For example:\n\nCluster 1\n    - Host a: backend a1\n    - Host b: backend a1\n\nBoth hosts, a and b, receive the jobs using the same RPC topic (something like, cluster@backenda1). That is like that by design to guarantee high availability. If one of the hosts are down, the other will continue to receive the requests and continue processing them regardless of the other. That\u0027s why, you notice the behavior \"instead of just filtering out volume nodes that are disabled, it keeps those who are in a cluster that still contains enabled host.\"\n\nThe thing is, removing the host from the state map doesn\u0027t solve this in the general case. In A/A the scheduler doesn\u0027t address individual hosts at all - backend_state_map is keyed by cluster_name or host, and the RPC cast goes to the cluster topic - so for a cluster with other enabled members, work still lands on the shared queue and the disabled service will pick it up. Your patch does help when every member of the cluster is disabled (or it\u0027s a single-host cluster), but it can\u0027t express \"disable just this host\", because that isn\u0027t a thing the scheduler can target.\n\nToday the supported ways to stop work reaching a clustered backend are to disable the whole cluster (os-clusters disable, API 3.7+) or to stop the service on the host and let its peers take over. What I think is the real defect here is that os-services/disable happily accepts a clustered service and sets a flag that has no effect — that\u0027s what made this look like a scheduler bug.\n\nWould you be up for a patch that rejects (or at least warns on) disabling a clustered service and points at the cluster API instead? I\u0027d support that, and it addresses the operator surprise you originally hit.\n_________________________\n[1] https://docs.openstack.org/cinder/latest/contributor/high_availability.html#rpc-calls","commit_id":"4f9fa83dcd7d33cd2359cc7218439be73990eb13"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"bc074b2040965934942626632e779d06c391d7ad","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":6,"id":"d99d1b7a_604d8f32","updated":"2026-09-18 15:18:02.000000000","message":"I now understand why my patch is wrong. In cluster\u003d\u003csomething\u003e mode, there\u0027s only a single rabbitmq topic, and the scheduler isn\u0027t really involved. Rajat\u0027s patch over here should do a much better job at it:\n\nhttps://review.opendev.org/c/openstack/cinder/+/1005206\n\nthe only issue with it, is that it seems based on the report_state() periodic method, which can take up to the configured amount of time (10 seconds by default) to be processed. I guess that\u0027s ok on a first approach. I\u0027d prefer if hosts were listening to 2 topics though, but maybe this can be done as a 2nd approach.","commit_id":"12ed5077e54116bbcfee3c4c8543f1931db328da"},{"author":{"_account_id":34860,"name":"Amit Uniyal","email":"auniyal@redhat.com","username":"auniyal"},"change_message_id":"bb99a450994623226109194779b30a656f9d7991","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":6,"id":"3d2a13ec_dad31a98","updated":"2026-09-07 05:29:10.000000000","message":"could you add a reproducer showing actual impact - like volume being scheduled and created to disabled host - before and after this patch (so  in separate patch if possible) ?\n\nwhat the service/cluster setup was, how the host was disabled, and what showed\nvolumes still landing on it. right now it\u0027s hard to tell whether this\nis the right fix.\n\n---\n\n@thomas@goirand.fr - small process request - could you please leave the comments unresolved, until we agreed on something and settled them ?\nI find it easier to pickup a review when then open threads are still visible, rather than expanding reslved onces to check what was agreed","commit_id":"12ed5077e54116bbcfee3c4c8543f1931db328da"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"67b047c268020d8213cad5eed08a50a2d25b46cb","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":6,"id":"3a0e43f1_a1ff64ae","updated":"2026-09-02 13:04:31.000000000","message":"recheck","commit_id":"12ed5077e54116bbcfee3c4c8543f1931db328da"}],"cinder/scheduler/host_manager.py":[{"author":{"_account_id":10058,"name":"Erlon R. Cruz","email":"erlon.rodrigues.cruz@canonical.com","username":"sombrafam"},"change_message_id":"733e73c421447d2f3e44b33d3ac2862db38b7e03","unresolved":true,"context_lines":[{"line_number":692,"context_line":"        no_capabilities_backends \u003d set()"},{"line_number":693,"context_line":"        for service in volume_services.objects:"},{"line_number":694,"context_line":"            host \u003d service.host"},{"line_number":695,"context_line":"            if not service.is_up:"},{"line_number":696,"context_line":"                LOG.warning(\"volume service is down. (host: %s)\", host)"},{"line_number":697,"context_line":"                continue"},{"line_number":698,"context_line":""}],"source_content_type":"text/x-python","patch_set":1,"id":"3303b5d6_e8ba3a3f","line":695,"updated":"2026-03-25 12:44:05.000000000","message":"Can you merge your fix to this if?\n```\n  if not service.is_up or service.disabled:\n      LOG.warning(\"volume service down or disabled. (host: %s)\", host)\n```","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"5e96792b1635bb4e496074e786bc227ec8e242f8","unresolved":false,"context_lines":[{"line_number":692,"context_line":"        no_capabilities_backends \u003d set()"},{"line_number":693,"context_line":"        for service in volume_services.objects:"},{"line_number":694,"context_line":"            host \u003d service.host"},{"line_number":695,"context_line":"            if not service.is_up:"},{"line_number":696,"context_line":"                LOG.warning(\"volume service is down. (host: %s)\", host)"},{"line_number":697,"context_line":"                continue"},{"line_number":698,"context_line":""}],"source_content_type":"text/x-python","patch_set":1,"id":"2b619e7b_e26ba44c","line":695,"in_reply_to":"3303b5d6_e8ba3a3f","updated":"2026-03-27 08:09:34.000000000","message":"No, I prefer the way I did, so we can log differently if the service is down or disabled.","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":4523,"name":"Eric Harney","email":"eharney@redhat.com","username":"eharney"},"change_message_id":"62206690e8e1faccfb7d7367946a726334055fe3","unresolved":true,"context_lines":[{"line_number":696,"context_line":"                LOG.warning(\"volume service is down. (host: %s)\", host)"},{"line_number":697,"context_line":"                continue"},{"line_number":698,"context_line":""},{"line_number":699,"context_line":"            if service.disabled:"},{"line_number":700,"context_line":"                LOG.warning(\"volume service is disabled. (host: %s)\", host)"},{"line_number":701,"context_line":"                continue"},{"line_number":702,"context_line":""}],"source_content_type":"text/x-python","patch_set":1,"id":"7af2be6e_d0b0be44","line":699,"updated":"2026-04-21 14:44:47.000000000","message":"Didn\u0027t we call ServiceList.get_all with disabled: False above? Will this ever be true?","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":36171,"name":"jayaanand borra","display_name":"jayaanand borra","email":"jayaanand.borra@netapp.com","username":"jayaanan","status":"netapp"},"change_message_id":"0c8e65e6779060164f5e1de8764cbd696f4cae21","unresolved":true,"context_lines":[{"line_number":696,"context_line":"                LOG.warning(\"volume service is down. (host: %s)\", host)"},{"line_number":697,"context_line":"                continue"},{"line_number":698,"context_line":""},{"line_number":699,"context_line":"            if service.disabled:"},{"line_number":700,"context_line":"                LOG.warning(\"volume service is disabled. (host: %s)\", host)"},{"line_number":701,"context_line":"                continue"},{"line_number":702,"context_line":""}],"source_content_type":"text/x-python","patch_set":1,"id":"c8d82915_4df35ba1","line":699,"in_reply_to":"7af2be6e_d0b0be44","updated":"2026-04-22 05:38:52.000000000","message":"small window of possibility... ServiceList.get_all(disabled\u003dFalse) filters correctly at SQL level for this cycle and backend_state_map is a persistent in-memory dict that accumulates entries across in scheduler cycles.\nA service disabled between cycles will no longer appear in SQL results, but its stale entry remains in state_map from the previous cycle and the scheduler will keep sending jobs to it.","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":5314,"name":"Brian Rosmaita","email":"rosmaita.fossdev@gmail.com","username":"brian-rosmaita"},"change_message_id":"ca58f5d43103db7626819549fbecc3f4231bfd4b","unresolved":true,"context_lines":[{"line_number":696,"context_line":"                LOG.warning(\"volume service is down. (host: %s)\", host)"},{"line_number":697,"context_line":"                continue"},{"line_number":698,"context_line":""},{"line_number":699,"context_line":"            if service.disabled:"},{"line_number":700,"context_line":"                LOG.warning(\"volume service is disabled. (host: %s)\", host)"},{"line_number":701,"context_line":"                continue"},{"line_number":702,"context_line":""}],"source_content_type":"text/x-python","patch_set":1,"id":"e96bb456_71ee6e8c","line":699,"in_reply_to":"c8d82915_4df35ba1","updated":"2026-04-27 18:53:15.000000000","message":"I think the stale entries from the previous cycle are removed after this loop is completed, namely at lines 735-744 ... or am i misunderstanding your point?","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"ac850c6394aead966f5d8ec44c62184ca803f9ab","unresolved":false,"context_lines":[{"line_number":696,"context_line":"                LOG.warning(\"volume service is down. (host: %s)\", host)"},{"line_number":697,"context_line":"                continue"},{"line_number":698,"context_line":""},{"line_number":699,"context_line":"            if service.disabled:"},{"line_number":700,"context_line":"                LOG.warning(\"volume service is disabled. (host: %s)\", host)"},{"line_number":701,"context_line":"                continue"},{"line_number":702,"context_line":""}],"source_content_type":"text/x-python","patch_set":1,"id":"4eb977b2_071bb0bb","line":699,"in_reply_to":"e96bb456_71ee6e8c","updated":"2026-07-01 15:16:30.000000000","message":"As explained in the bug report, this is very miss-leading. If I understand correctly (please do double-check for me: I\u0027m not an expert in SQLAlchemy), if one carefully looks into how objects.ServiceList.get_all() works, it internally calls the filter _clustered_bool_field_filter(). While this does work for a single hostname with no cluster, if cluster\u003d is set in cinder.conf, and there are other hosts in the clusters that haven\u0027t been disabled, then the host isn\u0027t filtered. See _clustered_bool_field_filter() in cinder/db/sqlalchemy/api.py (called by _service_query()).","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":11904,"name":"Sean McGinnis","email":"sean.mcginnis@gmail.com","username":"SeanM"},"change_message_id":"359f5e8da4760311feb5fab33587cb482caffa6c","unresolved":true,"context_lines":[{"line_number":702,"context_line":"            # and up. We must then filter disabled services here so the"},{"line_number":703,"context_line":"            # scheduler never considers them as candidates."},{"line_number":704,"context_line":"            if service.disabled:"},{"line_number":705,"context_line":"                LOG.warning(\"volume service is disabled. (host: %s)\", host)"},{"line_number":706,"context_line":"                continue"},{"line_number":707,"context_line":""},{"line_number":708,"context_line":"            backend_key \u003d service.service_topic_queue"}],"source_content_type":"text/x-python","patch_set":3,"id":"40000552_6300101d","line":705,"updated":"2026-08-05 14:37:43.000000000","message":"Warning may be too high of a level if this is a \"normal\" expected state for this host.","commit_id":"c5f7b0127ea674327cdc1d5bf1906a72c3b3c3e8"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"fd57d4ee68e4100f244033658cc2da9a6d7a72d7","unresolved":false,"context_lines":[{"line_number":702,"context_line":"            # and up. We must then filter disabled services here so the"},{"line_number":703,"context_line":"            # scheduler never considers them as candidates."},{"line_number":704,"context_line":"            if service.disabled:"},{"line_number":705,"context_line":"                LOG.warning(\"volume service is disabled. (host: %s)\", host)"},{"line_number":706,"context_line":"                continue"},{"line_number":707,"context_line":""},{"line_number":708,"context_line":"            backend_key \u003d service.service_topic_queue"}],"source_content_type":"text/x-python","patch_set":3,"id":"f31dcd60_f849fa1e","line":705,"in_reply_to":"40000552_6300101d","updated":"2026-08-06 07:40:32.000000000","message":"Done","commit_id":"c5f7b0127ea674327cdc1d5bf1906a72c3b3c3e8"},{"author":{"_account_id":27615,"name":"Rajat Dhasmana","email":"rajatdhasmana@gmail.com","username":"whoami-rajat"},"change_message_id":"c4fbd9e9932921a28e1711d1ea4a291dde8cc8e9","unresolved":true,"context_lines":[{"line_number":696,"context_line":"                LOG.warning(\"volume service is down. (host: %s)\", host)"},{"line_number":697,"context_line":"                continue"},{"line_number":698,"context_line":""},{"line_number":699,"context_line":"            # Like service.is_up, ServiceList.get_all() can be misleading."},{"line_number":700,"context_line":"            # When multiple services belong to the same cluster, it may return"},{"line_number":701,"context_line":"            # disabled services if another service in the cluster is enabled"},{"line_number":702,"context_line":"            # and up. We must then filter disabled services here so the"}],"source_content_type":"text/x-python","patch_set":6,"id":"85ce5654_0231cbd3","line":699,"range":{"start_line":699,"start_character":12,"end_line":699,"end_character":74},"updated":"2026-09-12 18:32:29.000000000","message":"service.is_up is not a database field but calculated dynamically from the updated_at field so applying same reasoning for is_up and disabled is not the right comparison.","commit_id":"12ed5077e54116bbcfee3c4c8543f1931db328da"},{"author":{"_account_id":27615,"name":"Rajat Dhasmana","email":"rajatdhasmana@gmail.com","username":"whoami-rajat"},"change_message_id":"c4fbd9e9932921a28e1711d1ea4a291dde8cc8e9","unresolved":true,"context_lines":[{"line_number":697,"context_line":"                continue"},{"line_number":698,"context_line":""},{"line_number":699,"context_line":"            # Like service.is_up, ServiceList.get_all() can be misleading."},{"line_number":700,"context_line":"            # When multiple services belong to the same cluster, it may return"},{"line_number":701,"context_line":"            # disabled services if another service in the cluster is enabled"},{"line_number":702,"context_line":"            # and up. We must then filter disabled services here so the"},{"line_number":703,"context_line":"            # scheduler never considers them as candidates."},{"line_number":704,"context_line":"            if service.disabled:"},{"line_number":705,"context_line":"                LOG.debug(\"volume service is disabled. (host: %s)\", host)"},{"line_number":706,"context_line":"                continue"}],"source_content_type":"text/x-python","patch_set":6,"id":"ce778c70_838524ec","line":703,"range":{"start_line":700,"start_character":12,"end_line":703,"end_character":59},"updated":"2026-09-12 18:32:29.000000000","message":"This fix is good enough to \"patch\" the situation but I think we\u0027ve a bigger problem if this is happening. A service can be disabled and still be listening to the cluster queue which is the core issue here.","commit_id":"12ed5077e54116bbcfee3c4c8543f1931db328da"},{"author":{"_account_id":10058,"name":"Erlon R. Cruz","email":"erlon.rodrigues.cruz@canonical.com","username":"sombrafam"},"change_message_id":"c3d76ce5ad5345f442fb9f667c601bc29e28cd6e","unresolved":true,"context_lines":[{"line_number":697,"context_line":"                continue"},{"line_number":698,"context_line":""},{"line_number":699,"context_line":"            # Like service.is_up, ServiceList.get_all() can be misleading."},{"line_number":700,"context_line":"            # When multiple services belong to the same cluster, it may return"},{"line_number":701,"context_line":"            # disabled services if another service in the cluster is enabled"},{"line_number":702,"context_line":"            # and up. We must then filter disabled services here so the"},{"line_number":703,"context_line":"            # scheduler never considers them as candidates."},{"line_number":704,"context_line":"            if service.disabled:"},{"line_number":705,"context_line":"                LOG.debug(\"volume service is disabled. (host: %s)\", host)"},{"line_number":706,"context_line":"                continue"}],"source_content_type":"text/x-python","patch_set":6,"id":"f41b5f4e_a498b6bb","line":703,"range":{"start_line":700,"start_character":12,"end_line":703,"end_character":59},"in_reply_to":"ce778c70_838524ec","updated":"2026-09-16 14:30:26.000000000","message":"So, this fix the way it is would not solve the problem. What would solve the issue, would be one of the things we are proposing: Stop one of the host of the cluster, or as in your patch, blocks it from consume from the RPC queue.","commit_id":"12ed5077e54116bbcfee3c4c8543f1931db328da"},{"author":{"_account_id":34860,"name":"Amit Uniyal","email":"auniyal@redhat.com","username":"auniyal"},"change_message_id":"bb99a450994623226109194779b30a656f9d7991","unresolved":true,"context_lines":[{"line_number":703,"context_line":"            # scheduler never considers them as candidates."},{"line_number":704,"context_line":"            if service.disabled:"},{"line_number":705,"context_line":"                LOG.debug(\"volume service is disabled. (host: %s)\", host)"},{"line_number":706,"context_line":"                continue"},{"line_number":707,"context_line":""},{"line_number":708,"context_line":"            backend_key \u003d service.service_topic_queue"},{"line_number":709,"context_line":"            # We only pay attention to the first up service of a cluster since"}],"source_content_type":"text/x-python","patch_set":6,"id":"25e0f9ad_9da5fcc1","line":706,"updated":"2026-09-07 05:29:10.000000000","message":"this fix seems fine in general, though I also wonder wouldn\u0027t DB query already doing this - filtering disabled service!!\nand its kind of unclear if this addresses the bug. added tests doesn\u0027t demostrate/clear it.","commit_id":"12ed5077e54116bbcfee3c4c8543f1931db328da"}],"cinder/tests/unit/scheduler/test_host_manager.py":[{"author":{"_account_id":4523,"name":"Eric Harney","email":"eharney@redhat.com","username":"eharney"},"change_message_id":"5306e2e4672261c87f5b7738a2e5ae734adbca60","unresolved":true,"context_lines":[{"line_number":755,"context_line":"    @mock.patch(\u0027cinder.db.service_get_all\u0027)"},{"line_number":756,"context_line":"    @mock.patch(\u0027cinder.objects.service.Service.is_up\u0027,"},{"line_number":757,"context_line":"                new_callable\u003dmock.PropertyMock)"},{"line_number":758,"context_line":"    def test_get_all_backend_states(self, _mock_service_is_up,"},{"line_number":759,"context_line":"                                    _mock_service_get_all):"},{"line_number":760,"context_line":"        context \u003d \u0027fake_context\u0027"},{"line_number":761,"context_line":"        timestamp \u003d datetime.utcnow()"}],"source_content_type":"text/x-python","patch_set":1,"id":"20172704_67dfb078","line":758,"updated":"2026-04-21 15:04:10.000000000","message":"Maybe this test can be modified to include a disabled host.","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":6476,"name":"Thomas Goirand","email":"thomas@goirand.fr","username":"thomas-goirand"},"change_message_id":"8daffe896f65d650e88719fb9309660107316251","unresolved":false,"context_lines":[{"line_number":755,"context_line":"    @mock.patch(\u0027cinder.db.service_get_all\u0027)"},{"line_number":756,"context_line":"    @mock.patch(\u0027cinder.objects.service.Service.is_up\u0027,"},{"line_number":757,"context_line":"                new_callable\u003dmock.PropertyMock)"},{"line_number":758,"context_line":"    def test_get_all_backend_states(self, _mock_service_is_up,"},{"line_number":759,"context_line":"                                    _mock_service_get_all):"},{"line_number":760,"context_line":"        context \u003d \u0027fake_context\u0027"},{"line_number":761,"context_line":"        timestamp \u003d datetime.utcnow()"}],"source_content_type":"text/x-python","patch_set":1,"id":"e9a4e754_d078e3d6","line":758,"in_reply_to":"20172704_67dfb078","updated":"2026-07-03 07:51:15.000000000","message":"Very good idea, thanks!","commit_id":"86ff5183a5d219e12eaefe86861f53e81d147873"},{"author":{"_account_id":34860,"name":"Amit Uniyal","email":"auniyal@redhat.com","username":"auniyal"},"change_message_id":"bb99a450994623226109194779b30a656f9d7991","unresolved":true,"context_lines":[{"line_number":752,"context_line":"                    (\u0027non_clustered_host#_pool0\u0027, 4000)}"},{"line_number":753,"context_line":"        self.assertSetEqual(expected, result)"},{"line_number":754,"context_line":""},{"line_number":755,"context_line":"    @mock.patch(\u0027cinder.db.service_get_all\u0027)"},{"line_number":756,"context_line":"    @mock.patch(\u0027cinder.objects.service.Service.is_up\u0027,"},{"line_number":757,"context_line":"                new_callable\u003dmock.PropertyMock)"},{"line_number":758,"context_line":"    def test_get_all_backend_states(self, _mock_service_is_up,"}],"source_content_type":"text/x-python","patch_set":6,"id":"d04d1980_b8173f7c","line":755,"updated":"2026-09-07 05:29:10.000000000","message":"so all db query are already mocked here, so real filtering never runs.","commit_id":"12ed5077e54116bbcfee3c4c8543f1931db328da"},{"author":{"_account_id":34860,"name":"Amit Uniyal","email":"auniyal@redhat.com","username":"auniyal"},"change_message_id":"bb99a450994623226109194779b30a656f9d7991","unresolved":true,"context_lines":[{"line_number":815,"context_line":"                          provisioned_capacity_gb\u003d212),"},{"line_number":816,"context_line":"        }"},{"line_number":817,"context_line":"        # First test: service.is_up is always True, host5 is disabled,"},{"line_number":818,"context_line":"        # host4 has no capabilities and host5 is disabled."},{"line_number":819,"context_line":"        self.host_manager.service_states \u003d service_states"},{"line_number":820,"context_line":"        _mock_service_get_all.return_value \u003d services"},{"line_number":821,"context_line":"        _mock_service_is_up.return_value \u003d True"}],"source_content_type":"text/x-python","patch_set":6,"id":"7746f9b4_6b2aee7f","line":818,"range":{"start_line":818,"start_character":36,"end_line":818,"end_character":58},"updated":"2026-09-07 05:29:10.000000000","message":"nit: its said already in above comment.","commit_id":"12ed5077e54116bbcfee3c4c8543f1931db328da"}]}
