)]}'
{"/COMMIT_MSG":[{"author":{"_account_id":1131,"name":"Brian Haley","email":"haleyb.dev@gmail.com","username":"brian-haley"},"change_message_id":"b8efec2cb04b28706915a36ec1bc9195f17adbbe","unresolved":true,"context_lines":[{"line_number":4,"context_line":"Commit:     Marcin Wilk \u003cmarcin.wilk@canonical.com\u003e"},{"line_number":5,"context_line":"CommitDate: 2026-07-09 17:01:23 +0200"},{"line_number":6,"context_line":""},{"line_number":7,"context_line":"Denote highest priority chassis after crash and BFD failover"},{"line_number":8,"context_line":""},{"line_number":9,"context_line":"When the highest priority chassis crashes, BFD detects the problem"},{"line_number":10,"context_line":"quickly and fails over active gateway ports to the second highest"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":4,"id":"031abec0_0b24e01b","line":7,"range":{"start_line":7,"start_character":0,"end_line":7,"end_character":6},"updated":"2026-07-16 16:17:31.000000000","message":"s/Demote","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"d5554107803378da334a5cc83364e09061d2be3b","unresolved":true,"context_lines":[{"line_number":4,"context_line":"Commit:     Marcin Wilk \u003cmarcin.wilk@canonical.com\u003e"},{"line_number":5,"context_line":"CommitDate: 2026-07-09 17:01:23 +0200"},{"line_number":6,"context_line":""},{"line_number":7,"context_line":"Denote highest priority chassis after crash and BFD failover"},{"line_number":8,"context_line":""},{"line_number":9,"context_line":"When the highest priority chassis crashes, BFD detects the problem"},{"line_number":10,"context_line":"quickly and fails over active gateway ports to the second highest"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":4,"id":"0cad1715_6706372c","line":7,"range":{"start_line":7,"start_character":0,"end_line":7,"end_character":6},"in_reply_to":"031abec0_0b24e01b","updated":"2026-07-17 06:53:39.000000000","message":"Acknowledged","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"}],"/PATCHSET_LEVEL":[{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"a31d4be7a2ae7ec274bd435c5a38fe351f45cf69","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":1,"id":"548ae2d3_62f9cc6e","updated":"2026-05-25 18:04:48.000000000","message":"Hello, I\u0027d appreciate it if you could share your thoughts/comments about this patch.","commit_id":"58ba0e5f5700037826d455152d656dd795b5421e"},{"author":{"_account_id":16688,"name":"Rodolfo Alonso","email":"ralonsoh@redhat.com","username":"rodolfo-alonso-hernandez"},"change_message_id":"4ac0e25c1301082bfee119edc5c922881d74b820","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":1,"id":"a6ddf899_a04e9ec7","in_reply_to":"548ae2d3_62f9cc6e","updated":"2026-06-04 09:55:01.000000000","message":"I\u0027ve asked this question privately in my company team. We\u0027ll discuss about it. In any case, it could be nice to propose this change in the Neutron Drivers meeting (https://meetings.opendev.org/#Neutron_drivers_Meeting).\n\nI\u0027ll add a topic for this friday","commit_id":"58ba0e5f5700037826d455152d656dd795b5421e"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"07d8c90b27732f35b65cbeab87070325257b538b","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":1,"id":"bbb88c29_a7947c7d","in_reply_to":"a6ddf899_a04e9ec7","updated":"2026-06-04 10:18:03.000000000","message":"Thank you. I will attend the meeting tomorrow.","commit_id":"58ba0e5f5700037826d455152d656dd795b5421e"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"1775bb4e101e0fef9976774870ec4de4f0c36d10","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":2,"id":"a6a4e755_8beb77d7","updated":"2026-07-07 05:53:21.000000000","message":"recheck neutron-tempest-plugin-ovn timed out","commit_id":"1cce594de49233e57bcc148438734e1324bbf90a"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"049562548deb5fa5812f2d086ff1c3d9cb078845","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":3,"id":"fd540b34_79f705fb","updated":"2026-07-09 05:32:16.000000000","message":"recheck after the neutron-tempest-plugin-ovn bug lp2069718 has been fixed","commit_id":"4d63bc29feae024af919e54622f67f98a7ae8449"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"412bdaa3a2039e5e2d84dc91d688b467062dab7d","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"c27da78a_e508eafe","updated":"2026-07-10 09:15:22.000000000","message":"@ralonsoh@redhat.com @haleyb.dev@gmail.com\nI updated the patch and added a new config knob to make this an optional functionality, as it was discussed on the neutron drivers weekly the other day.\nI tested the changes on master branch (the new HC/HCG path) on devstack and the legacy path (gateway_chassis) on charmed Flamingo.\nPlease have a look and share your comments/suggestions. Thank you.","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"cad6cb445933125b27a59bfd19a1cc4a2eec4140","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"dd338f2b_72720ba5","updated":"2026-09-30 15:50:46.000000000","message":"@twilson@redhat.com, could you please take a look at Rodolfo\u0027s most recent comment? I would like to clarify the path forward with this patch. Thank you","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"ade6f9116ca16f3220ed43942e3ec1472abc8e37","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"a61f5bd2_6848afac","updated":"2026-08-11 13:13:31.000000000","message":"@twilson@redhat.com, pls check my response.","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":1131,"name":"Brian Haley","email":"haleyb.dev@gmail.com","username":"brian-haley"},"change_message_id":"b8efec2cb04b28706915a36ec1bc9195f17adbbe","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"ec5476e4_28c35538","updated":"2026-07-16 16:17:31.000000000","message":"Looks good, just nits, just want to get other\u0027s opinions.","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":5756,"name":"Terry Wilson","email":"twilson@redhat.com","username":"otherwiseguy"},"change_message_id":"7b17d11bb42acb43c8d10e50a93bdf2853f8c08a","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"c802c236_1be05a45","updated":"2026-08-06 15:39:03.000000000","message":"One question I have is that the exam","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"f46eb626321b0a4dde536f1ffc9c6bb7e14041ea","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":4,"id":"275a55d7_e92ce24a","updated":"2026-07-10 06:48:30.000000000","message":"recheck random live-migration timeout","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":5756,"name":"Terry Wilson","email":"twilson@redhat.com","username":"otherwiseguy"},"change_message_id":"983441cd7598c58a0136b142ea00b3dc842e7057","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":4,"id":"7ca384cb_39c774dc","in_reply_to":"dd338f2b_72720ba5","updated":"2026-10-02 22:49:16.000000000","message":"To me this still feels like trying to solve the problem at the wrong layer. I just don\u0027t trust PB.chassis as the signal that there has been a failover that we should react to. Also, due to the very asynchronous nature of OVS/OVN/OVSDB, I think we could run into problems like \"we can\u0027t be sure we can update the priorities in the db before the chassis comes back up\". So in that case, we\u0027d actually cause more churn by demoting the Chassis after it came back, which I think might then read as another failover (I haven\u0027t traced that, just seems possible)\n\nOn the \u0027alive\u0027 checking, the fact that agent_down_time\u003d75 by default, and often on large clusters might be set to (insanely) high values, due to just how poorly using NB_Global.nb_cfg is, especially with ovn-monitor-all\u003dTrue, makes me think its something we can really reliably use--especially for the worst-case scenario: a highest-priority chassis that gets marked down/up very frequently and causes a lot of bouncing.\n\nI also get that there isn\u0027t an OVN solution yet, and waiting on one is also not great. So, maybe a disabled-by-default option for people who really need it right now makes sense. \n\nThis feels like one in a long line of things where it\u0027d be really nice if we had some kind of real RPC from the ovn agent. OVN agent has access to the local OVS, so it can read BFD. It could communicate that info back. (We could technically do that today by writing to Chassis_Private.external_ids, but please don\u0027t do that--neutron subscribing to that on every worker is a cancer I\u0027m trying to cure.)\n\nIf we didn\u0027t want to use \"real rpc\" to avoid dependencies, technically ovsdb-server can host multiple DBs at once and you can define your own schema...we could have an ovn-agent-monitor process that ran in one place on the controlplane and didn\u0027t affect northd/ovn-controller/normal neutron workers. I\u0027m not necessarily suggesting this as wise, just something I recently thought of. Anyway, maybe OVN agent RPC of some kind is something that should be a PTG discussion.","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"}],"neutron/plugins/ml2/drivers/ovn/mech_driver/ovsdb/ovsdb_monitor.py":[{"author":{"_account_id":5756,"name":"Terry Wilson","email":"twilson@redhat.com","username":"otherwiseguy"},"change_message_id":"7b17d11bb42acb43c8d10e50a93bdf2853f8c08a","unresolved":true,"context_lines":[{"line_number":539,"context_line":"                     {\u0027router\u0027: router, \u0027host\u0027: host})"},{"line_number":540,"context_line":"            old_chassis \u003d getattr(old, \u0027chassis\u0027, None)"},{"line_number":541,"context_line":"            had_prev_chassis \u003d bool(old_chassis)"},{"line_number":542,"context_line":"            LOG.debug(\"BFD failover: port %s new_chassis\u003d%s \""},{"line_number":543,"context_line":"                      \"had_prev_chassis\u003d%s\","},{"line_number":544,"context_line":"                      row.logical_port, chassis[0].name, had_prev_chassis)"},{"line_number":545,"context_line":"            if (had_prev_chassis and"}],"source_content_type":"text/x-python","patch_set":4,"id":"bbde540c_b171c7c0","line":542,"updated":"2026-08-06 15:39:03.000000000","message":"I\u0027m a bit new to the scheduling code, so I have a question: Is BFD failover the only time we would match this event? It seems like it would happen any time the PB.chassis on a chassisredirect port changed. Wouldn\u0027t that also happen if, say, the HA_Chassis priority was changed via some other method, or a new chassis added with a higher priority? In this case, we\u0027d be demoting a chassis to the bottom and re-numbering despite it never crashing. Do we have enough information in neutron to really know why this happened? This feels a bit like something that should be an OVN feature, since it natively has access to all of this information.","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"ade6f9116ca16f3220ed43942e3ec1472abc8e37","unresolved":true,"context_lines":[{"line_number":539,"context_line":"                     {\u0027router\u0027: router, \u0027host\u0027: host})"},{"line_number":540,"context_line":"            old_chassis \u003d getattr(old, \u0027chassis\u0027, None)"},{"line_number":541,"context_line":"            had_prev_chassis \u003d bool(old_chassis)"},{"line_number":542,"context_line":"            LOG.debug(\"BFD failover: port %s new_chassis\u003d%s \""},{"line_number":543,"context_line":"                      \"had_prev_chassis\u003d%s\","},{"line_number":544,"context_line":"                      row.logical_port, chassis[0].name, had_prev_chassis)"},{"line_number":545,"context_line":"            if (had_prev_chassis and"}],"source_content_type":"text/x-python","patch_set":4,"id":"d1cc3028_7525ea71","line":542,"in_reply_to":"bbde540c_b171c7c0","updated":"2026-08-11 13:13:31.000000000","message":"@twilson@redhat.com thank you very much for looking into this and your invaluable comments.\nThe short answer is yes - there are other situations that would match this event and trigger priority demotion on a chassis that did not crash. Unfortunately, I overlooked those.\nAfter some extra research:\n1. I think adding a new chassis with a priority other than lowest is currently not possible. The scheduler refuses to promote a new chassis over the active one (to avoid traffic interruptions)\n2. out-of-band OVN DB manipulation is not supported as per the doc [1]. However 3.\n3. the OVN L3 router tool [2] allows an operator to completely reshuffle priorities (even on the active chassis). I completely lost it from my radar when working on this patch. But is a real example where your concerns apply.\n4. There are other situations that will also trigger demoting on a chassis that didn\u0027t crash, like:\n    a) removing a chassis from one AZ (updating ovn-cms-options.availability-zones value) - I tested this on devstack master branch,\n    b) removing a physnet / changing bridge-mappings on the active primary\nIndeed, neutron doesn\u0027t have enough information to precisely determine whether we are dealing with a failure or a legitimate chassis failover. \nAs you noticed, that information does not exist in the OVN SB either. It\u0027s an OVN feature to propagate the BFD state (from ovn-controller) of a geneve tunnel to the SN database.\nHaving said that, if accepted and implemented in OVN, it most likely won\u0027t be backported to the older/supported branches.\n\nOne possible option that comes to my mind is adding an extra check to the current patch that evaluates agent.is_alive state. This would allow for demoting only when the agent is dead. The potential downsides are:\n1. is_alive needs the agent_down_time to elapse, so an instantaneous chassis crash detection is not possible (default agent_down_time is 75s + some extra time for polling interval)\n2. because of 1., short network outages will still cause GW port bouncing (if the network outage is shorter than agent_down_time BFD would fail back the port)\n3. if there is an outage/hw failure on the data plane side only (ie NIC dedicated for the data plane) and the admin/oam connection remains active, the agent will never be considered as down -\u003e demoting will not happen -\u003e bouncing will happen when the data plane connection is active again\n\nThis approach would be sufficient for general HW failures/kernel crashes etc. However, before starting the implementation, I\u0027d like to get your comments (and other Neutron maintainers), if possible.\nThanks again for your support!\n\n[1] https://docs.openstack.org/neutron/latest/admin/ovn/l3_scheduler.html#re-schedule-logical-router-port-if-a-chassis-is-removed\n[2] https://bugs.launchpad.net/neutron/+bug/2103521","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"ad3630ee628b0f668cb5db5873bd7764f8dccaed","unresolved":true,"context_lines":[{"line_number":539,"context_line":"                     {\u0027router\u0027: router, \u0027host\u0027: host})"},{"line_number":540,"context_line":"            old_chassis \u003d getattr(old, \u0027chassis\u0027, None)"},{"line_number":541,"context_line":"            had_prev_chassis \u003d bool(old_chassis)"},{"line_number":542,"context_line":"            LOG.debug(\"BFD failover: port %s new_chassis\u003d%s \""},{"line_number":543,"context_line":"                      \"had_prev_chassis\u003d%s\","},{"line_number":544,"context_line":"                      row.logical_port, chassis[0].name, had_prev_chassis)"},{"line_number":545,"context_line":"            if (had_prev_chassis and"}],"source_content_type":"text/x-python","patch_set":4,"id":"705d65fd_241a2e17","line":542,"in_reply_to":"c0d7be0b_415891d9","updated":"2026-09-08 05:54:33.000000000","message":"Rodolfo,\nThank you for your comments. Could you please confirm that I correctly understood what you are proposing:\n\n1. You are generally ok if I redo my patch so it checks the `alive` status of the a before demoting it (and only demoting if it is dead)\n\n2. You are proposing/considering submitting an RFE for OVN (ovn-controller) to publish the tunnel state (BFD failure) to the SB db (you are still looking for Terry\u0027s comments here)\n\nIf I got any of the above wrong, please clarify. Thank you!","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":16688,"name":"Rodolfo Alonso","email":"ralonsoh@redhat.com","username":"rodolfo-alonso-hernandez"},"change_message_id":"029187c37b95ec2c768023069c979a73ac2f42a5","unresolved":true,"context_lines":[{"line_number":539,"context_line":"                     {\u0027router\u0027: router, \u0027host\u0027: host})"},{"line_number":540,"context_line":"            old_chassis \u003d getattr(old, \u0027chassis\u0027, None)"},{"line_number":541,"context_line":"            had_prev_chassis \u003d bool(old_chassis)"},{"line_number":542,"context_line":"            LOG.debug(\"BFD failover: port %s new_chassis\u003d%s \""},{"line_number":543,"context_line":"                      \"had_prev_chassis\u003d%s\","},{"line_number":544,"context_line":"                      row.logical_port, chassis[0].name, had_prev_chassis)"},{"line_number":545,"context_line":"            if (had_prev_chassis and"}],"source_content_type":"text/x-python","patch_set":4,"id":"c0d7be0b_415891d9","line":542,"in_reply_to":"d1cc3028_7525ea71","updated":"2026-08-27 14:47:24.000000000","message":"Since [1], as you mention in the [2] reference, it is possible to modify the existing HA_Chassis assignations and priorities. So Terry\u0027s concerns about what would happen with the previous chassis are legit. This code is not considering this API, it is just removing the previously assigned chassis.\n\nThere is no information about the BFD status in the NB/SB database. This could be done by reading locally in the OVS bfd_status.\n\nFor graceful stops (very rare), the ovn-controller deletes the `Chassis` and `Chassis_Private` registers. But this is most probably the corner case.\n\nIn order to demote a failed chassis, I would implement an event reading the `alive` status of each agent (as you suggest in 1). But at the cost, as you mentioned, of having a chassis bounce if the chassis is restored on time (under the `agent_down_time`).\n\nI can propose something (if Terry considers that as a good option): the ovn-controller, accessible by the OVN agent, classifies the tunnels as UP or DOWN. The OVN agent can update a new flag in `Chassis_Private` with this information. That could be a new extension or just a regular agent check. IMO, that is close to be a RFE.\n\n[1]https://review.opendev.org/c/openstack/neutron/+/982792\n[2]https://bugs.launchpad.net/neutron/+bug/2103521","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"}],"neutron/services/ovn_l3/plugin.py":[{"author":{"_account_id":1131,"name":"Brian Haley","email":"haleyb.dev@gmail.com","username":"brian-haley"},"change_message_id":"b8efec2cb04b28706915a36ec1bc9195f17adbbe","unresolved":true,"context_lines":[{"line_number":361,"context_line":"                      lrp_name)"},{"line_number":362,"context_line":"            return"},{"line_number":363,"context_line":""},{"line_number":364,"context_line":"        gw_chassis \u003d getattr(lrp, \u0027gateway_chassis\u0027, None)"},{"line_number":365,"context_line":"        if gw_chassis:"},{"line_number":366,"context_line":"            if not any(gc.chassis_name \u003d\u003d crashed_chassis_name"},{"line_number":367,"context_line":"                       for gc in gw_chassis):"}],"source_content_type":"text/x-python","patch_set":4,"id":"5925d7e4_9dfc7c74","line":364,"updated":"2026-07-16 16:17:31.000000000","message":"nit: if you have to respin, could un-indent all the code below as you did above:\n\nif not gw_chassis:\n    LOG.debug(\"BFD failover: LRP %s has no gateway_chassis entries, \"\n              \"skipping\", lrp_name)\n    return","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"},{"author":{"_account_id":30534,"name":"Marcin Wilk","email":"marcin.wilk@canonical.com","username":"wilkmar"},"change_message_id":"d5554107803378da334a5cc83364e09061d2be3b","unresolved":true,"context_lines":[{"line_number":361,"context_line":"                      lrp_name)"},{"line_number":362,"context_line":"            return"},{"line_number":363,"context_line":""},{"line_number":364,"context_line":"        gw_chassis \u003d getattr(lrp, \u0027gateway_chassis\u0027, None)"},{"line_number":365,"context_line":"        if gw_chassis:"},{"line_number":366,"context_line":"            if not any(gc.chassis_name \u003d\u003d crashed_chassis_name"},{"line_number":367,"context_line":"                       for gc in gw_chassis):"}],"source_content_type":"text/x-python","patch_set":4,"id":"0261b460_0f86a2f2","line":364,"in_reply_to":"5925d7e4_9dfc7c74","updated":"2026-07-17 06:53:39.000000000","message":"Acknowledged","commit_id":"f25ae40710a9c708e0e7d7d76b92fdec37b5d2e0"}]}
