)]}'
{"/COMMIT_MSG":[{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"07cfb6446dc1593863cf4835d0d40fcf2ba4aaf0","unresolved":true,"context_lines":[{"line_number":9,"context_line":"Neither Nova PCI passthrough nor Cyborg\u0027s existing drivers sanitize"},{"line_number":10,"context_line":"NVMe devices between tenants, and Cyborg\u0027s only native NVMe driver"},{"line_number":11,"context_line":"is Inspur-specific. This spec introduces NVMeDriver, a subclass of"},{"line_number":12,"context_line":"PciDriver, with five-tier secure erase on unbind: CES, BES, and OWS"},{"line_number":13,"context_line":"sanitize (controller-wide, covering deleted namespaces), then format"},{"line_number":14,"context_line":"--ses\u003d2 and --ses\u003d1 as fallbacks (active namespaces only)."},{"line_number":15,"context_line":""}],"source_content_type":"text/x-gerrit-commit-message","patch_set":8,"id":"0f810263_51e90a86","line":12,"updated":"2026-05-13 15:22:12.000000000","message":"I wasn\u0027t familiar with these acronyms, just for reference:\nCES \u003d Crypto Erase Suppor\nBES \u003d Block Erase Support\nOWS \u003d Overwrite Support\nSES \u003d Secure Erase Settings\n\nfrom the base specification https://nvmexpress.org/wp-content/uploads/NVM-Express-1_4c-2021.06.28-Ratified.pdf","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"e184f0d96aa4a41cde93298bfdcdb35b1f81d5d3","unresolved":true,"context_lines":[{"line_number":9,"context_line":"Neither Nova PCI passthrough nor Cyborg\u0027s existing drivers sanitize"},{"line_number":10,"context_line":"NVMe devices between tenants, and Cyborg\u0027s only native NVMe driver"},{"line_number":11,"context_line":"is Inspur-specific. This spec introduces NVMeDriver, a subclass of"},{"line_number":12,"context_line":"PciDriver, with five-tier secure erase on unbind: CES, BES, and OWS"},{"line_number":13,"context_line":"sanitize (controller-wide, covering deleted namespaces), then format"},{"line_number":14,"context_line":"--ses\u003d2 and --ses\u003d1 as fallbacks (active namespaces only)."},{"line_number":15,"context_line":""}],"source_content_type":"text/x-gerrit-commit-message","patch_set":8,"id":"d2896dbe_c3e2b830","line":12,"in_reply_to":"0f810263_51e90a86","updated":"2026-05-19 06:14:44.000000000","message":"Thank you, added them in the commit message.","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"a57e6fab0658fa63713df9e34b405bc7e555cbc3","unresolved":false,"context_lines":[{"line_number":9,"context_line":"Neither Nova PCI passthrough nor Cyborg\u0027s existing drivers sanitize"},{"line_number":10,"context_line":"NVMe devices between tenants, and Cyborg\u0027s only native NVMe driver"},{"line_number":11,"context_line":"is Inspur-specific. This spec introduces NVMeDriver, a subclass of"},{"line_number":12,"context_line":"PciDriver, with five-tier secure erase on unbind: CES, BES, and OWS"},{"line_number":13,"context_line":"sanitize (controller-wide, covering deleted namespaces), then format"},{"line_number":14,"context_line":"--ses\u003d2 and --ses\u003d1 as fallbacks (active namespaces only)."},{"line_number":15,"context_line":""}],"source_content_type":"text/x-gerrit-commit-message","patch_set":8,"id":"41c3be2d_4dd2227f","line":12,"in_reply_to":"d2896dbe_c3e2b830","updated":"2026-05-19 07:25:22.000000000","message":"Done","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"cbbd477b89720f5289c15024c36f3d305e1bddb1","unresolved":true,"context_lines":[{"line_number":14,"context_line":"sanitize (controller-wide, covering deleted namespaces), then format"},{"line_number":15,"context_line":"--ses\u003d2 and --ses\u003d1 as fallbacks (active namespaces only)."},{"line_number":16,"context_line":""},{"line_number":17,"context_line":"A new \"cleaning\" status and cleanup_failed column guard the lifecycle."},{"line_number":18,"context_line":"Cyborg sets reserved\u003dtotal in Placement before releasing the ARQ"},{"line_number":19,"context_line":"(Accelerator Request) allocation to close the scheduling race window."},{"line_number":20,"context_line":"Failed cleanups land in \"maintaining\" and are retried via"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":10,"id":"a45954e0_88bbf7d0","line":17,"range":{"start_line":17,"start_character":0,"end_line":17,"end_character":69},"updated":"2026-05-19 07:32:51.000000000","message":"I am currently using cleanup_failed status to track the failed cleanup device and using existing maintaining status. I am not sure this is the correct way to track cleanup failed devices.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"bdc24e7c58526707f7015bb40d75fc3c2d312416","unresolved":false,"context_lines":[{"line_number":14,"context_line":"sanitize (controller-wide, covering deleted namespaces), then format"},{"line_number":15,"context_line":"--ses\u003d2 and --ses\u003d1 as fallbacks (active namespaces only)."},{"line_number":16,"context_line":""},{"line_number":17,"context_line":"A new \"cleaning\" status and cleanup_failed column guard the lifecycle."},{"line_number":18,"context_line":"Cyborg sets reserved\u003dtotal in Placement before releasing the ARQ"},{"line_number":19,"context_line":"(Accelerator Request) allocation to close the scheduling race window."},{"line_number":20,"context_line":"Failed cleanups land in \"maintaining\" and are retried via"}],"source_content_type":"text/x-gerrit-commit-message","patch_set":10,"id":"d3bbd28b_502ac895","line":17,"range":{"start_line":17,"start_character":0,"end_line":17,"end_character":69},"in_reply_to":"a45954e0_88bbf7d0","updated":"2026-06-19 05:10:08.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"}],"/PATCHSET_LEVEL":[{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"238d2dc7bfdc76b0cb4d1c756cf8ff85a3c2ddca","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":10,"id":"5e2e8d3e_dda7a656","updated":"2026-06-08 06:16:32.000000000","message":"Thank you for the feedback, Addressing the comments.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"e589fdf9e9d0dbc3ea17a46ce3b62b05d2ecdefe","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":10,"id":"c34367eb_353a14a6","updated":"2026-05-20 15:35:40.000000000","message":"a few concerns mostly about negative testing that I think we\u0027re missing in the spec.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[],"source_content_type":"","patch_set":10,"id":"ab395670_76bc9659","updated":"2026-06-03 16:06:13.000000000","message":"as meta comment the majroyitn of this docuement shoudl be prose with sub heading wehre need not list\n\nthe excptions ot that are the usecase and work items \n\nthe reset shoud avoid lists unless it make a point clearer.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"bdc24e7c58526707f7015bb40d75fc3c2d312416","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":10,"id":"2e6470ca_9b127181","in_reply_to":"ab395670_76bc9659","updated":"2026-06-19 05:10:08.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"58060b030dcbe7b315bec92b80263016f27fb90c","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":11,"id":"b2fee572_23eb52bb","updated":"2026-06-16 07:18:16.000000000","message":"Need to reply the comments!","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":11,"id":"0422774a_7f28ed0f","updated":"2026-06-19 04:57:23.000000000","message":"Thank you Sean, Sylvain, Joan for the detailed feedback, I have addressed most of them, If it get missed, do let me know. thank you!","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":11,"id":"efb317db_1655b157","updated":"2026-06-17 04:45:31.000000000","message":"just cleaing up some fo the comments ill do a full review of this over the next day or two","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":11,"id":"c7a5e777_06e25e00","updated":"2026-06-17 15:27:52.000000000","message":"still mostly just cleanign uip the preior comments","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"5fadc81cf9d167ddd76e3a00378f41c272d789d9","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":15,"id":"a2dd951c_64fccd6f","updated":"2026-06-23 14:29:45.000000000","message":"Need one more review from my end.","commit_id":"dceca2a5eaabb579f37a88c0c08898cd09e87764"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"21ad868201ec441db039643b4a2840cedf8807bf","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":17,"id":"2f41db06_d1165d72","updated":"2026-06-26 13:47:12.000000000","message":"Thank you @jgilaber@redhat.com for the comments, I will push a new revision based on the feedback.","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b6c8fda169c0ced92db5fa328a947458aedfc883","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":17,"id":"0846859d_f76dd479","updated":"2026-06-24 06:05:16.000000000","message":"Thank you @smooney@redhat.com for all the feedback, I tried to address most of them, If I missed anything do let me know. \n\nThe Spec is now ready for review. thank you!","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"b42cb2d4e2f69f928932dd1e3a1110f6609a8765","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":21,"id":"d4a81ba5_870f3dc3","updated":"2026-06-30 15:39:37.000000000","message":"lgtm, the proposal is clear, thanks!","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":21,"id":"9460a479_515020a3","updated":"2026-06-30 19:46:58.000000000","message":"this is very close\n\nwe need to tighten the upgrade storay a bit both at the rpc and db level otherwise i think tis is mostly good to go.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":22,"id":"46af331f_f1c280e8","updated":"2026-07-01 07:56:58.000000000","message":"Thank you Melanie, Sean for the review.\n\nI have addressed most of them. I have left few comments unresolved to verify the content of those section.","commit_id":"8909064b10ee6b305e0b90e1ff03348b4a91187c"},{"author":{"_account_id":4690,"name":"melanie witt","display_name":"melwitt","email":"melwittt@gmail.com","username":"melwitt"},"change_message_id":"4f8fb4cb3e872599868ea52bfc612d6328c196ca","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":25,"id":"d2d55e72_d76da621","updated":"2026-07-01 18:41:31.000000000","message":"Sorry to -1 this but the introduction of HTTP 501 since PS21 as an expected/normal API response does not seem right to me ... thoughts?","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":4690,"name":"melanie witt","display_name":"melwitt","email":"melwittt@gmail.com","username":"melwitt"},"change_message_id":"3522c793119b9f70dd77beea16df396f5720ba8b","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":26,"id":"bc468ae2_2b0c63f9","updated":"2026-07-02 16:48:27.000000000","message":"For formality: thank you for updating HTTP 501 \u003d\u003e HTTP 400, spec looks good.","commit_id":"b6c43f73c5d042cf2c8a10834e534b8f3bfe826a"}],"specs/2026.2/approved/generic-nvme-driver-with-secure-cleanup.rst":[{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"07cfb6446dc1593863cf4835d0d40fcf2ba4aaf0","unresolved":true,"context_lines":[{"line_number":45,"context_line":"nvme driver. With this the operator gets locked in with inspur vendor only."},{"line_number":46,"context_line":"There is no way to use other nvme hardware vendor devices with Cyborg."},{"line_number":47,"context_line":""},{"line_number":48,"context_line":"Alternatively Operator can configure any NVME PCI driver via nova pci"},{"line_number":49,"context_line":"passthrough device_spec configuration and use it with the instance."},{"line_number":50,"context_line":""},{"line_number":51,"context_line":"Once the instance gets deleted, there is no way to automatically securely"}],"source_content_type":"text/x-rst","patch_set":8,"id":"ea549a9a_0acb1a29","line":48,"updated":"2026-05-13 15:22:12.000000000","message":"should `driver` be replaced with `device` in this sentence? Also, I would add a mention that this alternative approach does not use cyborg, it relies solely on Nova to do the passthrough","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"e184f0d96aa4a41cde93298bfdcdb35b1f81d5d3","unresolved":false,"context_lines":[{"line_number":45,"context_line":"nvme driver. With this the operator gets locked in with inspur vendor only."},{"line_number":46,"context_line":"There is no way to use other nvme hardware vendor devices with Cyborg."},{"line_number":47,"context_line":""},{"line_number":48,"context_line":"Alternatively Operator can configure any NVME PCI driver via nova pci"},{"line_number":49,"context_line":"passthrough device_spec configuration and use it with the instance."},{"line_number":50,"context_line":""},{"line_number":51,"context_line":"Once the instance gets deleted, there is no way to automatically securely"}],"source_content_type":"text/x-rst","patch_set":8,"id":"68a02e2e_10fe34db","line":48,"in_reply_to":"ea549a9a_0acb1a29","updated":"2026-05-19 06:14:44.000000000","message":"Done","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"07cfb6446dc1593863cf4835d0d40fcf2ba4aaf0","unresolved":true,"context_lines":[{"line_number":51,"context_line":"Once the instance gets deleted, there is no way to automatically securely"},{"line_number":52,"context_line":"cleanup the nvme device for further use. The operator needs to manually"},{"line_number":53,"context_line":"cleanup the nvme device after deallocation and also manage the life cycle"},{"line_number":54,"context_line":"manually so that dirty nvme does not get assigned to other instance which"},{"line_number":55,"context_line":"is time consuming and error prone."},{"line_number":56,"context_line":""},{"line_number":57,"context_line":"Without a cleanup story, operators cannot rely on Cyborg to enforce reuse"}],"source_content_type":"text/x-rst","patch_set":8,"id":"9113d437_b9fbb79a","line":54,"updated":"2026-05-13 15:22:12.000000000","message":"what does this manual life cycle management entail? Making sure that the nvme is not bound to another instance while it\u0027s being cleaned up?","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"a57e6fab0658fa63713df9e34b405bc7e555cbc3","unresolved":false,"context_lines":[{"line_number":51,"context_line":"Once the instance gets deleted, there is no way to automatically securely"},{"line_number":52,"context_line":"cleanup the nvme device for further use. The operator needs to manually"},{"line_number":53,"context_line":"cleanup the nvme device after deallocation and also manage the life cycle"},{"line_number":54,"context_line":"manually so that dirty nvme does not get assigned to other instance which"},{"line_number":55,"context_line":"is time consuming and error prone."},{"line_number":56,"context_line":""},{"line_number":57,"context_line":"Without a cleanup story, operators cannot rely on Cyborg to enforce reuse"}],"source_content_type":"text/x-rst","patch_set":8,"id":"1f5fcfef_9ff2d286","line":54,"in_reply_to":"3cbe6fd9_b5af6d6f","updated":"2026-05-19 07:25:22.000000000","message":"Done","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"e184f0d96aa4a41cde93298bfdcdb35b1f81d5d3","unresolved":true,"context_lines":[{"line_number":51,"context_line":"Once the instance gets deleted, there is no way to automatically securely"},{"line_number":52,"context_line":"cleanup the nvme device for further use. The operator needs to manually"},{"line_number":53,"context_line":"cleanup the nvme device after deallocation and also manage the life cycle"},{"line_number":54,"context_line":"manually so that dirty nvme does not get assigned to other instance which"},{"line_number":55,"context_line":"is time consuming and error prone."},{"line_number":56,"context_line":""},{"line_number":57,"context_line":"Without a cleanup story, operators cannot rely on Cyborg to enforce reuse"}],"source_content_type":"text/x-rst","patch_set":8,"id":"3cbe6fd9_b5af6d6f","line":54,"in_reply_to":"9113d437_b9fbb79a","updated":"2026-05-19 06:14:44.000000000","message":"Yes correct, Done","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"07cfb6446dc1593863cf4835d0d40fcf2ba4aaf0","unresolved":true,"context_lines":[{"line_number":70,"context_line":"  request using a device profile, Cyborg should allocate a clean NVMe"},{"line_number":71,"context_line":"  device and bind it to the instance. When the instance is deleted,"},{"line_number":72,"context_line":"  Cyborg should securely sanitize the device before making it available"},{"line_number":73,"context_line":"  for reallocation. It should not perform resize, cold and live migration"},{"line_number":74,"context_line":"  of the instances."},{"line_number":75,"context_line":""},{"line_number":76,"context_line":"* A deployment with mixed NVMe hardware from different vendors should be"}],"source_content_type":"text/x-rst","patch_set":8,"id":"9f73e00b_d3e96b2a","line":73,"updated":"2026-05-13 15:22:12.000000000","message":"is the last sentence about not performing resizes and migrations just reflecting current limitations or we want to purposely limit those operations by design?","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"a57e6fab0658fa63713df9e34b405bc7e555cbc3","unresolved":false,"context_lines":[{"line_number":70,"context_line":"  request using a device profile, Cyborg should allocate a clean NVMe"},{"line_number":71,"context_line":"  device and bind it to the instance. When the instance is deleted,"},{"line_number":72,"context_line":"  Cyborg should securely sanitize the device before making it available"},{"line_number":73,"context_line":"  for reallocation. It should not perform resize, cold and live migration"},{"line_number":74,"context_line":"  of the instances."},{"line_number":75,"context_line":""},{"line_number":76,"context_line":"* A deployment with mixed NVMe hardware from different vendors should be"}],"source_content_type":"text/x-rst","patch_set":8,"id":"782a56f2_4514781c","line":73,"in_reply_to":"0c9be070_33ac8c61","updated":"2026-05-19 07:25:22.000000000","message":"Done","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"e184f0d96aa4a41cde93298bfdcdb35b1f81d5d3","unresolved":true,"context_lines":[{"line_number":70,"context_line":"  request using a device profile, Cyborg should allocate a clean NVMe"},{"line_number":71,"context_line":"  device and bind it to the instance. When the instance is deleted,"},{"line_number":72,"context_line":"  Cyborg should securely sanitize the device before making it available"},{"line_number":73,"context_line":"  for reallocation. It should not perform resize, cold and live migration"},{"line_number":74,"context_line":"  of the instances."},{"line_number":75,"context_line":""},{"line_number":76,"context_line":"* A deployment with mixed NVMe hardware from different vendors should be"}],"source_content_type":"text/x-rst","patch_set":8,"id":"0c9be070_33ac8c61","line":73,"in_reply_to":"9f73e00b_d3e96b2a","updated":"2026-05-19 06:14:44.000000000","message":"Fixed it with notes.","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"07cfb6446dc1593863cf4835d0d40fcf2ba4aaf0","unresolved":true,"context_lines":[{"line_number":105,"context_line":"all the PCI devices and filter out the NVME device. The `cleanup()` method"},{"line_number":106,"context_line":"will implement NVME cleanup using `nvme-cli`."},{"line_number":107,"context_line":""},{"line_number":108,"context_line":"Since the PCI driver already has built-in logic for bind and unbind a"},{"line_number":109,"context_line":"device from an instance. The nvme driver can use the same."},{"line_number":110,"context_line":""},{"line_number":111,"context_line":"NVMe Device Lifecycle Flow"}],"source_content_type":"text/x-rst","patch_set":8,"id":"90facb0d_32c2e4a8","line":108,"updated":"2026-05-13 15:22:12.000000000","message":"is this accurate? I looked at the flow when receiving an unbind request from nova and I don\u0027t see any specific path for pci devices in https://github.com/openstack/cyborg/blob/186cdd2b76aff57233660042f71761383285e5e8/cyborg/objects/ext_arq.py#L321 nor in https://github.com/openstack/cyborg/blob/master/cyborg/objects/ext_arq.py#L287. The code there is shared for all drivers, so IIUC adding a cleanup call there will need to be guarded to ensure it\u0027s only done for AttachHandles of type PCI","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"e184f0d96aa4a41cde93298bfdcdb35b1f81d5d3","unresolved":true,"context_lines":[{"line_number":105,"context_line":"all the PCI devices and filter out the NVME device. The `cleanup()` method"},{"line_number":106,"context_line":"will implement NVME cleanup using `nvme-cli`."},{"line_number":107,"context_line":""},{"line_number":108,"context_line":"Since the PCI driver already has built-in logic for bind and unbind a"},{"line_number":109,"context_line":"device from an instance. The nvme driver can use the same."},{"line_number":110,"context_line":""},{"line_number":111,"context_line":"NVMe Device Lifecycle Flow"}],"source_content_type":"text/x-rst","patch_set":8,"id":"d6765d6e_606fb27d","line":108,"in_reply_to":"90facb0d_32c2e4a8","updated":"2026-05-19 06:14:44.000000000","message":"thank you for pointing that out, fix it.","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"a57e6fab0658fa63713df9e34b405bc7e555cbc3","unresolved":false,"context_lines":[{"line_number":105,"context_line":"all the PCI devices and filter out the NVME device. The `cleanup()` method"},{"line_number":106,"context_line":"will implement NVME cleanup using `nvme-cli`."},{"line_number":107,"context_line":""},{"line_number":108,"context_line":"Since the PCI driver already has built-in logic for bind and unbind a"},{"line_number":109,"context_line":"device from an instance. The nvme driver can use the same."},{"line_number":110,"context_line":""},{"line_number":111,"context_line":"NVMe Device Lifecycle Flow"}],"source_content_type":"text/x-rst","patch_set":8,"id":"067be0c3_c28bb9f7","line":108,"in_reply_to":"d6765d6e_606fb27d","updated":"2026-05-19 07:25:22.000000000","message":"Done","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"07cfb6446dc1593863cf4835d0d40fcf2ba4aaf0","unresolved":true,"context_lines":[{"line_number":468,"context_line":"Data model impact"},{"line_number":469,"context_line":"-----------------"},{"line_number":470,"context_line":""},{"line_number":471,"context_line":"One additive Alembic migration is needed:"},{"line_number":472,"context_line":""},{"line_number":473,"context_line":"* Add ``NVME`` to the ``type`` column ENUM/CHECK constraint."},{"line_number":474,"context_line":"* Add ``\"cleaning\"`` to the ``status`` column ENUM/CHECK constraint."}],"source_content_type":"text/x-rst","patch_set":8,"id":"15504a81_d37a063c","line":471,"updated":"2026-05-13 15:22:12.000000000","message":"I think it would be good to mention that all the changes are made to the `Device` db model. It can be inferred from the last sentence of the section but it would not hurt to mention it explicitely","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"a57e6fab0658fa63713df9e34b405bc7e555cbc3","unresolved":false,"context_lines":[{"line_number":468,"context_line":"Data model impact"},{"line_number":469,"context_line":"-----------------"},{"line_number":470,"context_line":""},{"line_number":471,"context_line":"One additive Alembic migration is needed:"},{"line_number":472,"context_line":""},{"line_number":473,"context_line":"* Add ``NVME`` to the ``type`` column ENUM/CHECK constraint."},{"line_number":474,"context_line":"* Add ``\"cleaning\"`` to the ``status`` column ENUM/CHECK constraint."}],"source_content_type":"text/x-rst","patch_set":8,"id":"e2c5a59d_f20dc2fe","line":471,"in_reply_to":"15504a81_d37a063c","updated":"2026-05-19 07:25:22.000000000","message":"Done","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"07cfb6446dc1593863cf4835d0d40fcf2ba4aaf0","unresolved":true,"context_lines":[{"line_number":600,"context_line":"  ``nvme_driver`` in the ``[agent]`` section, and restart the cyborg-agent."},{"line_number":601,"context_line":"  Cyborg will then discover and manage the device."},{"line_number":602,"context_line":""},{"line_number":603,"context_line":"* **Existing Inspur driver deployments**: The Inspur NVMe driver is"},{"line_number":604,"context_line":"  unaffected. To also use the generic NVMe driver, add ``nvme_driver`` to"},{"line_number":605,"context_line":"  ``[agent] enabled_drivers``. Both drivers can coexist."},{"line_number":606,"context_line":""}],"source_content_type":"text/x-rst","patch_set":8,"id":"b2e0f8c6_d43bdcc0","line":603,"updated":"2026-05-13 15:22:12.000000000","message":"What would happen if an operator by mistake enables both the generic driver and the inspur one and both discover the same device? Will it be duplicated in the database?","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":600,"context_line":"  ``nvme_driver`` in the ``[agent]`` section, and restart the cyborg-agent."},{"line_number":601,"context_line":"  Cyborg will then discover and manage the device."},{"line_number":602,"context_line":""},{"line_number":603,"context_line":"* **Existing Inspur driver deployments**: The Inspur NVMe driver is"},{"line_number":604,"context_line":"  unaffected. To also use the generic NVMe driver, add ``nvme_driver`` to"},{"line_number":605,"context_line":"  ``[agent] enabled_drivers``. Both drivers can coexist."},{"line_number":606,"context_line":""}],"source_content_type":"text/x-rst","patch_set":8,"id":"29be3410_faef6308","line":603,"in_reply_to":"2728dd69_6614cb8a","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"a57e6fab0658fa63713df9e34b405bc7e555cbc3","unresolved":true,"context_lines":[{"line_number":600,"context_line":"  ``nvme_driver`` in the ``[agent]`` section, and restart the cyborg-agent."},{"line_number":601,"context_line":"  Cyborg will then discover and manage the device."},{"line_number":602,"context_line":""},{"line_number":603,"context_line":"* **Existing Inspur driver deployments**: The Inspur NVMe driver is"},{"line_number":604,"context_line":"  unaffected. To also use the generic NVMe driver, add ``nvme_driver`` to"},{"line_number":605,"context_line":"  ``[agent] enabled_drivers``. Both drivers can coexist."},{"line_number":606,"context_line":""}],"source_content_type":"text/x-rst","patch_set":8,"id":"f2a0e945_008de6cd","line":603,"in_reply_to":"b2e0f8c6_d43bdcc0","updated":"2026-05-19 07:25:22.000000000","message":"It will raise conflict once we start the cyborg agent. Fix it now.","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":600,"context_line":"  ``nvme_driver`` in the ``[agent]`` section, and restart the cyborg-agent."},{"line_number":601,"context_line":"  Cyborg will then discover and manage the device."},{"line_number":602,"context_line":""},{"line_number":603,"context_line":"* **Existing Inspur driver deployments**: The Inspur NVMe driver is"},{"line_number":604,"context_line":"  unaffected. To also use the generic NVMe driver, add ``nvme_driver`` to"},{"line_number":605,"context_line":"  ``[agent] enabled_drivers``. Both drivers can coexist."},{"line_number":606,"context_line":""}],"source_content_type":"text/x-rst","patch_set":8,"id":"2728dd69_6614cb8a","line":603,"in_reply_to":"f2a0e945_008de6cd","updated":"2026-06-03 16:06:13.000000000","message":"so this is valid to keep, what i would callout here is the intent to deprecate the inspru and base ssd driver for future feature devleopement but no other impact.","commit_id":"746905ed7bd9a96bf90fb0dc25771bb53d7c1469"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":13,"context_line":"    +spec/generic-nvme-driver-with-secure-cleanup"},{"line_number":14,"context_line":""},{"line_number":15,"context_line":"Blueprint: `generic-nvme-driver-with-secure-cleanup-bp`_"},{"line_number":16,"context_line":""},{"line_number":17,"context_line":"In OpenStack, we can add nvme devices as a PCI device via cyborg,"},{"line_number":18,"context_line":"as a device_spec via nova pci passthrough or use the dedicated inspur"},{"line_number":19,"context_line":"nvme cyborg driver with inspur nvme device to a dedicated flavor in order"},{"line_number":20,"context_line":"to use it with Instance. Once the instance gets deleted, there is no way"},{"line_number":21,"context_line":"to securely erase data from nvme device with OpenStack. The same nvme"},{"line_number":22,"context_line":"device gets allocated to other instance without cleanup. It may cause"},{"line_number":23,"context_line":"issues with new application running on new instance with data corruption."},{"line_number":24,"context_line":"Basically there is no way to manage the life cycle of nvme device in"},{"line_number":25,"context_line":"OpenStack."},{"line_number":26,"context_line":""},{"line_number":27,"context_line":"Also there is no other way to use other vendor provided nvme hardware"},{"line_number":28,"context_line":"with Cyborg."},{"line_number":29,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"30146eba_91b3bd82","line":26,"range":{"start_line":16,"start_character":1,"end_line":26,"end_character":1},"updated":"2026-06-03 16:06:13.000000000","message":"tehcnnally nova does not supprot using any stateful deviec like an nvme ssd via its current pci passthough feature even when markign the device as one time use\n\ni exiplcitly objected to allowing that to be considered supproted by the project so we shoudl not suggest its actully a supproted usecase\n\nthe same shoudl be true for cybrog pci driver\n\nits possibel to use them to od this today but if you have any issues as a result its entirly on you and it is not somehtign i would personally be comfortable allowing a customer to do.\n\n\nthe inspur driver was intended for this usecase but to me it never met the multitance requrieemnt for inclution in cybrog since it does not do device cleaning\n\nso i would like to eventually deprecate adn remove that driver or rebase it on top of your new driver goign forward as your new driver will supprot multi tenancy so it can entrily surplant it going forward.\n\n\n\n```suggestion\nIn OpenStack there are several way to pass-though generic pci devices\nincluding a dedicated in-spur nvme cyborg driver, nova pci pass-though and cyborg generic pci drvier. None of these existing mechanisms support multi tenancy when used with nvme devices nor are suitable for use in a public cloud as a result.\nOnce the instance gets deleted, there is no wayto securely erase data from nvme device with OpenStack. The same nvme device gets allocated to other instance\nwithout cleanup. This can leak sensitive information between tenant and may cause issues with workload by hijacking the boot process and compromising the guest.\nBasically there is no way to manage the life cycle of nvme device in\nOpenStack.\n```","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":13,"context_line":"    +spec/generic-nvme-driver-with-secure-cleanup"},{"line_number":14,"context_line":""},{"line_number":15,"context_line":"Blueprint: `generic-nvme-driver-with-secure-cleanup-bp`_"},{"line_number":16,"context_line":""},{"line_number":17,"context_line":"In OpenStack, we can add nvme devices as a PCI device via cyborg,"},{"line_number":18,"context_line":"as a device_spec via nova pci passthrough or use the dedicated inspur"},{"line_number":19,"context_line":"nvme cyborg driver with inspur nvme device to a dedicated flavor in order"},{"line_number":20,"context_line":"to use it with Instance. Once the instance gets deleted, there is no way"},{"line_number":21,"context_line":"to securely erase data from nvme device with OpenStack. The same nvme"},{"line_number":22,"context_line":"device gets allocated to other instance without cleanup. It may cause"},{"line_number":23,"context_line":"issues with new application running on new instance with data corruption."},{"line_number":24,"context_line":"Basically there is no way to manage the life cycle of nvme device in"},{"line_number":25,"context_line":"OpenStack."},{"line_number":26,"context_line":""},{"line_number":27,"context_line":"Also there is no other way to use other vendor provided nvme hardware"},{"line_number":28,"context_line":"with Cyborg."},{"line_number":29,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"1d40c123_dfbf45e5","line":26,"range":{"start_line":16,"start_character":1,"end_line":26,"end_character":1},"in_reply_to":"30146eba_91b3bd82","updated":"2026-06-17 04:45:31.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":24,"context_line":"Basically there is no way to manage the life cycle of nvme device in"},{"line_number":25,"context_line":"OpenStack."},{"line_number":26,"context_line":""},{"line_number":27,"context_line":"Also there is no other way to use other vendor provided nvme hardware"},{"line_number":28,"context_line":"with Cyborg."},{"line_number":29,"context_line":""},{"line_number":30,"context_line":"This blueprint proposes a generic Cyborg NVMe driver that will manage the"},{"line_number":31,"context_line":"whole life cycle of nvme device(s). It includes always binding a clean"}],"source_content_type":"text/x-rst","patch_set":10,"id":"b96244d9_9553d0c8","line":28,"range":{"start_line":27,"start_character":0,"end_line":28,"end_character":12},"updated":"2026-06-03 16:06:13.000000000","message":"you can just use the pci driver\ni woudl delete this line","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":24,"context_line":"Basically there is no way to manage the life cycle of nvme device in"},{"line_number":25,"context_line":"OpenStack."},{"line_number":26,"context_line":""},{"line_number":27,"context_line":"Also there is no other way to use other vendor provided nvme hardware"},{"line_number":28,"context_line":"with Cyborg."},{"line_number":29,"context_line":""},{"line_number":30,"context_line":"This blueprint proposes a generic Cyborg NVMe driver that will manage the"},{"line_number":31,"context_line":"whole life cycle of nvme device(s). It includes always binding a clean"}],"source_content_type":"text/x-rst","patch_set":10,"id":"4a94802d_393ec682","line":28,"range":{"start_line":27,"start_character":0,"end_line":28,"end_character":12},"in_reply_to":"b96244d9_9553d0c8","updated":"2026-06-17 04:45:31.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":30,"context_line":"This blueprint proposes a generic Cyborg NVMe driver that will manage the"},{"line_number":31,"context_line":"whole life cycle of nvme device(s). It includes always binding a clean"},{"line_number":32,"context_line":"nvme device to the new instance and securely cleaning up the nvme device"},{"line_number":33,"context_line":"after deallocation for reuse."},{"line_number":34,"context_line":""},{"line_number":35,"context_line":"Problem description"},{"line_number":36,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"}],"source_content_type":"text/x-rst","patch_set":10,"id":"a18b7fb3_18ad9083","line":33,"updated":"2026-06-03 16:06:13.000000000","message":"+1\n\nin the future we may want ot expand this to other lifecycle operations like snapshot/restore or programming a glance image to the nvme device but i agree with declaring those as out of scope for the short to medium term.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":30,"context_line":"This blueprint proposes a generic Cyborg NVMe driver that will manage the"},{"line_number":31,"context_line":"whole life cycle of nvme device(s). It includes always binding a clean"},{"line_number":32,"context_line":"nvme device to the new instance and securely cleaning up the nvme device"},{"line_number":33,"context_line":"after deallocation for reuse."},{"line_number":34,"context_line":""},{"line_number":35,"context_line":"Problem description"},{"line_number":36,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"}],"source_content_type":"text/x-rst","patch_set":10,"id":"1f8e27a2_5f41e95d","line":33,"in_reply_to":"a18b7fb3_18ad9083","updated":"2026-06-17 04:45:31.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":36,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":37,"context_line":""},{"line_number":38,"context_line":"OpenStack lacks automated lifecycle management for NVMe devices. When a"},{"line_number":39,"context_line":"Cyborg-managed or Nova PCI passthrough NVMe device is detached from an"},{"line_number":40,"context_line":"instance, tenant data remains on the device. Without automated sanitization,"},{"line_number":41,"context_line":"operators must manually track device allocation, run ``nvme sanitize`` or"},{"line_number":42,"context_line":"``nvme format`` commands via SSH after instance deletion, verify cleanup"}],"source_content_type":"text/x-rst","patch_set":10,"id":"ac1b420a_5918d785","line":39,"range":{"start_line":39,"start_character":17,"end_line":39,"end_character":62},"updated":"2026-06-03 16:06:13.000000000","message":"again this is possibel but offically unsupproted by nova.\n\nwe do not supprot nvme device pashtough upstream even when usign OTU devices.\nthat was a hard line form me when we dicussed that feature so we shoudl jsut remvoe this form the spec.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":36,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":37,"context_line":""},{"line_number":38,"context_line":"OpenStack lacks automated lifecycle management for NVMe devices. When a"},{"line_number":39,"context_line":"Cyborg-managed or Nova PCI passthrough NVMe device is detached from an"},{"line_number":40,"context_line":"instance, tenant data remains on the device. Without automated sanitization,"},{"line_number":41,"context_line":"operators must manually track device allocation, run ``nvme sanitize`` or"},{"line_number":42,"context_line":"``nvme format`` commands via SSH after instance deletion, verify cleanup"}],"source_content_type":"text/x-rst","patch_set":10,"id":"5dedfa0d_df781c99","line":39,"range":{"start_line":39,"start_character":17,"end_line":39,"end_character":62},"in_reply_to":"ac1b420a_5918d785","updated":"2026-06-17 04:45:31.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":false,"context_lines":[{"line_number":41,"context_line":"operators must manually track device allocation, run ``nvme sanitize`` or"},{"line_number":42,"context_line":"``nvme format`` commands via SSH after instance deletion, verify cleanup"},{"line_number":43,"context_line":"completion, and ensure devices are not reallocated during cleanup. This"},{"line_number":44,"context_line":"manual process is time-consuming, error-prone, and does not scale."},{"line_number":45,"context_line":""},{"line_number":46,"context_line":"Additionally, Cyborg\u0027s existing Inspur NVMe driver locks operators into a"},{"line_number":47,"context_line":"single hardware vendor. Operators cannot manage NVMe devices from other"}],"source_content_type":"text/x-rst","patch_set":10,"id":"aa2cf20a_03b624c5","line":44,"updated":"2026-06-03 16:06:13.000000000","message":"this is why it not supproted upstream in nova.\n\nwe do not consider that to meet our requirement for multi tenancy and providing multi teanacy is one fo the core painpoint that this new feature will adress.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":false,"context_lines":[{"line_number":77,"context_line":"* As a Cyborg developer, I want a base driver cleanup interface that"},{"line_number":78,"context_line":"  vendor-specific drivers can override, so cleanup behavior remains"},{"line_number":79,"context_line":"  extensible without changing the conductor unbind contract."},{"line_number":80,"context_line":""},{"line_number":81,"context_line":"Proposed change"},{"line_number":82,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":83,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"6d0f9bb9_d61192bb","line":80,"updated":"2026-06-03 16:06:13.000000000","message":"+1 i agree these are all good usecase to enable and motivate this spec well","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":82,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":83,"context_line":""},{"line_number":84,"context_line":"This blueprint adds a generic NVMe driver (NVMeDriver) by inheriting from"},{"line_number":85,"context_line":"PciDriver. A no-op ``cleanup(pci_addr)`` method is added to PciDriver so"},{"line_number":86,"context_line":"that existing drivers are unaffected on unbind. Vendor-specific drivers can"},{"line_number":87,"context_line":"be further created by inheriting from NVMeDriver::"},{"line_number":88,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"c820056c_40526ea8","line":85,"range":{"start_line":85,"start_character":10,"end_line":85,"end_character":69},"updated":"2026-06-03 16:06:13.000000000","message":"the cyborg agent should not have an driver awarenese\nit does nto always meet this goal today but it shoudl not need to care that its a pci or nvme drvier\n\nwhat that means is the geneirc cleanup funciton shoudl be added to the base generic driver as a simiple noop for backward compatiablity not the PciDrvier\nit will get it via inheritnace.\n\nso add it here\n\nhttps://github.com/openstack/cyborg/blob/2875d3c12d4484e9336ba5084f32f2acf83a2366/cyborg/accelerator/drivers/driver.py\n\nwe can use the fake drivre to test this as well.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":82,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":83,"context_line":""},{"line_number":84,"context_line":"This blueprint adds a generic NVMe driver (NVMeDriver) by inheriting from"},{"line_number":85,"context_line":"PciDriver. A no-op ``cleanup(pci_addr)`` method is added to PciDriver so"},{"line_number":86,"context_line":"that existing drivers are unaffected on unbind. Vendor-specific drivers can"},{"line_number":87,"context_line":"be further created by inheriting from NVMeDriver::"},{"line_number":88,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"65f37975_ddf3d805","line":85,"range":{"start_line":85,"start_character":10,"end_line":85,"end_character":69},"in_reply_to":"c820056c_40526ea8","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"e589fdf9e9d0dbc3ea17a46ce3b62b05d2ecdefe","unresolved":true,"context_lines":[{"line_number":144,"context_line":"                      ▼"},{"line_number":145,"context_line":"    ┌─────────────────────────────────────────────────┐"},{"line_number":146,"context_line":"    │  Cyborg conductor (unbind):                     │"},{"line_number":147,"context_line":"    │  1. device.status \u003d \u0027maintaining\u0027  (DB)         │"},{"line_number":148,"context_line":"    │  2. PUT /resource_providers/{rp}/inventories    │"},{"line_number":149,"context_line":"    │     reserved\u003dtotal                              │"},{"line_number":150,"context_line":"    │  3. dispatch cleanup RPC to agent (async)       │"}],"source_content_type":"text/x-rst","patch_set":10,"id":"1bc61657_71714b05","line":147,"range":{"start_line":147,"start_character":27,"end_line":147,"end_character":38},"updated":"2026-05-20 15:35:40.000000000","message":"is \u0027maintaining\u0027 the right state to use ? just an open question but as a cyborg operator, I\u0027d prefer to know the exact disk state (eg. \u0027cleaning\u0027)","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":144,"context_line":"                      ▼"},{"line_number":145,"context_line":"    ┌─────────────────────────────────────────────────┐"},{"line_number":146,"context_line":"    │  Cyborg conductor (unbind):                     │"},{"line_number":147,"context_line":"    │  1. device.status \u003d \u0027maintaining\u0027  (DB)         │"},{"line_number":148,"context_line":"    │  2. PUT /resource_providers/{rp}/inventories    │"},{"line_number":149,"context_line":"    │     reserved\u003dtotal                              │"},{"line_number":150,"context_line":"    │  3. dispatch cleanup RPC to agent (async)       │"}],"source_content_type":"text/x-rst","patch_set":10,"id":"d6bd26c3_d72fa55f","line":147,"range":{"start_line":147,"start_character":27,"end_line":147,"end_character":38},"in_reply_to":"1bc61657_71714b05","updated":"2026-06-03 16:06:13.000000000","message":"right this was feedback i belive i mention on irc \n\nthe device.status filed is currnlty used for enabling and disablity scudlign that feature i new and only partly implemented\n\nwe shoudl use cleaning but i do not want to reuse the exsitng filed \n\nlets add a new filed ot the device reosuce which si device_state that is an enum over avialable, allocated, cleaning and error states and keep the status field for the current enabeld/disabled status.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":144,"context_line":"                      ▼"},{"line_number":145,"context_line":"    ┌─────────────────────────────────────────────────┐"},{"line_number":146,"context_line":"    │  Cyborg conductor (unbind):                     │"},{"line_number":147,"context_line":"    │  1. device.status \u003d \u0027maintaining\u0027  (DB)         │"},{"line_number":148,"context_line":"    │  2. PUT /resource_providers/{rp}/inventories    │"},{"line_number":149,"context_line":"    │     reserved\u003dtotal                              │"},{"line_number":150,"context_line":"    │  3. dispatch cleanup RPC to agent (async)       │"}],"source_content_type":"text/x-rst","patch_set":10,"id":"50fdbe33_79710fdc","line":147,"range":{"start_line":147,"start_character":27,"end_line":147,"end_character":38},"in_reply_to":"d6bd26c3_d72fa55f","updated":"2026-06-17 04:45:31.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"e589fdf9e9d0dbc3ea17a46ce3b62b05d2ecdefe","unresolved":true,"context_lines":[{"line_number":146,"context_line":"    │  Cyborg conductor (unbind):                     │"},{"line_number":147,"context_line":"    │  1. device.status \u003d \u0027maintaining\u0027  (DB)         │"},{"line_number":148,"context_line":"    │  2. PUT /resource_providers/{rp}/inventories    │"},{"line_number":149,"context_line":"    │     reserved\u003dtotal                              │"},{"line_number":150,"context_line":"    │  3. dispatch cleanup RPC to agent (async)       │"},{"line_number":151,"context_line":"    └─────────────────┬───────────────────────────────┘"},{"line_number":152,"context_line":"                      │"}],"source_content_type":"text/x-rst","patch_set":10,"id":"eff88854_b2769e9e","line":149,"updated":"2026-05-20 15:35:40.000000000","message":"what happens if placement API returns a failure while device.status is on \u0027maintaining\u0027 ?","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":false,"context_lines":[{"line_number":146,"context_line":"    │  Cyborg conductor (unbind):                     │"},{"line_number":147,"context_line":"    │  1. device.status \u003d \u0027maintaining\u0027  (DB)         │"},{"line_number":148,"context_line":"    │  2. PUT /resource_providers/{rp}/inventories    │"},{"line_number":149,"context_line":"    │     reserved\u003dtotal                              │"},{"line_number":150,"context_line":"    │  3. dispatch cleanup RPC to agent (async)       │"},{"line_number":151,"context_line":"    └─────────────────┬───────────────────────────────┘"},{"line_number":152,"context_line":"                      │"}],"source_content_type":"text/x-rst","patch_set":10,"id":"6384c084_a25d3c59","line":149,"in_reply_to":"b0bf7eca_a5df7c26","updated":"2026-06-19 04:57:23.000000000","message":"Addressed in PS11. reserved\u003dtotal is set at bind time, and the bind path rejects binding if device_state is not available.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":146,"context_line":"    │  Cyborg conductor (unbind):                     │"},{"line_number":147,"context_line":"    │  1. device.status \u003d \u0027maintaining\u0027  (DB)         │"},{"line_number":148,"context_line":"    │  2. PUT /resource_providers/{rp}/inventories    │"},{"line_number":149,"context_line":"    │     reserved\u003dtotal                              │"},{"line_number":150,"context_line":"    │  3. dispatch cleanup RPC to agent (async)       │"},{"line_number":151,"context_line":"    └─────────────────┬───────────────────────────────┘"},{"line_number":152,"context_line":"                      │"}],"source_content_type":"text/x-rst","patch_set":10,"id":"b0bf7eca_a5df7c26","line":149,"in_reply_to":"eff88854_b2769e9e","updated":"2026-06-03 16:06:13.000000000","message":"on the cyborg side we can hae extra protection in the device bind path where we will fail the binding of a device if the device state is not in aviable.\n\nbut to your point we can retry on a generation conflcit but otherwise we will move the device state to error.\n\nhoweve ri want to set reserved\u003dtotal on device bind not on unbind\n\ni do no want the posibleity that we can allocate a device again once its bound without proprly cleaning so i want use to mark it with reserved\u003dtotoal at teh point of allcoation not deallcaotion.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":154,"context_line":"               instance deletion is complete)"},{"line_number":155,"context_line":"                      │"},{"line_number":156,"context_line":"                      ▼"},{"line_number":157,"context_line":"    ┌─────────────────────────────────────────────────┐"},{"line_number":158,"context_line":"    │  Cyborg agent:                                  │"},{"line_number":159,"context_line":"    │  resolve /dev/nvmeN via sysfs                   │"},{"line_number":160,"context_line":"    │  cleanup_nvme_privileged() runs (privsep)       │"},{"line_number":161,"context_line":"    │                                                 │"},{"line_number":162,"context_line":"    │  nvme id-ctrl → check sanicap bits 0-2           │"},{"line_number":163,"context_line":"    │  ┌─ bit 0 (Crypto Erase) ──► nvme sanitize -a 0x04  │"},{"line_number":164,"context_line":"    │  ├─ bit 1 (Block Erase) ──► nvme sanitize -a 0x02  │"},{"line_number":165,"context_line":"    │  ├─ bit 2 (Overwrite)   ──► nvme sanitize -a 0x03  │"},{"line_number":166,"context_line":"    │  ├─ ses\u003d2 support ─► nvme format -f --ses\u003d2     │"},{"line_number":167,"context_line":"    │  └─ fallback ──────► nvme format -f --ses\u003d1     │"},{"line_number":168,"context_line":"    │                                                 │"},{"line_number":169,"context_line":"    │  Bounded by [nvme] cleanup_timeout              │"},{"line_number":170,"context_line":"    └──────────┬──────────────────────┬───────────────┘"},{"line_number":171,"context_line":"               │ SUCCESS              │ FAILURE / TIMEOUT"},{"line_number":172,"context_line":"               ▼                      ▼"},{"line_number":173,"context_line":"    ┌─────────────────────┐  ┌────────────────────────────┐"},{"line_number":174,"context_line":"    │  reserved \u003d 0       │  │  reserved \u003d total          │"},{"line_number":175,"context_line":"    │  enable_device()    │  │  status → maintaining      │"},{"line_number":176,"context_line":"    │  status → enabled   │  │  cleanup_failed \u003d True     │"},{"line_number":177,"context_line":"    └─────────────────────┘  └──────────┬─────────────────┘"},{"line_number":178,"context_line":"                                        │ operator runs"},{"line_number":179,"context_line":"                                        │ cyborg-nvme-cleanup"},{"line_number":180,"context_line":"                                        ▼"},{"line_number":181,"context_line":"                             ┌────────────────────────────┐"},{"line_number":182,"context_line":"                             │  reserved\u003d0, status→enabled│"},{"line_number":183,"context_line":"                             │  cleanup_failed \u003d False    │"},{"line_number":184,"context_line":"                             └────────────────────────────┘"},{"line_number":185,"context_line":""},{"line_number":186,"context_line":"Scope"},{"line_number":187,"context_line":"-----"}],"source_content_type":"text/x-rst","patch_set":10,"id":"6944c739_55b85fb4","line":184,"range":{"start_line":157,"start_character":2,"end_line":184,"end_character":59},"updated":"2026-06-03 16:06:13.000000000","message":"we will need to update this\ni do not want to reuse any of the enable/disable infrastcure\n\nthat shoudl be contoleble indepentely of the device state so that an admin can disable a device so it will not be reallcoated when the current user delete there vm without impacting the device_state lifecycle flow.\n\nso status will remain only for enabeld/disabled so that admins can manage which device shoudl eb cons9omable for new worksload, and the device lifecycle will be tracted with a dedeicated device_state filed like nova vm_sate or task_sate.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":154,"context_line":"               instance deletion is complete)"},{"line_number":155,"context_line":"                      │"},{"line_number":156,"context_line":"                      ▼"},{"line_number":157,"context_line":"    ┌─────────────────────────────────────────────────┐"},{"line_number":158,"context_line":"    │  Cyborg agent:                                  │"},{"line_number":159,"context_line":"    │  resolve /dev/nvmeN via sysfs                   │"},{"line_number":160,"context_line":"    │  cleanup_nvme_privileged() runs (privsep)       │"},{"line_number":161,"context_line":"    │                                                 │"},{"line_number":162,"context_line":"    │  nvme id-ctrl → check sanicap bits 0-2           │"},{"line_number":163,"context_line":"    │  ┌─ bit 0 (Crypto Erase) ──► nvme sanitize -a 0x04  │"},{"line_number":164,"context_line":"    │  ├─ bit 1 (Block Erase) ──► nvme sanitize -a 0x02  │"},{"line_number":165,"context_line":"    │  ├─ bit 2 (Overwrite)   ──► nvme sanitize -a 0x03  │"},{"line_number":166,"context_line":"    │  ├─ ses\u003d2 support ─► nvme format -f --ses\u003d2     │"},{"line_number":167,"context_line":"    │  └─ fallback ──────► nvme format -f --ses\u003d1     │"},{"line_number":168,"context_line":"    │                                                 │"},{"line_number":169,"context_line":"    │  Bounded by [nvme] cleanup_timeout              │"},{"line_number":170,"context_line":"    └──────────┬──────────────────────┬───────────────┘"},{"line_number":171,"context_line":"               │ SUCCESS              │ FAILURE / TIMEOUT"},{"line_number":172,"context_line":"               ▼                      ▼"},{"line_number":173,"context_line":"    ┌─────────────────────┐  ┌────────────────────────────┐"},{"line_number":174,"context_line":"    │  reserved \u003d 0       │  │  reserved \u003d total          │"},{"line_number":175,"context_line":"    │  enable_device()    │  │  status → maintaining      │"},{"line_number":176,"context_line":"    │  status → enabled   │  │  cleanup_failed \u003d True     │"},{"line_number":177,"context_line":"    └─────────────────────┘  └──────────┬─────────────────┘"},{"line_number":178,"context_line":"                                        │ operator runs"},{"line_number":179,"context_line":"                                        │ cyborg-nvme-cleanup"},{"line_number":180,"context_line":"                                        ▼"},{"line_number":181,"context_line":"                             ┌────────────────────────────┐"},{"line_number":182,"context_line":"                             │  reserved\u003d0, status→enabled│"},{"line_number":183,"context_line":"                             │  cleanup_failed \u003d False    │"},{"line_number":184,"context_line":"                             └────────────────────────────┘"},{"line_number":185,"context_line":""},{"line_number":186,"context_line":"Scope"},{"line_number":187,"context_line":"-----"}],"source_content_type":"text/x-rst","patch_set":10,"id":"c345e4f4_e20ec407","line":184,"range":{"start_line":157,"start_character":2,"end_line":184,"end_character":59},"in_reply_to":"6944c739_55b85fb4","updated":"2026-06-17 04:45:31.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":190,"context_line":""},{"line_number":191,"context_line":"Out of scope:"},{"line_number":192,"context_line":""},{"line_number":193,"context_line":"* NVMe-oF/TCP/RDMA (fabric-attached storage)"},{"line_number":194,"context_line":"* Device content snapshot or restore"},{"line_number":195,"context_line":"* Instance resize and migration for instances with Cyborg-managed devices"},{"line_number":196,"context_line":"  (Cyborg devices are stateful and Cyborg does not provide data migration"}],"source_content_type":"text/x-rst","patch_set":10,"id":"d595ef02_53d4bd34","line":193,"updated":"2026-06-03 16:06:13.000000000","message":"this would be its own driver so yes this si correct to put out of scope for now.\nnote that cider also has nvme-of capabliteis so while we could provie an nvme-of driver in the future its not nessiarly needed given cinder can do that.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":190,"context_line":""},{"line_number":191,"context_line":"Out of scope:"},{"line_number":192,"context_line":""},{"line_number":193,"context_line":"* NVMe-oF/TCP/RDMA (fabric-attached storage)"},{"line_number":194,"context_line":"* Device content snapshot or restore"},{"line_number":195,"context_line":"* Instance resize and migration for instances with Cyborg-managed devices"},{"line_number":196,"context_line":"  (Cyborg devices are stateful and Cyborg does not provide data migration"}],"source_content_type":"text/x-rst","patch_set":10,"id":"142cab72_d261e0c4","line":193,"in_reply_to":"d595ef02_53d4bd34","updated":"2026-06-17 04:45:31.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":192,"context_line":""},{"line_number":193,"context_line":"* NVMe-oF/TCP/RDMA (fabric-attached storage)"},{"line_number":194,"context_line":"* Device content snapshot or restore"},{"line_number":195,"context_line":"* Instance resize and migration for instances with Cyborg-managed devices"},{"line_number":196,"context_line":"  (Cyborg devices are stateful and Cyborg does not provide data migration"},{"line_number":197,"context_line":"  capabilities. This is a general Cyborg limitation affecting all stateful"},{"line_number":198,"context_line":"  device types including NVMe, FPGA with bitstreams, etc.)"},{"line_number":199,"context_line":""},{"line_number":200,"context_line":""},{"line_number":201,"context_line":"Device Discovery"}],"source_content_type":"text/x-rst","patch_set":10,"id":"f60869e1_5915286b","line":198,"range":{"start_line":195,"start_character":0,"end_line":198,"end_character":58},"updated":"2026-06-03 16:06:13.000000000","message":"we can simplfy this for not, we do not suprpot resize of migration for any vm with cybrog devices today. we will look to add that next cycle but when we do that we can dicuss if we supprot state transefr or not and the relevent meanchics","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":192,"context_line":""},{"line_number":193,"context_line":"* NVMe-oF/TCP/RDMA (fabric-attached storage)"},{"line_number":194,"context_line":"* Device content snapshot or restore"},{"line_number":195,"context_line":"* Instance resize and migration for instances with Cyborg-managed devices"},{"line_number":196,"context_line":"  (Cyborg devices are stateful and Cyborg does not provide data migration"},{"line_number":197,"context_line":"  capabilities. This is a general Cyborg limitation affecting all stateful"},{"line_number":198,"context_line":"  device types including NVMe, FPGA with bitstreams, etc.)"},{"line_number":199,"context_line":""},{"line_number":200,"context_line":""},{"line_number":201,"context_line":"Device Discovery"}],"source_content_type":"text/x-rst","patch_set":10,"id":"43a78d8e_674df31a","line":198,"range":{"start_line":195,"start_character":0,"end_line":198,"end_character":58},"in_reply_to":"f60869e1_5915286b","updated":"2026-06-17 04:45:31.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"e589fdf9e9d0dbc3ea17a46ce3b62b05d2ecdefe","unresolved":false,"context_lines":[{"line_number":210,"context_line":"    [nvme]"},{"line_number":211,"context_line":"    device_spec \u003d {\"vendor_id\": \"8086\", \"product_id\": \"0001\"}   # vendor/product"},{"line_number":212,"context_line":"    # OR"},{"line_number":213,"context_line":"    device_spec \u003d {\"address\": \"*:0a:00.*\"}                      # PCI address glob"},{"line_number":214,"context_line":""},{"line_number":215,"context_line":"Discovery reads ``/sys/bus/pci/devices/`` and filters based on device_spec."},{"line_number":216,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"3ea7527d_ae92564d","line":213,"updated":"2026-05-20 15:35:40.000000000","message":"should the device_spec mutually exclusive from the [pci] device_spec list ? \nI hope so, or operators wouldn\u0027t know whether cyborg uses PCI or NVME driver...\n\n(later) oh sec, just saw the next paragraph 😅","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":false,"context_lines":[{"line_number":210,"context_line":"    [nvme]"},{"line_number":211,"context_line":"    device_spec \u003d {\"vendor_id\": \"8086\", \"product_id\": \"0001\"}   # vendor/product"},{"line_number":212,"context_line":"    # OR"},{"line_number":213,"context_line":"    device_spec \u003d {\"address\": \"*:0a:00.*\"}                      # PCI address glob"},{"line_number":214,"context_line":""},{"line_number":215,"context_line":"Discovery reads ``/sys/bus/pci/devices/`` and filters based on device_spec."},{"line_number":216,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"7932be56_879657e8","line":213,"in_reply_to":"3ea7527d_ae92564d","updated":"2026-06-03 16:06:13.000000000","message":"for any specific device yes but i dont belive nova or cybrog shoudl try an enfoce that.\n\nwe just need to docuemnt that clearly but any one device shoudl be manage by at most one of nova or cybrog. anything eles is operator error.\n\ncybrog cant gurad agains this and i would prefer to not complicate nova byt adding addtional check in nova for this.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"021287ad55d98a8a0e4369ab96195563565e4e5b","unresolved":true,"context_lines":[{"line_number":217,"context_line":"Driver Conflict Prevention"},{"line_number":218,"context_line":"---------------------------"},{"line_number":219,"context_line":""},{"line_number":220,"context_line":"Agent startup validates that generic and vendor NVMe drivers are not both"},{"line_number":221,"context_line":"enabled (raises ``InvalidConfiguration``). For mixed deployments, separate"},{"line_number":222,"context_line":"device scopes using ``device_spec`` filtering (e.g., vendor-specific driver"},{"line_number":223,"context_line":"gets ``vendor_id\u003d1bd4``, generic gets ``vendor_id\u003d!1bd4``)."}],"source_content_type":"text/x-rst","patch_set":10,"id":"88a7b95c_af0b05b5","line":220,"updated":"2026-06-02 10:16:48.000000000","message":"isn\u0027t this first sentence and the next one contradicting each other? The way I understand it, the first one says that an operator will not be able to enable the generic nvme driver and a vendor one at the same time, but then it talks about mixed deployments. I think there can be use cases where we could want to have devices discovered by the generic driver and others by vendor drivers","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":217,"context_line":"Driver Conflict Prevention"},{"line_number":218,"context_line":"---------------------------"},{"line_number":219,"context_line":""},{"line_number":220,"context_line":"Agent startup validates that generic and vendor NVMe drivers are not both"},{"line_number":221,"context_line":"enabled (raises ``InvalidConfiguration``). For mixed deployments, separate"},{"line_number":222,"context_line":"device scopes using ``device_spec`` filtering (e.g., vendor-specific driver"},{"line_number":223,"context_line":"gets ``vendor_id\u003d1bd4``, generic gets ``vendor_id\u003d!1bd4``)."}],"source_content_type":"text/x-rst","patch_set":10,"id":"2b97cd71_b93a9bd2","line":220,"in_reply_to":"2584efd5_43fe3ba3","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":217,"context_line":"Driver Conflict Prevention"},{"line_number":218,"context_line":"---------------------------"},{"line_number":219,"context_line":""},{"line_number":220,"context_line":"Agent startup validates that generic and vendor NVMe drivers are not both"},{"line_number":221,"context_line":"enabled (raises ``InvalidConfiguration``). For mixed deployments, separate"},{"line_number":222,"context_line":"device scopes using ``device_spec`` filtering (e.g., vendor-specific driver"},{"line_number":223,"context_line":"gets ``vendor_id\u003d1bd4``, generic gets ``vendor_id\u003d!1bd4``)."}],"source_content_type":"text/x-rst","patch_set":10,"id":"2584efd5_43fe3ba3","line":220,"in_reply_to":"88a7b95c_af0b05b5","updated":"2026-06-03 16:06:13.000000000","message":"we dont actully need to prevent this\n\nas long as you dont use the genric nvme dirver to manage the inspr device this just add complexity for no benift\n\nisnstead what i woudl prefer to do is deprecate the inspru driver in this release and remvoe it in favor of the generic on in 2027.2","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":220,"context_line":"Agent startup validates that generic and vendor NVMe drivers are not both"},{"line_number":221,"context_line":"enabled (raises ``InvalidConfiguration``). For mixed deployments, separate"},{"line_number":222,"context_line":"device scopes using ``device_spec`` filtering (e.g., vendor-specific driver"},{"line_number":223,"context_line":"gets ``vendor_id\u003d1bd4``, generic gets ``vendor_id\u003d!1bd4``)."},{"line_number":224,"context_line":""},{"line_number":225,"context_line":"Placement Integration"},{"line_number":226,"context_line":"---------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"30f96c7f_23a018b3","line":223,"range":{"start_line":223,"start_character":38,"end_line":223,"end_character":57},"updated":"2026-06-03 16:06:13.000000000","message":"lets not suggest ``vendor_id\u003d!1bd4`` is valid syntax its not\n\nlets jus trepalce this entir paragrp with something like this\n\n```\nFor any given device managing it with 2 driver simulatinoutly is condier invalid.\nthat applies to all drivers including the existing insprur and pci drivers.\nAs part of this spec the cybrog agent will also be updated to fail to start up if more then 1 driver returns the same device with an invalide configurtion excpetion. \n\nThe existing ssd and inspur drivers will be deprecated in favor of the generic\nnvme driver. Both driver can be used on the same system if an only if they are managing indepent devices. The upgrade docs will be updated with example of how to manage the inspur device with the new generic nvme driver.\n\nAs nova does not currenlty supprot resizing instnaces with cyborg manged devices we will not remove the vendor sepcific driver until after that feature gap is adressed but no new devleopment will be done on the vendor specific version outside fo bug fixes. This will be comunciated clearly to operators and contibutors via the docs.\n\n```","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":220,"context_line":"Agent startup validates that generic and vendor NVMe drivers are not both"},{"line_number":221,"context_line":"enabled (raises ``InvalidConfiguration``). For mixed deployments, separate"},{"line_number":222,"context_line":"device scopes using ``device_spec`` filtering (e.g., vendor-specific driver"},{"line_number":223,"context_line":"gets ``vendor_id\u003d1bd4``, generic gets ``vendor_id\u003d!1bd4``)."},{"line_number":224,"context_line":""},{"line_number":225,"context_line":"Placement Integration"},{"line_number":226,"context_line":"---------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"23d20cc5_52bb15c8","line":223,"range":{"start_line":223,"start_character":38,"end_line":223,"end_character":57},"in_reply_to":"30f96c7f_23a018b3","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"e589fdf9e9d0dbc3ea17a46ce3b62b05d2ecdefe","unresolved":false,"context_lines":[{"line_number":221,"context_line":"enabled (raises ``InvalidConfiguration``). For mixed deployments, separate"},{"line_number":222,"context_line":"device scopes using ``device_spec`` filtering (e.g., vendor-specific driver"},{"line_number":223,"context_line":"gets ``vendor_id\u003d1bd4``, generic gets ``vendor_id\u003d!1bd4``)."},{"line_number":224,"context_line":""},{"line_number":225,"context_line":"Placement Integration"},{"line_number":226,"context_line":"---------------------"},{"line_number":227,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"7d9b6930_5e9c918d","line":224,"updated":"2026-05-20 15:35:40.000000000","message":"++","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":227,"context_line":""},{"line_number":228,"context_line":"Devices are marked ``type\u003dNVME`` for cleanup path routing. Placement traits"},{"line_number":229,"context_line":"follow VGPU pattern: ``OWNER_CYBORG`` and"},{"line_number":230,"context_line":"``CUSTOM_NVME_\u003cVENDOR\u003e_\u003cVENDOR_ID\u003e_\u003cPRODUCT_ID\u003e``. Resource provider and"},{"line_number":231,"context_line":"deployable names use format ``\u003chostname\u003e_\u003cpci_address\u003e`` (e.g.,"},{"line_number":232,"context_line":"``compute-1_0000:01:00.0``)."},{"line_number":233,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"e13d2418_02365f54","line":230,"range":{"start_line":230,"start_character":0,"end_line":230,"end_character":49},"updated":"2026-06-03 16:06:13.000000000","message":"no\nthis is not correct\nthe vgpu driver is incorrectly encoding the VENDOR id and Prodcut id as traits when tehy shoudl be encoded in the resouce class\n\n`CUSTOM_NVME_\u003cVENDOR_ID\u003e_\u003cPRODUCT_ID\u003e`.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":227,"context_line":""},{"line_number":228,"context_line":"Devices are marked ``type\u003dNVME`` for cleanup path routing. Placement traits"},{"line_number":229,"context_line":"follow VGPU pattern: ``OWNER_CYBORG`` and"},{"line_number":230,"context_line":"``CUSTOM_NVME_\u003cVENDOR\u003e_\u003cVENDOR_ID\u003e_\u003cPRODUCT_ID\u003e``. Resource provider and"},{"line_number":231,"context_line":"deployable names use format ``\u003chostname\u003e_\u003cpci_address\u003e`` (e.g.,"},{"line_number":232,"context_line":"``compute-1_0000:01:00.0``)."},{"line_number":233,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"a3edb8e0_f5a7db0a","line":230,"range":{"start_line":230,"start_character":0,"end_line":230,"end_character":49},"in_reply_to":"e13d2418_02365f54","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":229,"context_line":"follow VGPU pattern: ``OWNER_CYBORG`` and"},{"line_number":230,"context_line":"``CUSTOM_NVME_\u003cVENDOR\u003e_\u003cVENDOR_ID\u003e_\u003cPRODUCT_ID\u003e``. Resource provider and"},{"line_number":231,"context_line":"deployable names use format ``\u003chostname\u003e_\u003cpci_address\u003e`` (e.g.,"},{"line_number":232,"context_line":"``compute-1_0000:01:00.0``)."},{"line_number":233,"context_line":""},{"line_number":234,"context_line":"Cleanup on Unbind"},{"line_number":235,"context_line":"-----------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"ef126b48_0dbac892","line":232,"range":{"start_line":232,"start_character":0,"end_line":232,"end_character":26},"updated":"2026-06-03 16:06:13.000000000","message":"yes this is the same convetion we use for pci device tracking in placment in nova\n\nthat will also ensure that a device cant be manage by both nova and cybrog as we will have a name collison if you try\n\none thing we shoud lcapture is this is the pci adress fo the nvme dvices phsyical funciton.\n\nwhile we are not supporting partionign the nvme device with namespace or VFS in this version if we were to do that we would jsut have an inventory of the VF/nameapces reousce grouped under the PF adrsss anyway so this is the correct nameing scheme to use\n\ncan we add a clarifying sentent to note this is the pf adress.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":229,"context_line":"follow VGPU pattern: ``OWNER_CYBORG`` and"},{"line_number":230,"context_line":"``CUSTOM_NVME_\u003cVENDOR\u003e_\u003cVENDOR_ID\u003e_\u003cPRODUCT_ID\u003e``. Resource provider and"},{"line_number":231,"context_line":"deployable names use format ``\u003chostname\u003e_\u003cpci_address\u003e`` (e.g.,"},{"line_number":232,"context_line":"``compute-1_0000:01:00.0``)."},{"line_number":233,"context_line":""},{"line_number":234,"context_line":"Cleanup on Unbind"},{"line_number":235,"context_line":"-----------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"3fc1f6b7_eabba132","line":232,"range":{"start_line":232,"start_character":0,"end_line":232,"end_character":26},"in_reply_to":"ef126b48_0dbac892","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":238,"context_line":"the MDEV pattern, but uses async RPC due to long cleanup duration (minutes"},{"line_number":239,"context_line":"to hours)."},{"line_number":240,"context_line":""},{"line_number":241,"context_line":"The conductor flow: (1) checks if ``device.type \u003d\u003d \u0027NVME\u0027`` in"},{"line_number":242,"context_line":"``_deallocate_attach_handle()``; (2) sets ``device.status \u003d \u0027maintaining\u0027``"},{"line_number":243,"context_line":"and Placement ``reserved\u003dtotal``; (3) dispatches"},{"line_number":244,"context_line":"``cctxt.cast(\u0027cleanup_nvme_device\u0027, ...)`` fire-and-forget; (4) returns"},{"line_number":245,"context_line":"immediately so Nova unbind completes without blocking."},{"line_number":246,"context_line":""},{"line_number":247,"context_line":"Agent execution runs under privsep with oslo.concurrency locking, using a"},{"line_number":248,"context_line":"five-tier cleanup strategy based on NVMe controller capabilities (sanicap"}],"source_content_type":"text/x-rst","patch_set":10,"id":"d73b2240_f7b17447","line":245,"range":{"start_line":241,"start_character":0,"end_line":245,"end_character":54},"updated":"2026-06-03 16:06:13.000000000","message":"so couple of comments\n\nfirst we shoudl reformat this so it more readble\n\nsecond as part of the CVE i actully change how unbidn work t o remove the prc to th e conductor as no device has any action on unbind previously that required it.\n\nso we will have to restore that flow \n\nthe qustion i hae si shodl we alwasy do that or make it condtionl on the type.\n\nif we were to make ti condtional on the device beign unbond i would prefer not to do it on type but rather have the concept fo device capablity\n\n`supprots_cleaning`, `supprots programming`, `supprots_live_migration`\n\nthrid as noted before the reseved\u003dtotal shoudl be doen as part of the bind not as part fo unbind, im undecied if the transtion of the device_state shoudl be done via conducotr ot the cybrog agent as part of the cast, i think the agent woudl be more corect but in that case the conductor should likely mvoe it form `allocated` to `pending_cleaning`.\n\nso the lifecycle is \n\navailable -\u003e allocated -\u003e pending_cleaning -\u003e cleaning and then either available or error.\n\npending_cleaning allow use later to have an api to allow ues to request cleaning again form the error state via the same flow, the api will call the conductor, it will revert the state to the pending_cleaning sate and dispatch the async cast to the cyborg agent to process.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":238,"context_line":"the MDEV pattern, but uses async RPC due to long cleanup duration (minutes"},{"line_number":239,"context_line":"to hours)."},{"line_number":240,"context_line":""},{"line_number":241,"context_line":"The conductor flow: (1) checks if ``device.type \u003d\u003d \u0027NVME\u0027`` in"},{"line_number":242,"context_line":"``_deallocate_attach_handle()``; (2) sets ``device.status \u003d \u0027maintaining\u0027``"},{"line_number":243,"context_line":"and Placement ``reserved\u003dtotal``; (3) dispatches"},{"line_number":244,"context_line":"``cctxt.cast(\u0027cleanup_nvme_device\u0027, ...)`` fire-and-forget; (4) returns"},{"line_number":245,"context_line":"immediately so Nova unbind completes without blocking."},{"line_number":246,"context_line":""},{"line_number":247,"context_line":"Agent execution runs under privsep with oslo.concurrency locking, using a"},{"line_number":248,"context_line":"five-tier cleanup strategy based on NVMe controller capabilities (sanicap"}],"source_content_type":"text/x-rst","patch_set":10,"id":"d03a9d8f_135b8330","line":245,"range":{"start_line":241,"start_character":0,"end_line":245,"end_character":54},"in_reply_to":"d73b2240_f7b17447","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":244,"context_line":"``cctxt.cast(\u0027cleanup_nvme_device\u0027, ...)`` fire-and-forget; (4) returns"},{"line_number":245,"context_line":"immediately so Nova unbind completes without blocking."},{"line_number":246,"context_line":""},{"line_number":247,"context_line":"Agent execution runs under privsep with oslo.concurrency locking, using a"},{"line_number":248,"context_line":"five-tier cleanup strategy based on NVMe controller capabilities (sanicap"},{"line_number":249,"context_line":"bits):"},{"line_number":250,"context_line":""},{"line_number":251,"context_line":"1. CES (Crypto Erase) - ``nvme sanitize -a 0x04``"},{"line_number":252,"context_line":"2. BES (Block Erase) - ``nvme sanitize -a 0x02``"},{"line_number":253,"context_line":"3. OWS (Overwrite) - ``nvme sanitize -a 0x03``"},{"line_number":254,"context_line":"4. Format SES\u003d2 - ``nvme format -f --ses\u003d2``"},{"line_number":255,"context_line":"5. Format SES\u003d1 - ``nvme format -f --ses\u003d1`` (fallback)"},{"line_number":256,"context_line":""},{"line_number":257,"context_line":"Sanitize commands use ``--output-format\u003djson`` for stable parsing. Agent"},{"line_number":258,"context_line":"periodic task (60s interval) reads ``nvme sanitize-log`` once per cycle for"}],"source_content_type":"text/x-rst","patch_set":10,"id":"ffe7ffab_a49d5422","line":255,"range":{"start_line":247,"start_character":1,"end_line":255,"end_character":55},"updated":"2026-06-03 16:06:13.000000000","message":"so we can support all of the above, but we should discover what is reported when we are enumarting the devices and report this vai traits to placment\ninternally we can can record the cleaning type using the atirbutes api on the device.\n\nat runtime we shoudl not try multipel cleanign approch and just put the device in error if the expected one does not work.\n\nwe do not want to silently fall back to a weaker or more write intensive approach without knowing about it\n\nwe shoudl create standard tratis by creating a new nvme folder under the HW namespace\n\nhttps://github.com/openstack/os-traits/tree/master/os_traits/hw\n\nand then create 5 new tratis in an __init__.py like this\n\nhttps://github.com/openstack/os-traits/blob/master/os_traits/hw/pci/__init__.py\n\n```\nTRAITS \u003d [\n    \u0027CES\u0027,  # crypto erase\n    \u0027BES\u0027,  # block erase\n    ...\n]\n```","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":244,"context_line":"``cctxt.cast(\u0027cleanup_nvme_device\u0027, ...)`` fire-and-forget; (4) returns"},{"line_number":245,"context_line":"immediately so Nova unbind completes without blocking."},{"line_number":246,"context_line":""},{"line_number":247,"context_line":"Agent execution runs under privsep with oslo.concurrency locking, using a"},{"line_number":248,"context_line":"five-tier cleanup strategy based on NVMe controller capabilities (sanicap"},{"line_number":249,"context_line":"bits):"},{"line_number":250,"context_line":""},{"line_number":251,"context_line":"1. CES (Crypto Erase) - ``nvme sanitize -a 0x04``"},{"line_number":252,"context_line":"2. BES (Block Erase) - ``nvme sanitize -a 0x02``"},{"line_number":253,"context_line":"3. OWS (Overwrite) - ``nvme sanitize -a 0x03``"},{"line_number":254,"context_line":"4. Format SES\u003d2 - ``nvme format -f --ses\u003d2``"},{"line_number":255,"context_line":"5. Format SES\u003d1 - ``nvme format -f --ses\u003d1`` (fallback)"},{"line_number":256,"context_line":""},{"line_number":257,"context_line":"Sanitize commands use ``--output-format\u003djson`` for stable parsing. Agent"},{"line_number":258,"context_line":"periodic task (60s interval) reads ``nvme sanitize-log`` once per cycle for"}],"source_content_type":"text/x-rst","patch_set":10,"id":"533d9cb2_fef2b607","line":255,"range":{"start_line":247,"start_character":1,"end_line":255,"end_character":55},"in_reply_to":"ffe7ffab_a49d5422","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":254,"context_line":"4. Format SES\u003d2 - ``nvme format -f --ses\u003d2``"},{"line_number":255,"context_line":"5. Format SES\u003d1 - ``nvme format -f --ses\u003d1`` (fallback)"},{"line_number":256,"context_line":""},{"line_number":257,"context_line":"Sanitize commands use ``--output-format\u003djson`` for stable parsing. Agent"},{"line_number":258,"context_line":"periodic task (60s interval) reads ``nvme sanitize-log`` once per cycle for"},{"line_number":259,"context_line":"devices in ``_active_cleanups`` (in-memory dict). On completion/failure, the"},{"line_number":260,"context_line":"agent calls conductor RPC to update device status."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":".. warning::"}],"source_content_type":"text/x-rst","patch_set":10,"id":"42f710ed_ae064bbe","line":259,"range":{"start_line":257,"start_character":67,"end_line":259,"end_character":49},"updated":"2026-06-03 16:06:13.000000000","message":"this i am very much not ok with\n\nthe cybrog agent shoudl use a futruist thread pools to dispatch the cleanign in the background and block on the result. i dont want use to have any perosic monierting of pasign fo the sanitze log\n\non compute agent startup we can either restart cleaning on any devices in cleaing or put them in error but i dont like the idea of useign perodic as part of the cleaning process.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":254,"context_line":"4. Format SES\u003d2 - ``nvme format -f --ses\u003d2``"},{"line_number":255,"context_line":"5. Format SES\u003d1 - ``nvme format -f --ses\u003d1`` (fallback)"},{"line_number":256,"context_line":""},{"line_number":257,"context_line":"Sanitize commands use ``--output-format\u003djson`` for stable parsing. Agent"},{"line_number":258,"context_line":"periodic task (60s interval) reads ``nvme sanitize-log`` once per cycle for"},{"line_number":259,"context_line":"devices in ``_active_cleanups`` (in-memory dict). On completion/failure, the"},{"line_number":260,"context_line":"agent calls conductor RPC to update device status."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":".. warning::"}],"source_content_type":"text/x-rst","patch_set":10,"id":"f6ea8e24_f76cb506","line":259,"range":{"start_line":257,"start_character":67,"end_line":259,"end_character":49},"in_reply_to":"42f710ed_ae064bbe","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":258,"context_line":"periodic task (60s interval) reads ``nvme sanitize-log`` once per cycle for"},{"line_number":259,"context_line":"devices in ``_active_cleanups`` (in-memory dict). On completion/failure, the"},{"line_number":260,"context_line":"agent calls conductor RPC to update device status."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":".. warning::"},{"line_number":263,"context_line":"   ``nvme format`` (``--ses\u003d2`` and ``--ses\u003d1``) erases only **currently"},{"line_number":264,"context_line":"   active namespaces**. Namespaces deleted by the tenant before instance"},{"line_number":265,"context_line":"   removal are **not** covered. All three ``nvme sanitize`` variants perform"},{"line_number":266,"context_line":"   a controller-wide erase including deleted namespaces. Operators with"},{"line_number":267,"context_line":"   strict data sanitisation requirements should restrict ``[nvme]"},{"line_number":268,"context_line":"   device_spec`` to hardware that reports at least one ``sanicap`` bit."},{"line_number":269,"context_line":""},{"line_number":270,"context_line":"Device Status State Machine"},{"line_number":271,"context_line":"----------------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"f9ed6a4a_c73ab7ba","line":268,"range":{"start_line":261,"start_character":1,"end_line":268,"end_character":71},"updated":"2026-06-03 16:06:13.000000000","message":"i wonder if instead of that wether we whis skip supproting this and isntead have the final fallback be calling shred or dd on teh device with zeros or soemthign else.\n\nthis may be vaild but i think we can provide a beter fallbac then using ses\u003d 1 or 2","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b65481c2244c0e91192130e709bbffab6db0e09f","unresolved":false,"context_lines":[{"line_number":258,"context_line":"periodic task (60s interval) reads ``nvme sanitize-log`` once per cycle for"},{"line_number":259,"context_line":"devices in ``_active_cleanups`` (in-memory dict). On completion/failure, the"},{"line_number":260,"context_line":"agent calls conductor RPC to update device status."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":".. warning::"},{"line_number":263,"context_line":"   ``nvme format`` (``--ses\u003d2`` and ``--ses\u003d1``) erases only **currently"},{"line_number":264,"context_line":"   active namespaces**. Namespaces deleted by the tenant before instance"},{"line_number":265,"context_line":"   removal are **not** covered. All three ``nvme sanitize`` variants perform"},{"line_number":266,"context_line":"   a controller-wide erase including deleted namespaces. Operators with"},{"line_number":267,"context_line":"   strict data sanitisation requirements should restrict ``[nvme]"},{"line_number":268,"context_line":"   device_spec`` to hardware that reports at least one ``sanicap`` bit."},{"line_number":269,"context_line":""},{"line_number":270,"context_line":"Device Status State Machine"},{"line_number":271,"context_line":"----------------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"7b83847e_3f02ce81","line":268,"range":{"start_line":261,"start_character":1,"end_line":268,"end_character":71},"in_reply_to":"95bcda45_7a0566ed","updated":"2026-06-24 06:03:25.000000000","message":"This is now written to support CES, BES, WZS and shread as a fallback mechanism. It also adds clear_mode and clear_method to tune cleaning.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"e088b55bdaf9f17c7baeb3bcfb4a7c877757b0c9","unresolved":true,"context_lines":[{"line_number":258,"context_line":"periodic task (60s interval) reads ``nvme sanitize-log`` once per cycle for"},{"line_number":259,"context_line":"devices in ``_active_cleanups`` (in-memory dict). On completion/failure, the"},{"line_number":260,"context_line":"agent calls conductor RPC to update device status."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":".. warning::"},{"line_number":263,"context_line":"   ``nvme format`` (``--ses\u003d2`` and ``--ses\u003d1``) erases only **currently"},{"line_number":264,"context_line":"   active namespaces**. Namespaces deleted by the tenant before instance"},{"line_number":265,"context_line":"   removal are **not** covered. All three ``nvme sanitize`` variants perform"},{"line_number":266,"context_line":"   a controller-wide erase including deleted namespaces. Operators with"},{"line_number":267,"context_line":"   strict data sanitisation requirements should restrict ``[nvme]"},{"line_number":268,"context_line":"   device_spec`` to hardware that reports at least one ``sanicap`` bit."},{"line_number":269,"context_line":""},{"line_number":270,"context_line":"Device Status State Machine"},{"line_number":271,"context_line":"----------------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"95bcda45_7a0566ed","line":268,"range":{"start_line":261,"start_character":1,"end_line":268,"end_character":71},"in_reply_to":"bb632a39_877ce3f0","updated":"2026-06-22 06:12:16.000000000","message":"Update: I have updated the spec to follow following cleanup chain:\nsanitize (CES → BES → OWS) → format ses\u003d2 and then move it to error state and let, operator decide.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"121686438d032edfd72c2ed7d5bb48c42e1b9c87","unresolved":true,"context_lines":[{"line_number":258,"context_line":"periodic task (60s interval) reads ``nvme sanitize-log`` once per cycle for"},{"line_number":259,"context_line":"devices in ``_active_cleanups`` (in-memory dict). On completion/failure, the"},{"line_number":260,"context_line":"agent calls conductor RPC to update device status."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":".. warning::"},{"line_number":263,"context_line":"   ``nvme format`` (``--ses\u003d2`` and ``--ses\u003d1``) erases only **currently"},{"line_number":264,"context_line":"   active namespaces**. Namespaces deleted by the tenant before instance"},{"line_number":265,"context_line":"   removal are **not** covered. All three ``nvme sanitize`` variants perform"},{"line_number":266,"context_line":"   a controller-wide erase including deleted namespaces. Operators with"},{"line_number":267,"context_line":"   strict data sanitisation requirements should restrict ``[nvme]"},{"line_number":268,"context_line":"   device_spec`` to hardware that reports at least one ``sanicap`` bit."},{"line_number":269,"context_line":""},{"line_number":270,"context_line":"Device Status State Machine"},{"line_number":271,"context_line":"----------------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"bb632a39_877ce3f0","line":268,"range":{"start_line":261,"start_character":1,"end_line":268,"end_character":71},"in_reply_to":"becc051b_c4643f1a","updated":"2026-06-19 05:05:03.000000000","message":"sorry commented at wrong place.\n\nI am in favor of dropping ses\u003d1 in favor of shred/dd. \n\nI am thinking about keeping nvme format ses\u003d2 (as it is available in devstack nvme vm) because it\u0027s a real\n  crypto erase when the device supports it (fna bit 2 \u003d 1), and some devices support format crypto erase but not sanitize. \n  \nThe cleanup chain is now: sanitize (CES → BES → OWS) → format ses\u003d2 (if supported) → shred/dd (opt-in via [nvme] best_effort_erase) → error.\n  \nLet me know your thoughts on this.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":true,"context_lines":[{"line_number":258,"context_line":"periodic task (60s interval) reads ``nvme sanitize-log`` once per cycle for"},{"line_number":259,"context_line":"devices in ``_active_cleanups`` (in-memory dict). On completion/failure, the"},{"line_number":260,"context_line":"agent calls conductor RPC to update device status."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":".. warning::"},{"line_number":263,"context_line":"   ``nvme format`` (``--ses\u003d2`` and ``--ses\u003d1``) erases only **currently"},{"line_number":264,"context_line":"   active namespaces**. Namespaces deleted by the tenant before instance"},{"line_number":265,"context_line":"   removal are **not** covered. All three ``nvme sanitize`` variants perform"},{"line_number":266,"context_line":"   a controller-wide erase including deleted namespaces. Operators with"},{"line_number":267,"context_line":"   strict data sanitisation requirements should restrict ``[nvme]"},{"line_number":268,"context_line":"   device_spec`` to hardware that reports at least one ``sanicap`` bit."},{"line_number":269,"context_line":""},{"line_number":270,"context_line":"Device Status State Machine"},{"line_number":271,"context_line":"----------------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"becc051b_c4643f1a","line":268,"range":{"start_line":261,"start_character":1,"end_line":268,"end_character":71},"in_reply_to":"f9ed6a4a_c73ab7ba","updated":"2026-06-19 04:57:23.000000000","message":"Done,\n\nThe conductor only dispatches the RPC cast now, all device_state transitions during cleanup are managed by the agent. \n\nAdded periodic reconciliation in the agent to catch missed RPCs. \n\nPlacement reserved\u003dtotal at bind time prevents re-allocation during the window.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":267,"context_line":"   strict data sanitisation requirements should restrict ``[nvme]"},{"line_number":268,"context_line":"   device_spec`` to hardware that reports at least one ``sanicap`` bit."},{"line_number":269,"context_line":""},{"line_number":270,"context_line":"Device Status State Machine"},{"line_number":271,"context_line":"----------------------------"},{"line_number":272,"context_line":""},{"line_number":273,"context_line":"The NVMe device uses the existing ``status`` field::"},{"line_number":274,"context_line":""},{"line_number":275,"context_line":"    enabled     ──operator disable──►  maintaining"},{"line_number":276,"context_line":"    enabled     ──instance delete──►   maintaining (cleanup runs async)"},{"line_number":277,"context_line":"    maintaining ──cleanup success──►   enabled (via agent callback)"},{"line_number":278,"context_line":"    maintaining ──cleanup failure──►   maintaining (cleanup_failed\u003dTrue)"},{"line_number":279,"context_line":"    maintaining ──operator enable──►   enabled (blocked if cleanup_failed\u003dTrue)"},{"line_number":280,"context_line":""},{"line_number":281,"context_line":"No new ``cleaning`` status is added; devices move to ``maintaining`` when"},{"line_number":282,"context_line":"cleanup starts, matching existing operator disable behavior. The"},{"line_number":283,"context_line":"``cleanup_failed`` boolean flag distinguishes cleanup failures from"},{"line_number":284,"context_line":"operator-initiated maintenance. If ``cleanup_failed\u003dTrue``, ``POST"},{"line_number":285,"context_line":"/v2/devices/{uuid}/enable`` returns HTTP 409; operators must use"},{"line_number":286,"context_line":"``cyborg-nvme-cleanup`` CLI to retry. Cleanup runs asynchronously, with the"},{"line_number":287,"context_line":"device remaining in ``maintaining`` until completion or failure."},{"line_number":288,"context_line":""},{"line_number":289,"context_line":"Timeout, Reconciliation, and Crash Recovery"},{"line_number":290,"context_line":"--------------------------------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"bc8dc18c_1f44b0e5","line":287,"range":{"start_line":270,"start_character":0,"end_line":287,"end_character":64},"updated":"2026-06-03 16:06:13.000000000","message":"again this breaks the mental model of enabled/disables for schdulabel and device state so this will need to be rewriten based on my feedback above.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":267,"context_line":"   strict data sanitisation requirements should restrict ``[nvme]"},{"line_number":268,"context_line":"   device_spec`` to hardware that reports at least one ``sanicap`` bit."},{"line_number":269,"context_line":""},{"line_number":270,"context_line":"Device Status State Machine"},{"line_number":271,"context_line":"----------------------------"},{"line_number":272,"context_line":""},{"line_number":273,"context_line":"The NVMe device uses the existing ``status`` field::"},{"line_number":274,"context_line":""},{"line_number":275,"context_line":"    enabled     ──operator disable──►  maintaining"},{"line_number":276,"context_line":"    enabled     ──instance delete──►   maintaining (cleanup runs async)"},{"line_number":277,"context_line":"    maintaining ──cleanup success──►   enabled (via agent callback)"},{"line_number":278,"context_line":"    maintaining ──cleanup failure──►   maintaining (cleanup_failed\u003dTrue)"},{"line_number":279,"context_line":"    maintaining ──operator enable──►   enabled (blocked if cleanup_failed\u003dTrue)"},{"line_number":280,"context_line":""},{"line_number":281,"context_line":"No new ``cleaning`` status is added; devices move to ``maintaining`` when"},{"line_number":282,"context_line":"cleanup starts, matching existing operator disable behavior. The"},{"line_number":283,"context_line":"``cleanup_failed`` boolean flag distinguishes cleanup failures from"},{"line_number":284,"context_line":"operator-initiated maintenance. If ``cleanup_failed\u003dTrue``, ``POST"},{"line_number":285,"context_line":"/v2/devices/{uuid}/enable`` returns HTTP 409; operators must use"},{"line_number":286,"context_line":"``cyborg-nvme-cleanup`` CLI to retry. Cleanup runs asynchronously, with the"},{"line_number":287,"context_line":"device remaining in ``maintaining`` until completion or failure."},{"line_number":288,"context_line":""},{"line_number":289,"context_line":"Timeout, Reconciliation, and Crash Recovery"},{"line_number":290,"context_line":"--------------------------------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"7c9a71bd_83b8f30f","line":287,"range":{"start_line":270,"start_character":0,"end_line":287,"end_character":64},"in_reply_to":"bc8dc18c_1f44b0e5","updated":"2026-06-17 15:27:52.000000000","message":"this has been updated os ill resolve this comment and we can review the new version","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":289,"context_line":"Timeout, Reconciliation, and Crash Recovery"},{"line_number":290,"context_line":"--------------------------------------------"},{"line_number":291,"context_line":""},{"line_number":292,"context_line":"NVMe cleanup operations can take minutes to hours depending on drive size and"},{"line_number":293,"context_line":"sanitize method. Cleanup is bounded by a configurable timeout::"},{"line_number":294,"context_line":""},{"line_number":295,"context_line":"    [nvme]"}],"source_content_type":"text/x-rst","patch_set":10,"id":"3d1a2d61_b9a530af","line":292,"range":{"start_line":292,"start_character":33,"end_line":292,"end_character":49},"updated":"2026-06-03 16:06:13.000000000","message":"minutes yes hour is not likely.\neven for very very large nvme devicves i woudl not expect it to be hours","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":289,"context_line":"Timeout, Reconciliation, and Crash Recovery"},{"line_number":290,"context_line":"--------------------------------------------"},{"line_number":291,"context_line":""},{"line_number":292,"context_line":"NVMe cleanup operations can take minutes to hours depending on drive size and"},{"line_number":293,"context_line":"sanitize method. Cleanup is bounded by a configurable timeout::"},{"line_number":294,"context_line":""},{"line_number":295,"context_line":"    [nvme]"}],"source_content_type":"text/x-rst","patch_set":10,"id":"0e2e0df0_11355393","line":292,"range":{"start_line":292,"start_character":33,"end_line":292,"end_character":49},"in_reply_to":"3d1a2d61_b9a530af","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":293,"context_line":"sanitize method. Cleanup is bounded by a configurable timeout::"},{"line_number":294,"context_line":""},{"line_number":295,"context_line":"    [nvme]"},{"line_number":296,"context_line":"    cleanup_timeout \u003d 7200      # seconds (default: 2 hours)"},{"line_number":297,"context_line":""},{"line_number":298,"context_line":"The conductor runs a periodic reconciliation task (5 minute interval) that"},{"line_number":299,"context_line":"detects stuck cleanup operations:"}],"source_content_type":"text/x-rst","patch_set":10,"id":"e1a5f956_3c651c8e","line":296,"range":{"start_line":296,"start_character":22,"end_line":296,"end_character":26},"updated":"2026-06-03 16:06:13.000000000","message":"a better default woudl be 15 minutes\nits very unlikely that it woudl take longer then that","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":293,"context_line":"sanitize method. Cleanup is bounded by a configurable timeout::"},{"line_number":294,"context_line":""},{"line_number":295,"context_line":"    [nvme]"},{"line_number":296,"context_line":"    cleanup_timeout \u003d 7200      # seconds (default: 2 hours)"},{"line_number":297,"context_line":""},{"line_number":298,"context_line":"The conductor runs a periodic reconciliation task (5 minute interval) that"},{"line_number":299,"context_line":"detects stuck cleanup operations:"}],"source_content_type":"text/x-rst","patch_set":10,"id":"432d14d0_c76c6b19","line":296,"range":{"start_line":296,"start_character":22,"end_line":296,"end_character":26},"in_reply_to":"e1a5f956_3c651c8e","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":295,"context_line":"    [nvme]"},{"line_number":296,"context_line":"    cleanup_timeout \u003d 7200      # seconds (default: 2 hours)"},{"line_number":297,"context_line":""},{"line_number":298,"context_line":"The conductor runs a periodic reconciliation task (5 minute interval) that"},{"line_number":299,"context_line":"detects stuck cleanup operations:"},{"line_number":300,"context_line":""},{"line_number":301,"context_line":"1. Query devices with ``status\u003dmaintaining`` AND Placement ``reserved\u003dtotal``"},{"line_number":302,"context_line":"2. Calculate elapsed time: ``(current_time - device.updated_at)``"}],"source_content_type":"text/x-rst","patch_set":10,"id":"937770df_5f2ea24c","line":299,"range":{"start_line":298,"start_character":0,"end_line":299,"end_character":33},"updated":"2026-06-03 16:06:13.000000000","message":"im hard no on adding a perodic reconsiation task for this to the conductor and i would prefer not to have one on the comute agent. so -2 on that approch","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":295,"context_line":"    [nvme]"},{"line_number":296,"context_line":"    cleanup_timeout \u003d 7200      # seconds (default: 2 hours)"},{"line_number":297,"context_line":""},{"line_number":298,"context_line":"The conductor runs a periodic reconciliation task (5 minute interval) that"},{"line_number":299,"context_line":"detects stuck cleanup operations:"},{"line_number":300,"context_line":""},{"line_number":301,"context_line":"1. Query devices with ``status\u003dmaintaining`` AND Placement ``reserved\u003dtotal``"},{"line_number":302,"context_line":"2. Calculate elapsed time: ``(current_time - device.updated_at)``"}],"source_content_type":"text/x-rst","patch_set":10,"id":"60449bba_05a464cc","line":299,"range":{"start_line":298,"start_character":0,"end_line":299,"end_character":33},"in_reply_to":"937770df_5f2ea24c","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"021287ad55d98a8a0e4369ab96195563565e4e5b","unresolved":true,"context_lines":[{"line_number":298,"context_line":"The conductor runs a periodic reconciliation task (5 minute interval) that"},{"line_number":299,"context_line":"detects stuck cleanup operations:"},{"line_number":300,"context_line":""},{"line_number":301,"context_line":"1. Query devices with ``status\u003dmaintaining`` AND Placement ``reserved\u003dtotal``"},{"line_number":302,"context_line":"2. Calculate elapsed time: ``(current_time - device.updated_at)``"},{"line_number":303,"context_line":"3. If elapsed time exceeds ``cleanup_timeout``, mark ``cleanup_failed\u003dTrue``"},{"line_number":304,"context_line":"4. Log warning with device UUID for operator intervention"}],"source_content_type":"text/x-rst","patch_set":10,"id":"acdb3ae9_0bf88db3","line":301,"updated":"2026-06-02 10:16:48.000000000","message":"do we need to check the placement `reserved` state? Is there some case where a device is disabled but it\u0027s not totally reserved in placement?","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":true,"context_lines":[{"line_number":298,"context_line":"The conductor runs a periodic reconciliation task (5 minute interval) that"},{"line_number":299,"context_line":"detects stuck cleanup operations:"},{"line_number":300,"context_line":""},{"line_number":301,"context_line":"1. Query devices with ``status\u003dmaintaining`` AND Placement ``reserved\u003dtotal``"},{"line_number":302,"context_line":"2. Calculate elapsed time: ``(current_time - device.updated_at)``"},{"line_number":303,"context_line":"3. If elapsed time exceeds ``cleanup_timeout``, mark ``cleanup_failed\u003dTrue``"},{"line_number":304,"context_line":"4. Log warning with device UUID for operator intervention"}],"source_content_type":"text/x-rst","patch_set":10,"id":"9ed2e69d_a2050a1d","line":301,"in_reply_to":"9e5b07c3_f02b9ddf","updated":"2026-06-19 04:57:23.000000000","message":"Added a check if reserved\u003dtotal in Placement but device_state is available (or vice versa), \n\nthat means a bug in the bind or cleanup code path.\n\nThe agent logs a warning during init_host() so operators can investigate","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"31578484e635ad0071270e48a7a8fe51c36b9d69","unresolved":false,"context_lines":[{"line_number":298,"context_line":"The conductor runs a periodic reconciliation task (5 minute interval) that"},{"line_number":299,"context_line":"detects stuck cleanup operations:"},{"line_number":300,"context_line":""},{"line_number":301,"context_line":"1. Query devices with ``status\u003dmaintaining`` AND Placement ``reserved\u003dtotal``"},{"line_number":302,"context_line":"2. Calculate elapsed time: ``(current_time - device.updated_at)``"},{"line_number":303,"context_line":"3. If elapsed time exceeds ``cleanup_timeout``, mark ``cleanup_failed\u003dTrue``"},{"line_number":304,"context_line":"4. Log warning with device UUID for operator intervention"}],"source_content_type":"text/x-rst","patch_set":10,"id":"bdd3ae9f_a364898d","line":301,"in_reply_to":"9ed2e69d_a2050a1d","updated":"2026-06-26 10:18:08.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":true,"context_lines":[{"line_number":298,"context_line":"The conductor runs a periodic reconciliation task (5 minute interval) that"},{"line_number":299,"context_line":"detects stuck cleanup operations:"},{"line_number":300,"context_line":""},{"line_number":301,"context_line":"1. Query devices with ``status\u003dmaintaining`` AND Placement ``reserved\u003dtotal``"},{"line_number":302,"context_line":"2. Calculate elapsed time: ``(current_time - device.updated_at)``"},{"line_number":303,"context_line":"3. If elapsed time exceeds ``cleanup_timeout``, mark ``cleanup_failed\u003dTrue``"},{"line_number":304,"context_line":"4. Log warning with device UUID for operator intervention"}],"source_content_type":"text/x-rst","patch_set":10,"id":"9e5b07c3_f02b9ddf","line":301,"in_reply_to":"acdb3ae9_0bf88db3","updated":"2026-06-17 15:27:52.000000000","message":"there should not be. that would be a bug.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":298,"context_line":"The conductor runs a periodic reconciliation task (5 minute interval) that"},{"line_number":299,"context_line":"detects stuck cleanup operations:"},{"line_number":300,"context_line":""},{"line_number":301,"context_line":"1. Query devices with ``status\u003dmaintaining`` AND Placement ``reserved\u003dtotal``"},{"line_number":302,"context_line":"2. Calculate elapsed time: ``(current_time - device.updated_at)``"},{"line_number":303,"context_line":"3. If elapsed time exceeds ``cleanup_timeout``, mark ``cleanup_failed\u003dTrue``"},{"line_number":304,"context_line":"4. Log warning with device UUID for operator intervention"},{"line_number":305,"context_line":""},{"line_number":306,"context_line":"This design eliminates the need to persist ``cleanup_method`` or track"},{"line_number":307,"context_line":"in-progress operations across service restarts. The ``device.updated_at``"}],"source_content_type":"text/x-rst","patch_set":10,"id":"28f166be_00ae8097","line":304,"range":{"start_line":301,"start_character":1,"end_line":304,"end_character":57},"updated":"2026-06-03 16:06:13.000000000","message":"no this shoudl not be monitored by the conductor at all.\n\nwe shoudl have the cleaning task be run by the compute agnet via a thread pool\n\nthat will return a future that we can wait on with a time out and we can also pass a time out to the command when executing the nvme cli.\n\nthe call shoudl be lockign with noe async mondiring\n\nif the commen successd then the compute agent will set reserved\u003d0 and device_state aviable if it faile it will move it to error.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":298,"context_line":"The conductor runs a periodic reconciliation task (5 minute interval) that"},{"line_number":299,"context_line":"detects stuck cleanup operations:"},{"line_number":300,"context_line":""},{"line_number":301,"context_line":"1. Query devices with ``status\u003dmaintaining`` AND Placement ``reserved\u003dtotal``"},{"line_number":302,"context_line":"2. Calculate elapsed time: ``(current_time - device.updated_at)``"},{"line_number":303,"context_line":"3. If elapsed time exceeds ``cleanup_timeout``, mark ``cleanup_failed\u003dTrue``"},{"line_number":304,"context_line":"4. Log warning with device UUID for operator intervention"},{"line_number":305,"context_line":""},{"line_number":306,"context_line":"This design eliminates the need to persist ``cleanup_method`` or track"},{"line_number":307,"context_line":"in-progress operations across service restarts. The ``device.updated_at``"}],"source_content_type":"text/x-rst","patch_set":10,"id":"9168d833_162c9ea6","line":304,"range":{"start_line":301,"start_character":1,"end_line":304,"end_character":57},"in_reply_to":"28f166be_00ae8097","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":304,"context_line":"4. Log warning with device UUID for operator intervention"},{"line_number":305,"context_line":""},{"line_number":306,"context_line":"This design eliminates the need to persist ``cleanup_method`` or track"},{"line_number":307,"context_line":"in-progress operations across service restarts. The ``device.updated_at``"},{"line_number":308,"context_line":"timestamp serves as the single source of truth for operation start time."},{"line_number":309,"context_line":""},{"line_number":310,"context_line":"**Crash recovery:** Agent-side state (``_active_cleanups`` in-memory dict) is"}],"source_content_type":"text/x-rst","patch_set":10,"id":"9c98a603_6e005d64","line":307,"updated":"2026-06-03 16:06:13.000000000","message":"this design makes the conductor the bottle neck for all device cleaning monitoring and a signel poitn of fialre. it not horzontially scalabel and not a design we shoudl use.  all local device state management should happen withing the cybrog agent when it comes to cleaning\n\nthis is architsctully incorrect for cybrog","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"121686438d032edfd72c2ed7d5bb48c42e1b9c87","unresolved":false,"context_lines":[{"line_number":304,"context_line":"4. Log warning with device UUID for operator intervention"},{"line_number":305,"context_line":""},{"line_number":306,"context_line":"This design eliminates the need to persist ``cleanup_method`` or track"},{"line_number":307,"context_line":"in-progress operations across service restarts. The ``device.updated_at``"},{"line_number":308,"context_line":"timestamp serves as the single source of truth for operation start time."},{"line_number":309,"context_line":""},{"line_number":310,"context_line":"**Crash recovery:** Agent-side state (``_active_cleanups`` in-memory dict) is"}],"source_content_type":"text/x-rst","patch_set":10,"id":"934d815e_5278eab7","line":307,"in_reply_to":"9c98a603_6e005d64","updated":"2026-06-19 05:05:03.000000000","message":"Done,\n\nThe conductor only dispatches the RPC cast now, all device_state transitions during cleanup are managed by the agent.\n\nAdded periodic reconciliation in the agent to catch missed RPCs.\n\nPlacement reserved\u003dtotal at bind time prevents re-allocation during the window.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":310,"context_line":"**Crash recovery:** Agent-side state (``_active_cleanups`` in-memory dict) is"},{"line_number":311,"context_line":"cleared on agent startup. Devices mid-cleanup when the agent crashed are"},{"line_number":312,"context_line":"detected by conductor reconciliation via the same timeout mechanism—no special"},{"line_number":313,"context_line":"crash recovery logic is needed."},{"line_number":314,"context_line":""},{"line_number":315,"context_line":"**Concurrency control:** Per-device locking uses ``@utils.synchronized()``"},{"line_number":316,"context_line":"(oslo.concurrency) to prevent multiple concurrent cleanup attempts on the same"}],"source_content_type":"text/x-rst","patch_set":10,"id":"0a1a8082_bb63a6ec","line":313,"updated":"2026-06-03 16:06:13.000000000","message":"no we need to detecti this in teh agent on restart not via the conductor.\n\nwe can choose to etehr restart the cleaing or move it to error and allow an operator to restart it via the rest api.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"bdc24e7c58526707f7015bb40d75fc3c2d312416","unresolved":false,"context_lines":[{"line_number":310,"context_line":"**Crash recovery:** Agent-side state (``_active_cleanups`` in-memory dict) is"},{"line_number":311,"context_line":"cleared on agent startup. Devices mid-cleanup when the agent crashed are"},{"line_number":312,"context_line":"detected by conductor reconciliation via the same timeout mechanism—no special"},{"line_number":313,"context_line":"crash recovery logic is needed."},{"line_number":314,"context_line":""},{"line_number":315,"context_line":"**Concurrency control:** Per-device locking uses ``@utils.synchronized()``"},{"line_number":316,"context_line":"(oslo.concurrency) to prevent multiple concurrent cleanup attempts on the same"}],"source_content_type":"text/x-rst","patch_set":10,"id":"3eec57f5_1fbe055d","line":313,"in_reply_to":"0a1a8082_bb63a6ec","updated":"2026-06-19 05:10:08.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":314,"context_line":""},{"line_number":315,"context_line":"**Concurrency control:** Per-device locking uses ``@utils.synchronized()``"},{"line_number":316,"context_line":"(oslo.concurrency) to prevent multiple concurrent cleanup attempts on the same"},{"line_number":317,"context_line":"device."},{"line_number":318,"context_line":""},{"line_number":319,"context_line":"**Manual recovery:** Operators retry failed cleanups using"},{"line_number":320,"context_line":"``cyborg-nvme-cleanup --device \u003cuuid\u003e`` (requires admin credentials). The CLI"}],"source_content_type":"text/x-rst","patch_set":10,"id":"0dad5b73_5631cca3","line":317,"updated":"2026-06-03 16:06:13.000000000","message":"sure we can do that usitn the device uuid fro the lock key.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":314,"context_line":""},{"line_number":315,"context_line":"**Concurrency control:** Per-device locking uses ``@utils.synchronized()``"},{"line_number":316,"context_line":"(oslo.concurrency) to prevent multiple concurrent cleanup attempts on the same"},{"line_number":317,"context_line":"device."},{"line_number":318,"context_line":""},{"line_number":319,"context_line":"**Manual recovery:** Operators retry failed cleanups using"},{"line_number":320,"context_line":"``cyborg-nvme-cleanup --device \u003cuuid\u003e`` (requires admin credentials). The CLI"}],"source_content_type":"text/x-rst","patch_set":10,"id":"56721102_20473a33","line":317,"in_reply_to":"0dad5b73_5631cca3","updated":"2026-06-17 04:45:31.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":318,"context_line":""},{"line_number":319,"context_line":"**Manual recovery:** Operators retry failed cleanups using"},{"line_number":320,"context_line":"``cyborg-nvme-cleanup --device \u003cuuid\u003e`` (requires admin credentials). The CLI"},{"line_number":321,"context_line":"tool re-triggers cleanup and resets ``cleanup_failed\u003dFalse`` on success."},{"line_number":322,"context_line":""},{"line_number":323,"context_line":"Adding Vendor-specific Drivers"},{"line_number":324,"context_line":"-------------------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"c2483a8d_d795dc1e","line":321,"updated":"2026-06-03 16:06:13.000000000","message":"im -1 on anding any driver specic cli to cybrog\n\nmy perfered manually recoveray would be providign /device/\u003cuuid\u003e/clean as a admin only api which will trigger the cleaning again.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":318,"context_line":""},{"line_number":319,"context_line":"**Manual recovery:** Operators retry failed cleanups using"},{"line_number":320,"context_line":"``cyborg-nvme-cleanup --device \u003cuuid\u003e`` (requires admin credentials). The CLI"},{"line_number":321,"context_line":"tool re-triggers cleanup and resets ``cleanup_failed\u003dFalse`` on success."},{"line_number":322,"context_line":""},{"line_number":323,"context_line":"Adding Vendor-specific Drivers"},{"line_number":324,"context_line":"-------------------------------"}],"source_content_type":"text/x-rst","patch_set":10,"id":"2094fce9_afd22b21","line":321,"in_reply_to":"c2483a8d_d795dc1e","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":false,"context_lines":[{"line_number":332,"context_line":""},{"line_number":333,"context_line":"* Perform cleanup synchronously in the conductor during unbind — blocks Nova"},{"line_number":334,"context_line":"  instance deletion for minutes to hours, exceeding API timeout windows."},{"line_number":335,"context_line":"  *Rejected*"},{"line_number":336,"context_line":""},{"line_number":337,"context_line":"* Add new \"cleaning\" device status — ``maintaining`` already covers both"},{"line_number":338,"context_line":"  operator and system maintenance; ``cleanup_failed`` boolean suffices."}],"source_content_type":"text/x-rst","patch_set":10,"id":"8713872e_a8eb83d1","line":335,"updated":"2026-06-03 16:06:13.000000000","message":"if it was not for the timescale of cleaning this would be the correct approch\nbut unfortunely it does need to be async","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":334,"context_line":"  instance deletion for minutes to hours, exceeding API timeout windows."},{"line_number":335,"context_line":"  *Rejected*"},{"line_number":336,"context_line":""},{"line_number":337,"context_line":"* Add new \"cleaning\" device status — ``maintaining`` already covers both"},{"line_number":338,"context_line":"  operator and system maintenance; ``cleanup_failed`` boolean suffices."},{"line_number":339,"context_line":"  *Rejected*"},{"line_number":340,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"dcb56cc3_a9c55153","line":337,"range":{"start_line":337,"start_character":39,"end_line":337,"end_character":50},"updated":"2026-06-03 16:06:13.000000000","message":"maintianing is cluky and does nto alight to how we name times in other apis so cleanign is more correct","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":334,"context_line":"  instance deletion for minutes to hours, exceeding API timeout windows."},{"line_number":335,"context_line":"  *Rejected*"},{"line_number":336,"context_line":""},{"line_number":337,"context_line":"* Add new \"cleaning\" device status — ``maintaining`` already covers both"},{"line_number":338,"context_line":"  operator and system maintenance; ``cleanup_failed`` boolean suffices."},{"line_number":339,"context_line":"  *Rejected*"},{"line_number":340,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"eda2482f_dc3733fe","line":337,"range":{"start_line":337,"start_character":39,"end_line":337,"end_character":50},"in_reply_to":"dcb56cc3_a9c55153","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"021287ad55d98a8a0e4369ab96195563565e4e5b","unresolved":true,"context_lines":[{"line_number":348,"context_line":"* Require vendor-specific drivers for all NVMe hardware — does not scale;"},{"line_number":349,"context_line":"  sanitize/format commands are standardized per NVM Express spec. *Rejected*"},{"line_number":350,"context_line":""},{"line_number":351,"context_line":"* Rely on Nova PCI passthrough alone without Cyborg cleanup — does not address"},{"line_number":352,"context_line":"  Cyborg-managed allocations via device profiles. *Rejected*"},{"line_number":353,"context_line":""},{"line_number":354,"context_line":"* Use Cyborg generic PCI driver with external cleanup automation — error-prone"}],"source_content_type":"text/x-rst","patch_set":10,"id":"d0b099e8_ab32d5aa","line":351,"updated":"2026-06-02 10:16:48.000000000","message":"this is a nit but I would keep the last two bullet points at the top, since they describe alternative options for the full spec, the others are alternative ideas for some aspects of the design","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":348,"context_line":"* Require vendor-specific drivers for all NVMe hardware — does not scale;"},{"line_number":349,"context_line":"  sanitize/format commands are standardized per NVM Express spec. *Rejected*"},{"line_number":350,"context_line":""},{"line_number":351,"context_line":"* Rely on Nova PCI passthrough alone without Cyborg cleanup — does not address"},{"line_number":352,"context_line":"  Cyborg-managed allocations via device profiles. *Rejected*"},{"line_number":353,"context_line":""},{"line_number":354,"context_line":"* Use Cyborg generic PCI driver with external cleanup automation — error-prone"}],"source_content_type":"text/x-rst","patch_set":10,"id":"bcdd78ff_a94332ef","line":351,"in_reply_to":"d0b099e8_ab32d5aa","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":385,"context_line":""},{"line_number":386,"context_line":"* ``GET /v2/devices``"},{"line_number":387,"context_line":"* ``GET /v2/devices/{uuid}``"},{"line_number":388,"context_line":"* ``POST /v2/devices/{uuid}/enable``"},{"line_number":389,"context_line":""},{"line_number":390,"context_line":"No request body changes are proposed."},{"line_number":391,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"5447ba27_5904068d","line":388,"updated":"2026-06-03 16:06:13.000000000","message":"-1 to modifying this\n\nas noted above enabled/disable and the status field have a diffent usecase the trakcing the device_state.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":385,"context_line":""},{"line_number":386,"context_line":"* ``GET /v2/devices``"},{"line_number":387,"context_line":"* ``GET /v2/devices/{uuid}``"},{"line_number":388,"context_line":"* ``POST /v2/devices/{uuid}/enable``"},{"line_number":389,"context_line":""},{"line_number":390,"context_line":"No request body changes are proposed."},{"line_number":391,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"63fc5c4e_cd728af0","line":388,"in_reply_to":"5447ba27_5904068d","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":393,"context_line":""},{"line_number":394,"context_line":"For API microversion ``2.4`` and later, device responses include:"},{"line_number":395,"context_line":""},{"line_number":396,"context_line":"* ``cleanup_failed`` — new boolean field, defaulting to ``false``"},{"line_number":397,"context_line":"* ``\"NVME\"`` — new valid value for the ``type`` field"},{"line_number":398,"context_line":""},{"line_number":399,"context_line":"**Example GET /v2/devices/{uuid} response (microversion 2.4+):**"}],"source_content_type":"text/x-rst","patch_set":10,"id":"9dbfd532_4bf8df28","line":396,"range":{"start_line":396,"start_character":4,"end_line":396,"end_character":18},"updated":"2026-06-03 16:06:13.000000000","message":"this shoudl be upsdate to say \n\ndevice_state\u003derror","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":393,"context_line":""},{"line_number":394,"context_line":"For API microversion ``2.4`` and later, device responses include:"},{"line_number":395,"context_line":""},{"line_number":396,"context_line":"* ``cleanup_failed`` — new boolean field, defaulting to ``false``"},{"line_number":397,"context_line":"* ``\"NVME\"`` — new valid value for the ``type`` field"},{"line_number":398,"context_line":""},{"line_number":399,"context_line":"**Example GET /v2/devices/{uuid} response (microversion 2.4+):**"}],"source_content_type":"text/x-rst","patch_set":10,"id":"07f62697_9b46c407","line":396,"range":{"start_line":396,"start_character":4,"end_line":396,"end_character":18},"in_reply_to":"9dbfd532_4bf8df28","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":394,"context_line":"For API microversion ``2.4`` and later, device responses include:"},{"line_number":395,"context_line":""},{"line_number":396,"context_line":"* ``cleanup_failed`` — new boolean field, defaulting to ``false``"},{"line_number":397,"context_line":"* ``\"NVME\"`` — new valid value for the ``type`` field"},{"line_number":398,"context_line":""},{"line_number":399,"context_line":"**Example GET /v2/devices/{uuid} response (microversion 2.4+):**"},{"line_number":400,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"433bab4a_1d39ae8e","line":397,"updated":"2026-06-03 16:06:13.000000000","message":"so by its self woud not require a microverion fro 2 reasons\n\nits a strign filed not an enum filed so we don\u0027t need a micro-version to extend and you can write out of tree driver so the value of type is always extensible without an API change.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":394,"context_line":"For API microversion ``2.4`` and later, device responses include:"},{"line_number":395,"context_line":""},{"line_number":396,"context_line":"* ``cleanup_failed`` — new boolean field, defaulting to ``false``"},{"line_number":397,"context_line":"* ``\"NVME\"`` — new valid value for the ``type`` field"},{"line_number":398,"context_line":""},{"line_number":399,"context_line":"**Example GET /v2/devices/{uuid} response (microversion 2.4+):**"},{"line_number":400,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"3aaa9c6d_1058eb4c","line":397,"in_reply_to":"433bab4a_1d39ae8e","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":405,"context_line":"    \"type\": \"NVME\","},{"line_number":406,"context_line":"    \"vendor\": \"8086\","},{"line_number":407,"context_line":"    \"model\": \"0001\","},{"line_number":408,"context_line":"    \"std_board_info\": \"{\\\"product_id\\\": \\\"0001\\\", \\\"controller\\\": \\\"nvme0\\\"}\","},{"line_number":409,"context_line":"    \"vendor_board_info\": null,"},{"line_number":410,"context_line":"    \"hostname\": \"compute-1\","},{"line_number":411,"context_line":"    \"status\": \"enabled\","},{"line_number":412,"context_line":"    \"cleanup_failed\": false,"}],"source_content_type":"text/x-rst","patch_set":10,"id":"410eda8c_ecfef4a3","line":409,"range":{"start_line":408,"start_character":0,"end_line":409,"end_character":30},"updated":"2026-06-03 16:06:13.000000000","message":"its out of scope for now but in a sepreate microverion we might repalce bot of these with a generic deriver depenet files\n\nfor now this is fine i do not want to encode new singel dirver filed in this but some question on the current respesntaiton\n\nwhy are you using contoler and pointing at nvme0 isntead of encodeing the pci adress\n\nthe pci adress is more useful as that will map to the placement resouce class and its the ting we pass to nova.\n\nim not sure we need to have the controller at all.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":405,"context_line":"    \"type\": \"NVME\","},{"line_number":406,"context_line":"    \"vendor\": \"8086\","},{"line_number":407,"context_line":"    \"model\": \"0001\","},{"line_number":408,"context_line":"    \"std_board_info\": \"{\\\"product_id\\\": \\\"0001\\\", \\\"controller\\\": \\\"nvme0\\\"}\","},{"line_number":409,"context_line":"    \"vendor_board_info\": null,"},{"line_number":410,"context_line":"    \"hostname\": \"compute-1\","},{"line_number":411,"context_line":"    \"status\": \"enabled\","},{"line_number":412,"context_line":"    \"cleanup_failed\": false,"}],"source_content_type":"text/x-rst","patch_set":10,"id":"db03e626_eccfd463","line":409,"range":{"start_line":408,"start_character":0,"end_line":409,"end_character":30},"in_reply_to":"410eda8c_ecfef4a3","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":430,"context_line":"    \"updated_at\": \"2026-05-15T16:30:15Z\""},{"line_number":431,"context_line":"  }"},{"line_number":432,"context_line":""},{"line_number":433,"context_line":"**POST /v2/devices/{uuid}/enable behavior:**"},{"line_number":434,"context_line":""},{"line_number":435,"context_line":"Returns HTTP 409 Conflict when ``cleanup_failed\u003dTrue``::"},{"line_number":436,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"ef28796d_ef827a66","line":433,"updated":"2026-06-03 16:06:13.000000000","message":"again instead of this we should have\n\nPOST /v2/devices/{uuid}/clean to manually triger cleaning but that shoudl not be requried in the standard flow\n\nthis shodl return a 409 conflict if the device is curetnly in the allocated state for now and be admin only.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":430,"context_line":"    \"updated_at\": \"2026-05-15T16:30:15Z\""},{"line_number":431,"context_line":"  }"},{"line_number":432,"context_line":""},{"line_number":433,"context_line":"**POST /v2/devices/{uuid}/enable behavior:**"},{"line_number":434,"context_line":""},{"line_number":435,"context_line":"Returns HTTP 409 Conflict when ``cleanup_failed\u003dTrue``::"},{"line_number":436,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"50f2fc27_127be7b5","line":433,"in_reply_to":"ef28796d_ef827a66","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"021287ad55d98a8a0e4369ab96195563565e4e5b","unresolved":true,"context_lines":[{"line_number":449,"context_line":""},{"line_number":450,"context_line":"This safety guard ensures operators use the ``cyborg-nvme-cleanup`` CLI to"},{"line_number":451,"context_line":"retry failed cleanups, which performs proper verification. Operators who need"},{"line_number":452,"context_line":"to force-enable a device without cleanup can set ``cleanup_failed\u003dFalse`` via"},{"line_number":453,"context_line":"direct database update as an escape hatch."},{"line_number":454,"context_line":""},{"line_number":455,"context_line":"Policy is unchanged."}],"source_content_type":"text/x-rst","patch_set":10,"id":"35a042f8_6abbd752","line":452,"updated":"2026-06-02 10:16:48.000000000","message":"is there any use case where we envision this might be needed? At first glance it seems dangerous to tell operators to write directly to the database","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":false,"context_lines":[{"line_number":449,"context_line":""},{"line_number":450,"context_line":"This safety guard ensures operators use the ``cyborg-nvme-cleanup`` CLI to"},{"line_number":451,"context_line":"retry failed cleanups, which performs proper verification. Operators who need"},{"line_number":452,"context_line":"to force-enable a device without cleanup can set ``cleanup_failed\u003dFalse`` via"},{"line_number":453,"context_line":"direct database update as an escape hatch."},{"line_number":454,"context_line":""},{"line_number":455,"context_line":"Policy is unchanged."}],"source_content_type":"text/x-rst","patch_set":10,"id":"7b08e86a_7312528d","line":452,"in_reply_to":"35a042f8_6abbd752","updated":"2026-06-19 04:57:23.000000000","message":"Addressed in PS11. \n\nThe direct DB update instruction has been replaced with the admin-only POST /v2/devices/{uuid}/clean API endpoint. \n\nOperators no longer need to touch the database directly.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":460,"context_line":"A new method is added to the agent RPC API (version bump ``1.0`` →"},{"line_number":461,"context_line":"``1.1``)::"},{"line_number":462,"context_line":""},{"line_number":463,"context_line":"    def cleanup_nvme_device(context, device_uuid, pci_addr):"},{"line_number":464,"context_line":"        \"\"\"Trigger async NVMe device cleanup."},{"line_number":465,"context_line":""},{"line_number":466,"context_line":"        :param context: request context"}],"source_content_type":"text/x-rst","patch_set":10,"id":"17fc7750_2f9fa522","line":463,"updated":"2026-06-03 16:06:13.000000000","message":"i do not want to add any driver specifc RPCs if we can avoid it\n\nso this shoudl be a gener cleanup_device function that take a device object","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":460,"context_line":"A new method is added to the agent RPC API (version bump ``1.0`` →"},{"line_number":461,"context_line":"``1.1``)::"},{"line_number":462,"context_line":""},{"line_number":463,"context_line":"    def cleanup_nvme_device(context, device_uuid, pci_addr):"},{"line_number":464,"context_line":"        \"\"\"Trigger async NVMe device cleanup."},{"line_number":465,"context_line":""},{"line_number":466,"context_line":"        :param context: request context"}],"source_content_type":"text/x-rst","patch_set":10,"id":"f650367d_c10fa435","line":463,"in_reply_to":"17fc7750_2f9fa522","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"e589fdf9e9d0dbc3ea17a46ce3b62b05d2ecdefe","unresolved":true,"context_lines":[{"line_number":477,"context_line":"        \u0027cleanup_nvme_device\u0027,"},{"line_number":478,"context_line":"        device_uuid\u003ddevice.uuid,"},{"line_number":479,"context_line":"        pci_addr\u003dpci_addr"},{"line_number":480,"context_line":"    )"},{"line_number":481,"context_line":""},{"line_number":482,"context_line":"The RPC returns immediately so Nova\u0027s unbind completes without blocking on"},{"line_number":483,"context_line":"cleanup. The agent updates device status directly via conductor RPC callbacks"}],"source_content_type":"text/x-rst","patch_set":10,"id":"3a3ea3cf_5a21f3b5","line":480,"updated":"2026-05-20 15:35:40.000000000","message":"do you support rolling upgrades in Cyborg ? if so, you should doublecheck whether the agent supports the 1.1 RPC version before you RPC cast it.\n\nIf so, you also need to explain about what kind of exception you would provide so.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":477,"context_line":"        \u0027cleanup_nvme_device\u0027,"},{"line_number":478,"context_line":"        device_uuid\u003ddevice.uuid,"},{"line_number":479,"context_line":"        pci_addr\u003dpci_addr"},{"line_number":480,"context_line":"    )"},{"line_number":481,"context_line":""},{"line_number":482,"context_line":"The RPC returns immediately so Nova\u0027s unbind completes without blocking on"},{"line_number":483,"context_line":"cleanup. The agent updates device status directly via conductor RPC callbacks"}],"source_content_type":"text/x-rst","patch_set":10,"id":"b10b2943_28a46f59","line":480,"in_reply_to":"254c849c_bebf0446","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":477,"context_line":"        \u0027cleanup_nvme_device\u0027,"},{"line_number":478,"context_line":"        device_uuid\u003ddevice.uuid,"},{"line_number":479,"context_line":"        pci_addr\u003dpci_addr"},{"line_number":480,"context_line":"    )"},{"line_number":481,"context_line":""},{"line_number":482,"context_line":"The RPC returns immediately so Nova\u0027s unbind completes without blocking on"},{"line_number":483,"context_line":"cleanup. The agent updates device status directly via conductor RPC callbacks"}],"source_content_type":"text/x-rst","patch_set":10,"id":"254c849c_bebf0446","line":480,"in_reply_to":"3a3ea3cf_5a21f3b5","updated":"2026-06-03 16:06:13.000000000","message":"cybrog has never had grenade testing until we got invovled so while we shoudl eventualy we have very littel supprot for that.\n\nwe do not have the equivlanet of compute service version or rpc version level like nova does.\n\nso effectivly no we have not supprot today for roling upgrade in cyborg\n\nin practice it work by acident because we have not really done any rpc changes in a while.  we do not have compute service recored equivlenet today as there is no db represetion fo the cybrog agetns at all.\n\nthis is not currently planded for the near term to resolve but perhas in or after 2027.1\n\nthe best we can do without changes that are very much ot of scope fo this is realy on docuemeation and maybe on the rpc client perpare to detect the mis alignemtn.\n\nupgardes are a major gap in cyborgs production readyness today","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":502,"context_line":"* **Denial of service:** Long-running subprocesses must be bounded by"},{"line_number":503,"context_line":"  configuration; concurrent cleanups per host must remain bounded."},{"line_number":504,"context_line":""},{"line_number":505,"context_line":""},{"line_number":506,"context_line":"Notifications impact"},{"line_number":507,"context_line":"--------------------"},{"line_number":508,"context_line":""},{"line_number":509,"context_line":"Operator notifications on cleanup completion/failure are deferred to a"},{"line_number":510,"context_line":"follow-on spec."},{"line_number":511,"context_line":""},{"line_number":512,"context_line":""},{"line_number":513,"context_line":"Other end user impact"}],"source_content_type":"text/x-rst","patch_set":10,"id":"ead14b7c_5103b6ea","line":510,"range":{"start_line":505,"start_character":1,"end_line":510,"end_character":14},"updated":"2026-06-03 16:06:13.000000000","message":"https://github.com/openstack/cyborg/blob/233f2a5b7e396c24bf15ae22f8cf64a480666f5c/cyborg/common/rpc.py#L117\n\nwhile syborg has som einital notificaon supprot\n\nit was not actully implemted out of the inial copy past when the repo was being created\n\nhttps://github.com/openstack/cyborg/commit/c7b24fda7fbf2c117bcd061be2c6dd9ae88acd56\n\n\nif notification were implemted properly in cybrog then we would not defer them as part fo this work\n\nim ok with defering htem in this case solely because we shoudl isntead have a singel spec to implement them end to end for all oeprations.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":502,"context_line":"* **Denial of service:** Long-running subprocesses must be bounded by"},{"line_number":503,"context_line":"  configuration; concurrent cleanups per host must remain bounded."},{"line_number":504,"context_line":""},{"line_number":505,"context_line":""},{"line_number":506,"context_line":"Notifications impact"},{"line_number":507,"context_line":"--------------------"},{"line_number":508,"context_line":""},{"line_number":509,"context_line":"Operator notifications on cleanup completion/failure are deferred to a"},{"line_number":510,"context_line":"follow-on spec."},{"line_number":511,"context_line":""},{"line_number":512,"context_line":""},{"line_number":513,"context_line":"Other end user impact"}],"source_content_type":"text/x-rst","patch_set":10,"id":"6741673f_ca78d94b","line":510,"range":{"start_line":505,"start_character":1,"end_line":510,"end_character":14},"in_reply_to":"ead14b7c_5103b6ea","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":551,"context_line":"--------------"},{"line_number":552,"context_line":""},{"line_number":553,"context_line":"* **Existing instances with NVMe devices**: For major upgrades, existing"},{"line_number":554,"context_line":"  workloads cannot directly use Cyborg-managed NVMe flavors. Operators"},{"line_number":555,"context_line":"  must take a snapshot of the VM and restore it with a new flavor that"},{"line_number":556,"context_line":"  includes the appropriate device profile for NVMe devices."},{"line_number":557,"context_line":""},{"line_number":558,"context_line":"* **Migration from Nova PCI passthrough**: If a cloud is currently using"},{"line_number":559,"context_line":"  NVMe devices via Nova PCI passthrough, operators must remove the device"}],"source_content_type":"text/x-rst","patch_set":10,"id":"5d1da2eb_8bda3409","line":556,"range":{"start_line":554,"start_character":61,"end_line":556,"end_character":59},"updated":"2026-06-03 16:06:13.000000000","message":"lets remove this and simply say that vms must be recreated with new flavor to adopt this feature.\n\ni dont want to perpetuate teh methign that is ever going to be possibel to upgrade form a nova manged deivce to a cybrog manged on or to change the driver that manages a device untile we have implemented resize in nova for cybrog.\n\nwe are not relacign the existing fucutionaltiy we are provideing a altrenitve parallel implemation that you can chose to use so there is now upgrae impact in that reagrad for this work. we are just addign a new driver\n\nany use of that new driver is a post upgrade activity to enabel and configure it and shoudl not be dicsss in the upgrade impact section\n\nthat is a better fit for the otehr deployer impact section.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":551,"context_line":"--------------"},{"line_number":552,"context_line":""},{"line_number":553,"context_line":"* **Existing instances with NVMe devices**: For major upgrades, existing"},{"line_number":554,"context_line":"  workloads cannot directly use Cyborg-managed NVMe flavors. Operators"},{"line_number":555,"context_line":"  must take a snapshot of the VM and restore it with a new flavor that"},{"line_number":556,"context_line":"  includes the appropriate device profile for NVMe devices."},{"line_number":557,"context_line":""},{"line_number":558,"context_line":"* **Migration from Nova PCI passthrough**: If a cloud is currently using"},{"line_number":559,"context_line":"  NVMe devices via Nova PCI passthrough, operators must remove the device"}],"source_content_type":"text/x-rst","patch_set":10,"id":"5aa2debc_98ea8fa4","line":556,"range":{"start_line":554,"start_character":61,"end_line":556,"end_character":59},"in_reply_to":"5d1da2eb_8bda3409","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":555,"context_line":"  must take a snapshot of the VM and restore it with a new flavor that"},{"line_number":556,"context_line":"  includes the appropriate device profile for NVMe devices."},{"line_number":557,"context_line":""},{"line_number":558,"context_line":"* **Migration from Nova PCI passthrough**: If a cloud is currently using"},{"line_number":559,"context_line":"  NVMe devices via Nova PCI passthrough, operators must remove the device"},{"line_number":560,"context_line":"  configuration from Nova side (``nova.conf``), stop the compute service,"},{"line_number":561,"context_line":"  add the device to ``cyborg.conf`` under ``[nvme] device_spec``, enable"},{"line_number":562,"context_line":"  ``nvme_driver`` in the ``[agent]`` section, and restart the cyborg-agent."},{"line_number":563,"context_line":"  Cyborg will then discover and manage the device."},{"line_number":564,"context_line":""},{"line_number":565,"context_line":"* **Existing Inspur driver deployments**: The Inspur NVMe driver is"},{"line_number":566,"context_line":"  unaffected. To also use the generic NVMe driver, add ``nvme_driver`` to"}],"source_content_type":"text/x-rst","patch_set":10,"id":"fea38c82_e6177e67","line":563,"range":{"start_line":558,"start_character":1,"end_line":563,"end_character":50},"updated":"2026-06-03 16:06:13.000000000","message":"this is not a thing\n\nnova does nto supprot nvme device via pci passhtouh and never will.\nits also not an upgrade activity its a post upgrade active\n\nupgrade do not invovle config changes we docuement this in teh greade throry of upgrade\n\nhttps://opendev.org/openstack/grenade#Theory%20of%20Upgrade\n\nall config changes happen psot upgrade.\n\nwe shoudl not docuemtn the nova  migration in this spec or this secton\n\nfi we want to describe that we can descirb it later in a dedicate doc in the cybrog repo but its misleading to present this as an upgrade related activy it is not.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":555,"context_line":"  must take a snapshot of the VM and restore it with a new flavor that"},{"line_number":556,"context_line":"  includes the appropriate device profile for NVMe devices."},{"line_number":557,"context_line":""},{"line_number":558,"context_line":"* **Migration from Nova PCI passthrough**: If a cloud is currently using"},{"line_number":559,"context_line":"  NVMe devices via Nova PCI passthrough, operators must remove the device"},{"line_number":560,"context_line":"  configuration from Nova side (``nova.conf``), stop the compute service,"},{"line_number":561,"context_line":"  add the device to ``cyborg.conf`` under ``[nvme] device_spec``, enable"},{"line_number":562,"context_line":"  ``nvme_driver`` in the ``[agent]`` section, and restart the cyborg-agent."},{"line_number":563,"context_line":"  Cyborg will then discover and manage the device."},{"line_number":564,"context_line":""},{"line_number":565,"context_line":"* **Existing Inspur driver deployments**: The Inspur NVMe driver is"},{"line_number":566,"context_line":"  unaffected. To also use the generic NVMe driver, add ``nvme_driver`` to"}],"source_content_type":"text/x-rst","patch_set":10,"id":"627f72f4_e82535cd","line":563,"range":{"start_line":558,"start_character":1,"end_line":563,"end_character":50},"in_reply_to":"fea38c82_e6177e67","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"e589fdf9e9d0dbc3ea17a46ce3b62b05d2ecdefe","unresolved":true,"context_lines":[{"line_number":570,"context_line":"  ``nvme_driver`` enabled. The ``Device`` object version bump requires"},{"line_number":571,"context_line":"  version-pinning to prevent new fields reaching old peers during rolling"},{"line_number":572,"context_line":"  upgrades."},{"line_number":573,"context_line":""},{"line_number":574,"context_line":""},{"line_number":575,"context_line":"Implementation"},{"line_number":576,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"}],"source_content_type":"text/x-rst","patch_set":10,"id":"f91e6b86_e9990e19","line":573,"updated":"2026-05-20 15:35:40.000000000","message":"what happens if you upgrade conductors first before your agents ?","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":570,"context_line":"  ``nvme_driver`` enabled. The ``Device`` object version bump requires"},{"line_number":571,"context_line":"  version-pinning to prevent new fields reaching old peers during rolling"},{"line_number":572,"context_line":"  upgrades."},{"line_number":573,"context_line":""},{"line_number":574,"context_line":""},{"line_number":575,"context_line":"Implementation"},{"line_number":576,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"}],"source_content_type":"text/x-rst","patch_set":10,"id":"8b6a94e0_25e8af2d","line":573,"in_reply_to":"7accdca0_cf50b957","updated":"2026-06-03 16:06:13.000000000","message":"the agents access the db via the conductor\nthe conductor and api must eb upgrade after the db schema migration and before teh agents.\n\nwhoever we do not have any strong rollign upgrade guarentees or supprot.\n\nwe added greade jobs in the last mont or two but historiclly cybrog has had no upgrade testing or documented upgrade path i added https://docs.openstack.org/cyborg/latest/admin/upgrade.html as an initally attepmt to intoduces one but effectilly assuem that cybrog upgrade are all or nothign with the intent to supprot n-1 and n-2 cybrog agent by the time we reach 2027.1","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":570,"context_line":"  ``nvme_driver`` enabled. The ``Device`` object version bump requires"},{"line_number":571,"context_line":"  version-pinning to prevent new fields reaching old peers during rolling"},{"line_number":572,"context_line":"  upgrades."},{"line_number":573,"context_line":""},{"line_number":574,"context_line":""},{"line_number":575,"context_line":"Implementation"},{"line_number":576,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"}],"source_content_type":"text/x-rst","patch_set":10,"id":"4bca298c_a7495871","line":573,"in_reply_to":"8b6a94e0_25e8af2d","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"021287ad55d98a8a0e4369ab96195563565e4e5b","unresolved":true,"context_lines":[{"line_number":570,"context_line":"  ``nvme_driver`` enabled. The ``Device`` object version bump requires"},{"line_number":571,"context_line":"  version-pinning to prevent new fields reaching old peers during rolling"},{"line_number":572,"context_line":"  upgrades."},{"line_number":573,"context_line":""},{"line_number":574,"context_line":""},{"line_number":575,"context_line":"Implementation"},{"line_number":576,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"}],"source_content_type":"text/x-rst","patch_set":10,"id":"7accdca0_cf50b957","line":573,"in_reply_to":"f91e6b86_e9990e19","updated":"2026-06-02 10:16:48.000000000","message":"for reference that is the recommenden procedure for upgrades in the docs https://docs.openstack.org/cyborg/latest/admin/upgrade.html:\n```\nTypical component order when rolling out a new version is: conductor and API first, then agents, so orchestration and the REST API match before dataplane agents pick up changes. Adjust for your deployment tooling if needed.\n```","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"e589fdf9e9d0dbc3ea17a46ce3b62b05d2ecdefe","unresolved":true,"context_lines":[{"line_number":579,"context_line":"-----------"},{"line_number":580,"context_line":""},{"line_number":581,"context_line":"Primary assignee:"},{"line_number":582,"context_line":"  None"},{"line_number":583,"context_line":""},{"line_number":584,"context_line":"Other contributors:"},{"line_number":585,"context_line":"  None"}],"source_content_type":"text/x-rst","patch_set":10,"id":"06d6184f_1fdf74e3","line":582,"updated":"2026-05-20 15:35:40.000000000","message":"hopefully not.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":579,"context_line":"-----------"},{"line_number":580,"context_line":""},{"line_number":581,"context_line":"Primary assignee:"},{"line_number":582,"context_line":"  None"},{"line_number":583,"context_line":""},{"line_number":584,"context_line":"Other contributors:"},{"line_number":585,"context_line":"  None"}],"source_content_type":"text/x-rst","patch_set":10,"id":"51641f66_261c01f1","line":582,"in_reply_to":"06d6184f_1fdf74e3","updated":"2026-06-03 16:06:13.000000000","message":"this should be chandan","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":579,"context_line":"-----------"},{"line_number":580,"context_line":""},{"line_number":581,"context_line":"Primary assignee:"},{"line_number":582,"context_line":"  None"},{"line_number":583,"context_line":""},{"line_number":584,"context_line":"Other contributors:"},{"line_number":585,"context_line":"  None"}],"source_content_type":"text/x-rst","patch_set":10,"id":"d10d4be9_f184766c","line":582,"in_reply_to":"51641f66_261c01f1","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":584,"context_line":"Other contributors:"},{"line_number":585,"context_line":"  None"},{"line_number":586,"context_line":""},{"line_number":587,"context_line":"Work Items"},{"line_number":588,"context_line":"----------"},{"line_number":589,"context_line":""},{"line_number":590,"context_line":"Implementation is organized into five phases with explicit dependencies."}],"source_content_type":"text/x-rst","patch_set":10,"id":"42ca7951_d127b4cc","line":587,"range":{"start_line":587,"start_character":0,"end_line":587,"end_character":2},"updated":"2026-06-03 16:06:13.000000000","message":"the work items section shoudl be short and call out the main phase fo the developemt\n\nthere shoudl be no new design or implemation detail captured in this section that have not areadly been expliend in the propose changes seciotn\n\nyou shoudl also avoid repeating infroamtion already coverd in the reset of the spec.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":584,"context_line":"Other contributors:"},{"line_number":585,"context_line":"  None"},{"line_number":586,"context_line":""},{"line_number":587,"context_line":"Work Items"},{"line_number":588,"context_line":"----------"},{"line_number":589,"context_line":""},{"line_number":590,"context_line":"Implementation is organized into five phases with explicit dependencies."}],"source_content_type":"text/x-rst","patch_set":10,"id":"700f510d_7069e454","line":587,"range":{"start_line":587,"start_character":0,"end_line":587,"end_character":2},"in_reply_to":"42ca7951_d127b4cc","updated":"2026-06-17 04:45:31.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":591,"context_line":""},{"line_number":592,"context_line":"**Phase 1: Data model and API foundation**"},{"line_number":593,"context_line":""},{"line_number":594,"context_line":"* Add Alembic migration adding ``NVME`` to ``device.type`` ENUM and"},{"line_number":595,"context_line":"  ``cleanup_failed`` boolean column (``NOT NULL DEFAULT FALSE``)"},{"line_number":596,"context_line":"* Bump ``Device`` oslo.versionedobjects version and add ``cleanup_failed``"},{"line_number":597,"context_line":"  field handling"},{"line_number":598,"context_line":"* Add Cyborg API microversion (expected to be ``2.4``)"},{"line_number":599,"context_line":"* Update Cyborg API version history for new device fields"},{"line_number":600,"context_line":"* Serialize ``cleanup_failed`` field in device API responses for"},{"line_number":601,"context_line":"  microversion ``2.4`` and later"}],"source_content_type":"text/x-rst","patch_set":10,"id":"0f0bdc4c_434ffce7","line":598,"range":{"start_line":594,"start_character":0,"end_line":598,"end_character":54},"updated":"2026-06-03 16:06:13.000000000","message":"so the api field is not defiend as an enuma so adding a value to the set of possibel value does not requrie a microverion even if we do need to exted the model \nhttps://github.com/openstack/cyborg/blob/master/cyborg/db/sqlalchemy/models.py#L86","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":591,"context_line":""},{"line_number":592,"context_line":"**Phase 1: Data model and API foundation**"},{"line_number":593,"context_line":""},{"line_number":594,"context_line":"* Add Alembic migration adding ``NVME`` to ``device.type`` ENUM and"},{"line_number":595,"context_line":"  ``cleanup_failed`` boolean column (``NOT NULL DEFAULT FALSE``)"},{"line_number":596,"context_line":"* Bump ``Device`` oslo.versionedobjects version and add ``cleanup_failed``"},{"line_number":597,"context_line":"  field handling"},{"line_number":598,"context_line":"* Add Cyborg API microversion (expected to be ``2.4``)"},{"line_number":599,"context_line":"* Update Cyborg API version history for new device fields"},{"line_number":600,"context_line":"* Serialize ``cleanup_failed`` field in device API responses for"},{"line_number":601,"context_line":"  microversion ``2.4`` and later"}],"source_content_type":"text/x-rst","patch_set":10,"id":"7440e78e_a29e6ad9","line":598,"range":{"start_line":594,"start_character":0,"end_line":598,"end_character":54},"in_reply_to":"0f0bdc4c_434ffce7","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":596,"context_line":"* Bump ``Device`` oslo.versionedobjects version and add ``cleanup_failed``"},{"line_number":597,"context_line":"  field handling"},{"line_number":598,"context_line":"* Add Cyborg API microversion (expected to be ``2.4``)"},{"line_number":599,"context_line":"* Update Cyborg API version history for new device fields"},{"line_number":600,"context_line":"* Serialize ``cleanup_failed`` field in device API responses for"},{"line_number":601,"context_line":"  microversion ``2.4`` and later"},{"line_number":602,"context_line":"* Add HTTP 409 guard on ``POST /v2/devices/{uuid}/enable`` when"},{"line_number":603,"context_line":"  ``cleanup_failed\u003dTrue``"},{"line_number":604,"context_line":"* Unit tests for new API microversion, field serialization, and error handling"},{"line_number":605,"context_line":""},{"line_number":606,"context_line":"**Phase 2: Driver base and discovery**"},{"line_number":607,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"632535c1_d1630175","line":604,"range":{"start_line":599,"start_character":1,"end_line":604,"end_character":78},"updated":"2026-06-03 16:06:13.000000000","message":"the api changes shoudl be the final phase only after all other work is compelte.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":596,"context_line":"* Bump ``Device`` oslo.versionedobjects version and add ``cleanup_failed``"},{"line_number":597,"context_line":"  field handling"},{"line_number":598,"context_line":"* Add Cyborg API microversion (expected to be ``2.4``)"},{"line_number":599,"context_line":"* Update Cyborg API version history for new device fields"},{"line_number":600,"context_line":"* Serialize ``cleanup_failed`` field in device API responses for"},{"line_number":601,"context_line":"  microversion ``2.4`` and later"},{"line_number":602,"context_line":"* Add HTTP 409 guard on ``POST /v2/devices/{uuid}/enable`` when"},{"line_number":603,"context_line":"  ``cleanup_failed\u003dTrue``"},{"line_number":604,"context_line":"* Unit tests for new API microversion, field serialization, and error handling"},{"line_number":605,"context_line":""},{"line_number":606,"context_line":"**Phase 2: Driver base and discovery**"},{"line_number":607,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"63f1ded3_432418c0","line":604,"range":{"start_line":599,"start_character":1,"end_line":604,"end_character":78},"in_reply_to":"632535c1_d1630175","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":607,"context_line":""},{"line_number":608,"context_line":"*Depends on Phase 1 (data model must exist for driver to set cleanup_failed)*"},{"line_number":609,"context_line":""},{"line_number":610,"context_line":"* Add no-op ``cleanup(pci_addr)`` method to ``PciDriver`` base class"},{"line_number":611,"context_line":"* Create ``NVMeDriver`` class inheriting from ``PciDriver`` with"},{"line_number":612,"context_line":"  ``type\u003d\u0027NVME\u0027``"},{"line_number":613,"context_line":"* Implement NVMe device discovery using ``/sys/bus/pci/devices/`` filtering"}],"source_content_type":"text/x-rst","patch_set":10,"id":"00af9b6c_fc9b2cb8","line":610,"range":{"start_line":610,"start_character":46,"end_line":610,"end_character":55},"updated":"2026-06-03 16:06:13.000000000","message":"generic driver","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":607,"context_line":""},{"line_number":608,"context_line":"*Depends on Phase 1 (data model must exist for driver to set cleanup_failed)*"},{"line_number":609,"context_line":""},{"line_number":610,"context_line":"* Add no-op ``cleanup(pci_addr)`` method to ``PciDriver`` base class"},{"line_number":611,"context_line":"* Create ``NVMeDriver`` class inheriting from ``PciDriver`` with"},{"line_number":612,"context_line":"  ``type\u003d\u0027NVME\u0027``"},{"line_number":613,"context_line":"* Implement NVMe device discovery using ``/sys/bus/pci/devices/`` filtering"}],"source_content_type":"text/x-rst","patch_set":10,"id":"7ded2777_8d0285cc","line":610,"range":{"start_line":610,"start_character":46,"end_line":610,"end_character":55},"in_reply_to":"00af9b6c_fc9b2cb8","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":613,"context_line":"* Implement NVMe device discovery using ``/sys/bus/pci/devices/`` filtering"},{"line_number":614,"context_line":"* Implement sysfs PCI-to-``/dev/nvmeN`` resolution for cleanup operations"},{"line_number":615,"context_line":"* Parse and validate ``[nvme] device_spec`` configuration (vendor_id/product_id"},{"line_number":616,"context_line":"  only for 2026.2; PCI address glob support deferred to PCI refactoring"},{"line_number":617,"context_line":"  blueprint)"},{"line_number":618,"context_line":"* Create Placement traits (``OWNER_CYBORG``,"},{"line_number":619,"context_line":"  ``CUSTOM_NVME_\u003cVENDOR\u003e_\u003cVENDOR_ID\u003e_\u003cPRODUCT_ID\u003e``)"},{"line_number":620,"context_line":"* Create resource providers and deployables following VGPU pattern"}],"source_content_type":"text/x-rst","patch_set":10,"id":"0984e4a7_04ba262c","line":617,"range":{"start_line":616,"start_character":18,"end_line":617,"end_character":12},"updated":"2026-06-03 16:06:13.000000000","message":"we are not currently working on this this cycle.\n\nwe shoudl support all of the adress version supproted by the pci driver today.\n\nwhile the adress shoudl not be requried it shoudl be strongly encurraged in the config option help text in teh event you use the same type of ssd as a host boot device or similar.\n\nwe may also want to supprot selecting the resouce class used via the device spec but im ok with deferring that to a diffent cycle.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":613,"context_line":"* Implement NVMe device discovery using ``/sys/bus/pci/devices/`` filtering"},{"line_number":614,"context_line":"* Implement sysfs PCI-to-``/dev/nvmeN`` resolution for cleanup operations"},{"line_number":615,"context_line":"* Parse and validate ``[nvme] device_spec`` configuration (vendor_id/product_id"},{"line_number":616,"context_line":"  only for 2026.2; PCI address glob support deferred to PCI refactoring"},{"line_number":617,"context_line":"  blueprint)"},{"line_number":618,"context_line":"* Create Placement traits (``OWNER_CYBORG``,"},{"line_number":619,"context_line":"  ``CUSTOM_NVME_\u003cVENDOR\u003e_\u003cVENDOR_ID\u003e_\u003cPRODUCT_ID\u003e``)"},{"line_number":620,"context_line":"* Create resource providers and deployables following VGPU pattern"}],"source_content_type":"text/x-rst","patch_set":10,"id":"4d50d93c_fa8b3b57","line":617,"range":{"start_line":616,"start_character":18,"end_line":617,"end_character":12},"in_reply_to":"0984e4a7_04ba262c","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":616,"context_line":"  only for 2026.2; PCI address glob support deferred to PCI refactoring"},{"line_number":617,"context_line":"  blueprint)"},{"line_number":618,"context_line":"* Create Placement traits (``OWNER_CYBORG``,"},{"line_number":619,"context_line":"  ``CUSTOM_NVME_\u003cVENDOR\u003e_\u003cVENDOR_ID\u003e_\u003cPRODUCT_ID\u003e``)"},{"line_number":620,"context_line":"* Create resource providers and deployables following VGPU pattern"},{"line_number":621,"context_line":"* Add agent startup validation preventing generic + vendor NVMe drivers both"},{"line_number":622,"context_line":"  enabled (raise ``InvalidConfiguration``)"}],"source_content_type":"text/x-rst","patch_set":10,"id":"e041fa7f_09d4c741","line":619,"range":{"start_line":619,"start_character":3,"end_line":619,"end_character":51},"updated":"2026-06-03 16:06:13.000000000","message":"-1 on modleing this as a trait\nlets keep it considtin with https://specs.openstack.org/openstack/nova-specs/specs/2023.1/implemented/pci-device-tracking-in-placement.html\n\nnext release i think im going to propos the v2 pci driver and that is one of the change i intneded to make in that work.\n\nwe want the ablity to create quota for cybrog devices based on unifed limits\n\nto do that we need to use device clases to encode the device type by default not traits.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":616,"context_line":"  only for 2026.2; PCI address glob support deferred to PCI refactoring"},{"line_number":617,"context_line":"  blueprint)"},{"line_number":618,"context_line":"* Create Placement traits (``OWNER_CYBORG``,"},{"line_number":619,"context_line":"  ``CUSTOM_NVME_\u003cVENDOR\u003e_\u003cVENDOR_ID\u003e_\u003cPRODUCT_ID\u003e``)"},{"line_number":620,"context_line":"* Create resource providers and deployables following VGPU pattern"},{"line_number":621,"context_line":"* Add agent startup validation preventing generic + vendor NVMe drivers both"},{"line_number":622,"context_line":"  enabled (raise ``InvalidConfiguration``)"}],"source_content_type":"text/x-rst","patch_set":10,"id":"44360bbd_4dfb3751","line":619,"range":{"start_line":619,"start_character":3,"end_line":619,"end_character":51},"in_reply_to":"e041fa7f_09d4c741","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":655,"context_line":"* Unit tests for capability detection, command selection, privsep execution,"},{"line_number":656,"context_line":"  and locking"},{"line_number":657,"context_line":""},{"line_number":658,"context_line":"**Phase 4: Reconciliation and recovery**"},{"line_number":659,"context_line":""},{"line_number":660,"context_line":"*Depends on Phase 3 (reconciliation needs cleanup RPC and agent callbacks)*"},{"line_number":661,"context_line":""},{"line_number":662,"context_line":"* Implement conductor periodic reconciliation task (5 minute interval) to"},{"line_number":663,"context_line":"  detect stuck cleanups"},{"line_number":664,"context_line":"* Query devices with ``status\u003dmaintaining`` AND Placement ``reserved\u003dtotal``"},{"line_number":665,"context_line":"* Compare ``(now - device.updated_at)`` against ``[nvme] cleanup_timeout``"},{"line_number":666,"context_line":"* Mark ``cleanup_failed\u003dTrue`` and log warning when timeout exceeded"},{"line_number":667,"context_line":"* Add ``cleanup_complete`` agent→conductor RPC callback to set"},{"line_number":668,"context_line":"  ``reserved\u003d0``, ``status\u003denabled``, ``cleanup_failed\u003dFalse``"},{"line_number":669,"context_line":"* Add ``cleanup_failed`` agent→conductor RPC callback to set"},{"line_number":670,"context_line":"  ``cleanup_failed\u003dTrue`` and log error"},{"line_number":671,"context_line":"* Implement ``cyborg-nvme-cleanup`` CLI tool for manual retry (requires admin"},{"line_number":672,"context_line":"  credentials)"},{"line_number":673,"context_line":"* CLI should re-trigger agent RPC and reset ``cleanup_failed\u003dFalse`` on"},{"line_number":674,"context_line":"  success"},{"line_number":675,"context_line":"* Integration tests for timeout detection, failure scenarios, and manual"},{"line_number":676,"context_line":"  recovery flows"},{"line_number":677,"context_line":"* Unit tests for reconciliation task, callback handlers, and CLI operations"},{"line_number":678,"context_line":""},{"line_number":679,"context_line":"**Phase 5: Configuration and documentation**"},{"line_number":680,"context_line":""},{"line_number":681,"context_line":"*Can proceed in parallel with Phase 4*"}],"source_content_type":"text/x-rst","patch_set":10,"id":"5af0be0e_48085e7a","line":678,"range":{"start_line":658,"start_character":0,"end_line":678,"end_character":1},"updated":"2026-06-03 16:06:13.000000000","message":"this section can be deleted entirly as i dont think any of this is correct to include in this design","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":655,"context_line":"* Unit tests for capability detection, command selection, privsep execution,"},{"line_number":656,"context_line":"  and locking"},{"line_number":657,"context_line":""},{"line_number":658,"context_line":"**Phase 4: Reconciliation and recovery**"},{"line_number":659,"context_line":""},{"line_number":660,"context_line":"*Depends on Phase 3 (reconciliation needs cleanup RPC and agent callbacks)*"},{"line_number":661,"context_line":""},{"line_number":662,"context_line":"* Implement conductor periodic reconciliation task (5 minute interval) to"},{"line_number":663,"context_line":"  detect stuck cleanups"},{"line_number":664,"context_line":"* Query devices with ``status\u003dmaintaining`` AND Placement ``reserved\u003dtotal``"},{"line_number":665,"context_line":"* Compare ``(now - device.updated_at)`` against ``[nvme] cleanup_timeout``"},{"line_number":666,"context_line":"* Mark ``cleanup_failed\u003dTrue`` and log warning when timeout exceeded"},{"line_number":667,"context_line":"* Add ``cleanup_complete`` agent→conductor RPC callback to set"},{"line_number":668,"context_line":"  ``reserved\u003d0``, ``status\u003denabled``, ``cleanup_failed\u003dFalse``"},{"line_number":669,"context_line":"* Add ``cleanup_failed`` agent→conductor RPC callback to set"},{"line_number":670,"context_line":"  ``cleanup_failed\u003dTrue`` and log error"},{"line_number":671,"context_line":"* Implement ``cyborg-nvme-cleanup`` CLI tool for manual retry (requires admin"},{"line_number":672,"context_line":"  credentials)"},{"line_number":673,"context_line":"* CLI should re-trigger agent RPC and reset ``cleanup_failed\u003dFalse`` on"},{"line_number":674,"context_line":"  success"},{"line_number":675,"context_line":"* Integration tests for timeout detection, failure scenarios, and manual"},{"line_number":676,"context_line":"  recovery flows"},{"line_number":677,"context_line":"* Unit tests for reconciliation task, callback handlers, and CLI operations"},{"line_number":678,"context_line":""},{"line_number":679,"context_line":"**Phase 5: Configuration and documentation**"},{"line_number":680,"context_line":""},{"line_number":681,"context_line":"*Can proceed in parallel with Phase 4*"}],"source_content_type":"text/x-rst","patch_set":10,"id":"45470048_f55c94d3","line":678,"range":{"start_line":658,"start_character":0,"end_line":678,"end_character":1},"in_reply_to":"5af0be0e_48085e7a","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":695,"context_line":"* Update Cyborg configuration reference documentation"},{"line_number":696,"context_line":"* Add Tempest API tests for new microversion"},{"line_number":697,"context_line":"* Write release notes covering migration paths and new capabilities"},{"line_number":698,"context_line":""},{"line_number":699,"context_line":""},{"line_number":700,"context_line":"Dependencies"},{"line_number":701,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"}],"source_content_type":"text/x-rst","patch_set":10,"id":"122d3029_cc663ddf","line":698,"updated":"2026-06-03 16:06:13.000000000","message":"so only after all of this work is don can we add the api changes.\n\nthey shoudl eb the last thing to do \n\nfirst db and object chantges, then driver changes then rpc changes and finally api\n\nthe config options shoudl be added in teh patch that adds the code to use them.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":false,"context_lines":[{"line_number":695,"context_line":"* Update Cyborg configuration reference documentation"},{"line_number":696,"context_line":"* Add Tempest API tests for new microversion"},{"line_number":697,"context_line":"* Write release notes covering migration paths and new capabilities"},{"line_number":698,"context_line":""},{"line_number":699,"context_line":""},{"line_number":700,"context_line":"Dependencies"},{"line_number":701,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"}],"source_content_type":"text/x-rst","patch_set":10,"id":"adf90161_9c6d65ed","line":698,"in_reply_to":"122d3029_cc663ddf","updated":"2026-06-17 04:45:31.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":7166,"name":"Sylvain Bauza","email":"sbauza@redhat.com","username":"sbauza"},"change_message_id":"e589fdf9e9d0dbc3ea17a46ce3b62b05d2ecdefe","unresolved":true,"context_lines":[{"line_number":701,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":702,"context_line":""},{"line_number":703,"context_line":"* ``nvme-cli \u003e\u003d 1.5`` must be installed on compute nodes; Linux kernel 4.9+"},{"line_number":704,"context_line":"  with ``CONFIG_BLK_DEV_NVME`` required."},{"line_number":705,"context_line":""},{"line_number":706,"context_line":"* A privileged execution path appropriate for destructive media operations"},{"line_number":707,"context_line":"  (the project\u0027s existing privsep integration)."}],"source_content_type":"text/x-rst","patch_set":10,"id":"7d75913b_7a5a0008","line":704,"updated":"2026-05-20 15:35:40.000000000","message":"what would happen if computes don\u0027t have this dependency ? should we provide an exception and if so, when ? (hopefully when we start the agent, right?)","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":701,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":702,"context_line":""},{"line_number":703,"context_line":"* ``nvme-cli \u003e\u003d 1.5`` must be installed on compute nodes; Linux kernel 4.9+"},{"line_number":704,"context_line":"  with ``CONFIG_BLK_DEV_NVME`` required."},{"line_number":705,"context_line":""},{"line_number":706,"context_line":"* A privileged execution path appropriate for destructive media operations"},{"line_number":707,"context_line":"  (the project\u0027s existing privsep integration)."}],"source_content_type":"text/x-rst","patch_set":10,"id":"bd44d690_456481a0","line":704,"in_reply_to":"7d75913b_7a5a0008","updated":"2026-06-03 16:06:13.000000000","message":"i think the kernel versionis too old to really care about practically\n\nbut yes i think in init_host in the agent start up we can intoduce a driver init_host function and loop over all the enable drivers and call it\n\nthe nvme driver can validate teh nvme-cli verison and maybe do a kernel check \n\ni feel like checkign for CONFIG_BLK_DEV_NVME is overkill as if that is not compiled in the device discovery will jsut fail to fiand any nvme devices\n\nso i woudl just check that nvme-cli is present and \u003e\u003d 1.5 and leave it at that.\n\nwe do somethign very simpler in nova to check for swtpm or some other clis we required.\n\n\ncurrently many of the drvier depend on lspci for example and dont check for that so we can add a default noop implmeation of init_host in the generic driver and then follwo up with driver specific assertions in the future.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":701,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":702,"context_line":""},{"line_number":703,"context_line":"* ``nvme-cli \u003e\u003d 1.5`` must be installed on compute nodes; Linux kernel 4.9+"},{"line_number":704,"context_line":"  with ``CONFIG_BLK_DEV_NVME`` required."},{"line_number":705,"context_line":""},{"line_number":706,"context_line":"* A privileged execution path appropriate for destructive media operations"},{"line_number":707,"context_line":"  (the project\u0027s existing privsep integration)."}],"source_content_type":"text/x-rst","patch_set":10,"id":"dccf5ecd_7919222c","line":704,"in_reply_to":"bd44d690_456481a0","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":702,"context_line":""},{"line_number":703,"context_line":"* ``nvme-cli \u003e\u003d 1.5`` must be installed on compute nodes; Linux kernel 4.9+"},{"line_number":704,"context_line":"  with ``CONFIG_BLK_DEV_NVME`` required."},{"line_number":705,"context_line":""},{"line_number":706,"context_line":"* A privileged execution path appropriate for destructive media operations"},{"line_number":707,"context_line":"  (the project\u0027s existing privsep integration)."},{"line_number":708,"context_line":""},{"line_number":709,"context_line":"* **PCI device_spec parsing refactoring** — richer matching formats"},{"line_number":710,"context_line":"  (glob/dict/array) require shared PCI parsing logic. Until that blueprint"}],"source_content_type":"text/x-rst","patch_set":10,"id":"1c1a9281_47882fd6","line":707,"range":{"start_line":705,"start_character":1,"end_line":707,"end_character":47},"updated":"2026-06-03 16:06:13.000000000","message":"this is a pretty meaningless statement can we just delete htis.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":702,"context_line":""},{"line_number":703,"context_line":"* ``nvme-cli \u003e\u003d 1.5`` must be installed on compute nodes; Linux kernel 4.9+"},{"line_number":704,"context_line":"  with ``CONFIG_BLK_DEV_NVME`` required."},{"line_number":705,"context_line":""},{"line_number":706,"context_line":"* A privileged execution path appropriate for destructive media operations"},{"line_number":707,"context_line":"  (the project\u0027s existing privsep integration)."},{"line_number":708,"context_line":""},{"line_number":709,"context_line":"* **PCI device_spec parsing refactoring** — richer matching formats"},{"line_number":710,"context_line":"  (glob/dict/array) require shared PCI parsing logic. Until that blueprint"}],"source_content_type":"text/x-rst","patch_set":10,"id":"e4dd7e13_397e0a17","line":707,"range":{"start_line":705,"start_character":1,"end_line":707,"end_character":47},"in_reply_to":"1c1a9281_47882fd6","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"021287ad55d98a8a0e4369ab96195563565e4e5b","unresolved":true,"context_lines":[{"line_number":707,"context_line":"  (the project\u0027s existing privsep integration)."},{"line_number":708,"context_line":""},{"line_number":709,"context_line":"* **PCI device_spec parsing refactoring** — richer matching formats"},{"line_number":710,"context_line":"  (glob/dict/array) require shared PCI parsing logic. Until that blueprint"},{"line_number":711,"context_line":"  lands, only ``{\"vendor_id\": ..., \"product_id\": ...}`` is supported."},{"line_number":712,"context_line":""},{"line_number":713,"context_line":"* No new Python dependencies are strictly required beyond what Cyborg already"}],"source_content_type":"text/x-rst","patch_set":10,"id":"8513c32c_ed0ec290","line":710,"updated":"2026-06-02 10:16:48.000000000","message":"do we have this blueprint already? If so it would be good to link it here","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":707,"context_line":"  (the project\u0027s existing privsep integration)."},{"line_number":708,"context_line":""},{"line_number":709,"context_line":"* **PCI device_spec parsing refactoring** — richer matching formats"},{"line_number":710,"context_line":"  (glob/dict/array) require shared PCI parsing logic. Until that blueprint"},{"line_number":711,"context_line":"  lands, only ``{\"vendor_id\": ..., \"product_id\": ...}`` is supported."},{"line_number":712,"context_line":""},{"line_number":713,"context_line":"* No new Python dependencies are strictly required beyond what Cyborg already"}],"source_content_type":"text/x-rst","patch_set":10,"id":"ba3e512f_06778b61","line":710,"in_reply_to":"2e188a74_bd8995d7","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"978265488b8d3b555248a566a56886f281abd363","unresolved":true,"context_lines":[{"line_number":707,"context_line":"  (the project\u0027s existing privsep integration)."},{"line_number":708,"context_line":""},{"line_number":709,"context_line":"* **PCI device_spec parsing refactoring** — richer matching formats"},{"line_number":710,"context_line":"  (glob/dict/array) require shared PCI parsing logic. Until that blueprint"},{"line_number":711,"context_line":"  lands, only ``{\"vendor_id\": ..., \"product_id\": ...}`` is supported."},{"line_number":712,"context_line":""},{"line_number":713,"context_line":"* No new Python dependencies are strictly required beyond what Cyborg already"}],"source_content_type":"text/x-rst","patch_set":10,"id":"fded4b23_5dac0cbd","line":710,"in_reply_to":"8513c32c_ed0ec290","updated":"2026-06-02 11:48:05.000000000","message":"No no, I need to improve the wording. We are tracking it as a bug https://bugs.launchpad.net/openstack-cyborg/+bug/2152545","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":707,"context_line":"  (the project\u0027s existing privsep integration)."},{"line_number":708,"context_line":""},{"line_number":709,"context_line":"* **PCI device_spec parsing refactoring** — richer matching formats"},{"line_number":710,"context_line":"  (glob/dict/array) require shared PCI parsing logic. Until that blueprint"},{"line_number":711,"context_line":"  lands, only ``{\"vendor_id\": ..., \"product_id\": ...}`` is supported."},{"line_number":712,"context_line":""},{"line_number":713,"context_line":"* No new Python dependencies are strictly required beyond what Cyborg already"}],"source_content_type":"text/x-rst","patch_set":10,"id":"2e188a74_bd8995d7","line":710,"in_reply_to":"fded4b23_5dac0cbd","updated":"2026-06-03 16:06:13.000000000","message":"so the pci driver already has this capablity\n\nhttps://docs.openstack.org/cyborg/latest/configuration/drivers.html#generic-pci-driver\n\nand not supprot adress form day 1 is a hard blocker for me.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":710,"context_line":"  (glob/dict/array) require shared PCI parsing logic. Until that blueprint"},{"line_number":711,"context_line":"  lands, only ``{\"vendor_id\": ..., \"product_id\": ...}`` is supported."},{"line_number":712,"context_line":""},{"line_number":713,"context_line":"* No new Python dependencies are strictly required beyond what Cyborg already"},{"line_number":714,"context_line":"  carries for subprocess-based tooling and configuration."},{"line_number":715,"context_line":""},{"line_number":716,"context_line":""},{"line_number":717,"context_line":"Testing"}],"source_content_type":"text/x-rst","patch_set":10,"id":"9015e566_0c9e1321","line":714,"range":{"start_line":713,"start_character":2,"end_line":714,"end_character":57},"updated":"2026-06-03 16:06:13.000000000","message":"on then we can remvoe this line","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":710,"context_line":"  (glob/dict/array) require shared PCI parsing logic. Until that blueprint"},{"line_number":711,"context_line":"  lands, only ``{\"vendor_id\": ..., \"product_id\": ...}`` is supported."},{"line_number":712,"context_line":""},{"line_number":713,"context_line":"* No new Python dependencies are strictly required beyond what Cyborg already"},{"line_number":714,"context_line":"  carries for subprocess-based tooling and configuration."},{"line_number":715,"context_line":""},{"line_number":716,"context_line":""},{"line_number":717,"context_line":"Testing"}],"source_content_type":"text/x-rst","patch_set":10,"id":"0bd7fc2f_e92221c1","line":714,"range":{"start_line":713,"start_character":2,"end_line":714,"end_character":57},"in_reply_to":"9015e566_0c9e1321","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":721,"context_line":"  scenarios including timeout, unsupported capability, Placement call failure,"},{"line_number":722,"context_line":"  and agent crash."},{"line_number":723,"context_line":""},{"line_number":724,"context_line":"* Functional tests via qemu NVMe model in nested virt covering the full"},{"line_number":725,"context_line":"  lifecycle (discovery → bind → cleanup → re-allocation). If not feasible in"},{"line_number":726,"context_line":"  gate CI, explicitly tracked as a coverage gap in the implementation patch."},{"line_number":727,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"b73733f6_77bbf800","line":724,"range":{"start_line":724,"start_character":2,"end_line":724,"end_character":12},"updated":"2026-06-03 16:06:13.000000000","message":"let defer this to the implmetaiotn review\n\nim not sure fi i want to condire this a functional tst ro not\n\nos-vif has fucntional tests liek this that actully create port on ovs\nnova has a diffent defintion of fucntional tsts where we mock all exetnal denpencies with fixtures to simulate them\n\nboth have there use but im not sure if we want to use fucntional for both type of test or if this end ot end type testing shoudl use a diffent name","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":721,"context_line":"  scenarios including timeout, unsupported capability, Placement call failure,"},{"line_number":722,"context_line":"  and agent crash."},{"line_number":723,"context_line":""},{"line_number":724,"context_line":"* Functional tests via qemu NVMe model in nested virt covering the full"},{"line_number":725,"context_line":"  lifecycle (discovery → bind → cleanup → re-allocation). If not feasible in"},{"line_number":726,"context_line":"  gate CI, explicitly tracked as a coverage gap in the implementation patch."},{"line_number":727,"context_line":""}],"source_content_type":"text/x-rst","patch_set":10,"id":"25865740_1e702fe9","line":724,"range":{"start_line":724,"start_character":2,"end_line":724,"end_character":12},"in_reply_to":"b73733f6_77bbf800","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":726,"context_line":"  gate CI, explicitly tracked as a coverage gap in the implementation patch."},{"line_number":727,"context_line":""},{"line_number":728,"context_line":"* Tempest API tests for the new microversion, ``cleanup_failed`` field,"},{"line_number":729,"context_line":"  HTTP 409 on blocked ``enable`` when ``cleanup_failed\u003dTrue``."},{"line_number":730,"context_line":""},{"line_number":731,"context_line":"* Cryptographic erasure proof on real hardware may require **third-party**"},{"line_number":732,"context_line":"  or manual validation outside the default gate."}],"source_content_type":"text/x-rst","patch_set":10,"id":"625be68f_c95b7a3c","line":729,"updated":"2026-06-03 16:06:13.000000000","message":"so negitive api test liek this are actrully disucraged in tempest\n\nthey shoudl be validate via api sample tests and unit tests using tox as the priamry way of testing api behvior.\n\nthere have even been dicsusion of remvoign some ot the negitive api test form tempest in the past so we can test this in our plugin but we shoudl test it in the project driectly.\n\nhttps://docs.openstack.org/tempest/latest/field_guide/api.html#why-are-these-tests-in-tempest","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":726,"context_line":"  gate CI, explicitly tracked as a coverage gap in the implementation patch."},{"line_number":727,"context_line":""},{"line_number":728,"context_line":"* Tempest API tests for the new microversion, ``cleanup_failed`` field,"},{"line_number":729,"context_line":"  HTTP 409 on blocked ``enable`` when ``cleanup_failed\u003dTrue``."},{"line_number":730,"context_line":""},{"line_number":731,"context_line":"* Cryptographic erasure proof on real hardware may require **third-party**"},{"line_number":732,"context_line":"  or manual validation outside the default gate."}],"source_content_type":"text/x-rst","patch_set":10,"id":"0c916ce3_9fa72866","line":729,"in_reply_to":"625be68f_c95b7a3c","updated":"2026-06-17 15:27:52.000000000","message":"Acknowledged","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":727,"context_line":""},{"line_number":728,"context_line":"* Tempest API tests for the new microversion, ``cleanup_failed`` field,"},{"line_number":729,"context_line":"  HTTP 409 on blocked ``enable`` when ``cleanup_failed\u003dTrue``."},{"line_number":730,"context_line":""},{"line_number":731,"context_line":"* Cryptographic erasure proof on real hardware may require **third-party**"},{"line_number":732,"context_line":"  or manual validation outside the default gate."},{"line_number":733,"context_line":""},{"line_number":734,"context_line":""},{"line_number":735,"context_line":"Documentation Impact"}],"source_content_type":"text/x-rst","patch_set":10,"id":"67e3e94b_df9a3835","line":732,"range":{"start_line":730,"start_character":1,"end_line":732,"end_character":48},"updated":"2026-06-03 16:06:13.000000000","message":"on this i want to draw a clear supprot bondary between cybrog and nvme-cli\n\ncyborg is  responsible for invoking it and trusting its output\nnvme-cli and the vendor kernel drivers and firmware is responbile for the secuire rearaser at the physical level\n\nwe shoudl aim to have have 1st or 3rd party verifcaiton of this indirtly via the cybrog or whithebox tempest plugins.\n\nwe will not merge this feature without at least manual verifciaon by runing those test or manually providion a vm wrtieign data to the nvme devicve deleteing the vm and assertign its removed\n\nthe data cna be as simple as doing mkfs to lay down a file system or echoing a know set of bytes to the first 100 bytes of the disk and confirmign its present and removed.","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":727,"context_line":""},{"line_number":728,"context_line":"* Tempest API tests for the new microversion, ``cleanup_failed`` field,"},{"line_number":729,"context_line":"  HTTP 409 on blocked ``enable`` when ``cleanup_failed\u003dTrue``."},{"line_number":730,"context_line":""},{"line_number":731,"context_line":"* Cryptographic erasure proof on real hardware may require **third-party**"},{"line_number":732,"context_line":"  or manual validation outside the default gate."},{"line_number":733,"context_line":""},{"line_number":734,"context_line":""},{"line_number":735,"context_line":"Documentation Impact"}],"source_content_type":"text/x-rst","patch_set":10,"id":"8e9c01d6_5dfa1df3","line":732,"range":{"start_line":730,"start_character":1,"end_line":732,"end_character":48},"in_reply_to":"67e3e94b_df9a3835","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"78d6aa70398fceec0681c49652e2c855039e5341","unresolved":true,"context_lines":[{"line_number":735,"context_line":"Documentation Impact"},{"line_number":736,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":737,"context_line":""},{"line_number":738,"context_line":"* Cyborg configuration and administrator guides: enabling the driver,"},{"line_number":739,"context_line":"  ``device_spec``, timeouts, ``nvme-cli`` installation, coexistence"},{"line_number":740,"context_line":"  with other drivers, and recovery from cleanup failure using"},{"line_number":741,"context_line":"  ``cyborg-nvme-cleanup``."},{"line_number":742,"context_line":""},{"line_number":743,"context_line":""},{"line_number":744,"context_line":"References"}],"source_content_type":"text/x-rst","patch_set":10,"id":"9d76104a_e430e30a","line":741,"range":{"start_line":738,"start_character":0,"end_line":741,"end_character":25},"updated":"2026-06-03 16:06:13.000000000","message":"pelase rewrite this as prose not as a bullet list","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":false,"context_lines":[{"line_number":735,"context_line":"Documentation Impact"},{"line_number":736,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":737,"context_line":""},{"line_number":738,"context_line":"* Cyborg configuration and administrator guides: enabling the driver,"},{"line_number":739,"context_line":"  ``device_spec``, timeouts, ``nvme-cli`` installation, coexistence"},{"line_number":740,"context_line":"  with other drivers, and recovery from cleanup failure using"},{"line_number":741,"context_line":"  ``cyborg-nvme-cleanup``."},{"line_number":742,"context_line":""},{"line_number":743,"context_line":""},{"line_number":744,"context_line":"References"}],"source_content_type":"text/x-rst","patch_set":10,"id":"a4711274_cea62c38","line":741,"range":{"start_line":738,"start_character":0,"end_line":741,"end_character":25},"in_reply_to":"9d76104a_e430e30a","updated":"2026-06-17 15:27:52.000000000","message":"Done","commit_id":"c33b0567a0eaaa10cbc47f8daf7d0fbda321f9c7"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":true,"context_lines":[{"line_number":87,"context_line":"``NVMeDriver``::"},{"line_number":88,"context_line":""},{"line_number":89,"context_line":"        GenericDriver (cyborg/accelerator/drivers/driver.py)"},{"line_number":90,"context_line":"        ├── init_host()             → not implemented; subclasses override"},{"line_number":91,"context_line":"        ├── discover()              → subclasses override"},{"line_number":92,"context_line":"        └── cleanup(device)         → not implemented; subclasses override"},{"line_number":93,"context_line":""}],"source_content_type":"text/x-rst","patch_set":11,"id":"6e7f9b27_8e2be788","line":90,"range":{"start_line":90,"start_character":37,"end_line":90,"end_character":74},"updated":"2026-06-17 04:45:31.000000000","message":"im not entirly sure is we need this but its ok to add if w ehave a concreate usecase.\n\nwhen you say not implmented i woudl normally asusme that would raise an Nontimplemted expction that the manager woudl catch\n\n```\ndef init_host(self):\n    rasie NotImplemented()\n```\n\nthe altherivie woudl eb to implement it as \n```\ndef init_host(self):\n    pass\n```\nwhich is what i woudl have assuemd when you sadi a no-op before.\n\nthe former is more descriptive but slower.\nthe latter means the manager does not have a signal that the driver does nto supprot the methohd but we also may not care so either is fine.\n\nand we can leave the exeact details to the implemtion review if you prefer.\n\ni guess the other meaing could be that you not planing to implment it and interity but since you said \n\n  ├── discover()              → inherits PciDriver, filters NVMe\n  \nim assuming that is not the case","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":false,"context_lines":[{"line_number":87,"context_line":"``NVMeDriver``::"},{"line_number":88,"context_line":""},{"line_number":89,"context_line":"        GenericDriver (cyborg/accelerator/drivers/driver.py)"},{"line_number":90,"context_line":"        ├── init_host()             → not implemented; subclasses override"},{"line_number":91,"context_line":"        ├── discover()              → subclasses override"},{"line_number":92,"context_line":"        └── cleanup(device)         → not implemented; subclasses override"},{"line_number":93,"context_line":""}],"source_content_type":"text/x-rst","patch_set":11,"id":"e293dc8e_6ce083d5","line":90,"range":{"start_line":90,"start_character":37,"end_line":90,"end_character":74},"in_reply_to":"6e7f9b27_8e2be788","updated":"2026-06-19 04:57:23.000000000","message":"Fixed, \ninit_host() and cleanup() are no-op pass defaults, not raise NotImplementedError().  \n\n\nNVMeDriver.discover() is its own implementation (sysfs PCI enumeration filtered by device_spec with NVMe capability detection), not inherited from PciDriver. NVMeDriver is a standalone driver.","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":true,"context_lines":[{"line_number":170,"context_line":"    │  device_state → cleaning                        │"},{"line_number":171,"context_line":"    │  resolve NVMe device path from PCI address      │"},{"line_number":172,"context_line":"    │  dispatch cleanup to futurist thread pool       │"},{"line_number":173,"context_line":"    │  block on result with cleanup_timeout (900s)    │"},{"line_number":174,"context_line":"    │                                                 │"},{"line_number":175,"context_line":"    │  Use best available sanitize method:            │"},{"line_number":176,"context_line":"    │  ┌─ HW_NVME_CES → nvme sanitize -a 0x04        │"}],"source_content_type":"text/x-rst","patch_set":11,"id":"9a8ef7b1_b6bf9f91","line":173,"range":{"start_line":173,"start_character":44,"end_line":173,"end_character":51},"updated":"2026-06-17 04:45:31.000000000","message":"this should be a config option","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"b673c95737bca69a56a5bffb9ae658afe7cecb3a","unresolved":false,"context_lines":[{"line_number":170,"context_line":"    │  device_state → cleaning                        │"},{"line_number":171,"context_line":"    │  resolve NVMe device path from PCI address      │"},{"line_number":172,"context_line":"    │  dispatch cleanup to futurist thread pool       │"},{"line_number":173,"context_line":"    │  block on result with cleanup_timeout (900s)    │"},{"line_number":174,"context_line":"    │                                                 │"},{"line_number":175,"context_line":"    │  Use best available sanitize method:            │"},{"line_number":176,"context_line":"    │  ┌─ HW_NVME_CES → nvme sanitize -a 0x04        │"}],"source_content_type":"text/x-rst","patch_set":11,"id":"507711d8_d8c29ea3","line":173,"range":{"start_line":173,"start_character":44,"end_line":173,"end_character":51},"in_reply_to":"9a8ef7b1_b6bf9f91","updated":"2026-06-17 04:46:02.000000000","message":"this is covered later","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":true,"context_lines":[{"line_number":223,"context_line":"following the same pattern as Nova\u0027s libvirt driver which validates"},{"line_number":224,"context_line":"minimum libvirt and QEMU versions at startup. If ``nvme-cli`` is not"},{"line_number":225,"context_line":"found, the driver will log an error and skip NVMe device discovery"},{"line_number":226,"context_line":"gracefully rather than crashing the agent."},{"line_number":227,"context_line":""},{"line_number":228,"context_line":"Device Discovery"},{"line_number":229,"context_line":"----------------"}],"source_content_type":"text/x-rst","patch_set":11,"id":"7ebbc670_d21d9ac3","line":226,"updated":"2026-06-17 04:45:31.000000000","message":"so normlaly we woudl prefer a hard stop in this case\n\ni.e. if you configure nova for the libvirt driver and libvirt is not present or too old we would normally prevent startup\n\ni think that would be more correct to do here as well.\n\nwhat we woudl not do is prevent start up if a nvme device was not found\nbut if we have a runtime depencyle liek nvme-cli and the driver is confiugred ot use it if tis not precent that shoudl stop the agent.\n\nthe reason for the delta is hardware fialrues shoudl idelaly not prevent you form manging other devices but if if a tool we depeend on is not present we shoudl nto start and only find that out when we try to use it.","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":false,"context_lines":[{"line_number":223,"context_line":"following the same pattern as Nova\u0027s libvirt driver which validates"},{"line_number":224,"context_line":"minimum libvirt and QEMU versions at startup. If ``nvme-cli`` is not"},{"line_number":225,"context_line":"found, the driver will log an error and skip NVMe device discovery"},{"line_number":226,"context_line":"gracefully rather than crashing the agent."},{"line_number":227,"context_line":""},{"line_number":228,"context_line":"Device Discovery"},{"line_number":229,"context_line":"----------------"}],"source_content_type":"text/x-rst","patch_set":11,"id":"dc51eece_6b01570a","line":226,"in_reply_to":"7ebbc670_d21d9ac3","updated":"2026-06-19 04:57:23.000000000","message":"Updated. \n\nIf nvme-cli is not installed and the NVMe driver is configured in [agent] enabled_drivers, init_host() raises an exception and prevents agent startup, following Nova\u0027s InvalidConfiguration pattern for swtpm. \n  \nMissing hardware (no NVMe devices found) is handled gracefully by returning an empty list from discover().","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":true,"context_lines":[{"line_number":282,"context_line":"considered invalid. The Cyborg agent will fail to start up if more than"},{"line_number":283,"context_line":"one driver returns the same device, raising a new"},{"line_number":284,"context_line":"``InvalidConfiguration`` exception. No new development will be done on"},{"line_number":285,"context_line":"the deprecated drivers outside of bug fixes."},{"line_number":286,"context_line":""},{"line_number":287,"context_line":"NVMe Device Type"},{"line_number":288,"context_line":"----------------"}],"source_content_type":"text/x-rst","patch_set":11,"id":"ae13475c_a64b295a","line":285,"updated":"2026-06-17 04:45:31.000000000","message":"+1\n\nwe may delay the removal of the inspur driver if we have not implemtned resize\nfor flavore with cyborg device or come up with a diffent upgrade mechanisum but you have capture the intent correctly.\n\nit can remian deprecated indefintly until we have a good upgrade story","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":false,"context_lines":[{"line_number":282,"context_line":"considered invalid. The Cyborg agent will fail to start up if more than"},{"line_number":283,"context_line":"one driver returns the same device, raising a new"},{"line_number":284,"context_line":"``InvalidConfiguration`` exception. No new development will be done on"},{"line_number":285,"context_line":"the deprecated drivers outside of bug fixes."},{"line_number":286,"context_line":""},{"line_number":287,"context_line":"NVMe Device Type"},{"line_number":288,"context_line":"----------------"}],"source_content_type":"text/x-rst","patch_set":11,"id":"e28ccf06_bbd1a6cb","line":285,"in_reply_to":"ae13475c_a64b295a","updated":"2026-06-30 19:46:58.000000000","message":"Acknowledged","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":true,"context_lines":[{"line_number":322,"context_line":"    Resource provider: compute-1_0000:01:00.0"},{"line_number":323,"context_line":"    Resource class:    CUSTOM_NVME_8086_0001"},{"line_number":324,"context_line":"    Inventory:         total\u003d1"},{"line_number":325,"context_line":"    Traits:            OWNER_CYBORG, HW_NVME_CES, HW_NVME_BES"},{"line_number":326,"context_line":""},{"line_number":327,"context_line":"Cleanup on Unbind"},{"line_number":328,"context_line":"-----------------"}],"source_content_type":"text/x-rst","patch_set":11,"id":"975000ee_cea00fd6","line":325,"updated":"2026-06-17 04:45:31.000000000","message":"+1\n\nthanks that helps and alignes to what i was expecting\n\nas an aside we shoudl likely guard agasitn the case that the RP exists but does not have the OWNER_CYBORG trait and treat that as an error.\n\nthat shoudl not happen but could happen if the deivce is currenly manage by nova.","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":false,"context_lines":[{"line_number":322,"context_line":"    Resource provider: compute-1_0000:01:00.0"},{"line_number":323,"context_line":"    Resource class:    CUSTOM_NVME_8086_0001"},{"line_number":324,"context_line":"    Inventory:         total\u003d1"},{"line_number":325,"context_line":"    Traits:            OWNER_CYBORG, HW_NVME_CES, HW_NVME_BES"},{"line_number":326,"context_line":""},{"line_number":327,"context_line":"Cleanup on Unbind"},{"line_number":328,"context_line":"-----------------"}],"source_content_type":"text/x-rst","patch_set":11,"id":"5f21768c_41f2ffc5","line":325,"in_reply_to":"975000ee_cea00fd6","updated":"2026-06-19 04:57:23.000000000","message":"Done. During discovery, if an RP already exists for a PCI address but lacks the OWNER_CYBORG trait, the driver logs an error and skips that device.","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":true,"context_lines":[{"line_number":339,"context_line":"``device_state`` to ``cleaning``, resolves the NVMe device path from"},{"line_number":340,"context_line":"the PCI address via sysfs, and dispatches the cleanup to the thread"},{"line_number":341,"context_line":"pool with a configurable timeout (``[nvme] cleanup_timeout``, default"},{"line_number":342,"context_line":"900 seconds). All nvme-cli commands run under privsep"},{"line_number":343,"context_line":"(``sys_admin_pctxt``) with per-device locking"},{"line_number":344,"context_line":"(``@utils.synchronized()``) to prevent concurrent cleanup on the same"},{"line_number":345,"context_line":"device."}],"source_content_type":"text/x-rst","patch_set":11,"id":"bbfab3c2_360f20fa","line":342,"updated":"2026-06-17 04:45:31.000000000","message":"ah cool.\n\ncan you factor this out inot a code block to make this a litel clearer\n\n```\n.. code::\n\n  [nvme]\n  cleanup_timeout\u003d900\n```","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":false,"context_lines":[{"line_number":339,"context_line":"``device_state`` to ``cleaning``, resolves the NVMe device path from"},{"line_number":340,"context_line":"the PCI address via sysfs, and dispatches the cleanup to the thread"},{"line_number":341,"context_line":"pool with a configurable timeout (``[nvme] cleanup_timeout``, default"},{"line_number":342,"context_line":"900 seconds). All nvme-cli commands run under privsep"},{"line_number":343,"context_line":"(``sys_admin_pctxt``) with per-device locking"},{"line_number":344,"context_line":"(``@utils.synchronized()``) to prevent concurrent cleanup on the same"},{"line_number":345,"context_line":"device."}],"source_content_type":"text/x-rst","patch_set":11,"id":"387abcbc_80109b13","line":342,"in_reply_to":"bbfab3c2_360f20fa","updated":"2026-06-19 04:57:23.000000000","message":"Done","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"62abc69ea2d4fe61e8f70b6202e8328eb4f110c5","unresolved":true,"context_lines":[{"line_number":340,"context_line":"the PCI address via sysfs, and dispatches the cleanup to the thread"},{"line_number":341,"context_line":"pool with a configurable timeout (``[nvme] cleanup_timeout``, default"},{"line_number":342,"context_line":"900 seconds). All nvme-cli commands run under privsep"},{"line_number":343,"context_line":"(``sys_admin_pctxt``) with per-device locking"},{"line_number":344,"context_line":"(``@utils.synchronized()``) to prevent concurrent cleanup on the same"},{"line_number":345,"context_line":"device."},{"line_number":346,"context_line":""}],"source_content_type":"text/x-rst","patch_set":11,"id":"067c71eb_101bb653","line":343,"range":{"start_line":343,"start_character":0,"end_line":343,"end_character":2},"updated":"2026-06-17 04:45:31.000000000","message":"do we actully knoew if CAP_SYS_ADMIN is really need of if a less set of capablites woudl sufice\n\nthis is fine if its really requried but the less the beter..","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":true,"context_lines":[{"line_number":340,"context_line":"the PCI address via sysfs, and dispatches the cleanup to the thread"},{"line_number":341,"context_line":"pool with a configurable timeout (``[nvme] cleanup_timeout``, default"},{"line_number":342,"context_line":"900 seconds). All nvme-cli commands run under privsep"},{"line_number":343,"context_line":"(``sys_admin_pctxt``) with per-device locking"},{"line_number":344,"context_line":"(``@utils.synchronized()``) to prevent concurrent cleanup on the same"},{"line_number":345,"context_line":"device."},{"line_number":346,"context_line":""}],"source_content_type":"text/x-rst","patch_set":11,"id":"4ab34602_82b510a4","line":343,"range":{"start_line":343,"start_character":0,"end_line":343,"end_character":2},"in_reply_to":"067c71eb_101bb653","updated":"2026-06-19 04:57:23.000000000","message":"In my Devstack libvirt nvme setup vm,\n\nI have verified it CAP_SYS_ADMIN is the minimum required. \n\nThe NVMe controller device (/dev/nvmeN) is root-only (mode 0600), and the kernel checks capable(CAP_SYS_ADMIN) for admin ioctls — no lesser capability works. \n\nWe reuse the existing sys_admin_pctxt. Splitting to a more minimal privsep context can be done as a follow-up.","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":false,"context_lines":[{"line_number":340,"context_line":"the PCI address via sysfs, and dispatches the cleanup to the thread"},{"line_number":341,"context_line":"pool with a configurable timeout (``[nvme] cleanup_timeout``, default"},{"line_number":342,"context_line":"900 seconds). All nvme-cli commands run under privsep"},{"line_number":343,"context_line":"(``sys_admin_pctxt``) with per-device locking"},{"line_number":344,"context_line":"(``@utils.synchronized()``) to prevent concurrent cleanup on the same"},{"line_number":345,"context_line":"device."},{"line_number":346,"context_line":""}],"source_content_type":"text/x-rst","patch_set":11,"id":"c6b00446_b6b71bb3","line":343,"range":{"start_line":343,"start_character":0,"end_line":343,"end_character":2},"in_reply_to":"4ab34602_82b510a4","updated":"2026-06-30 19:46:58.000000000","message":"yes that spliting is someitng i want to do btu not as part fo this work\nwe can dicuss that next cycle\n\nthanks for confirming","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":true,"context_lines":[{"line_number":700,"context_line":"development will be done on the vendor-specific drivers outside of bug"},{"line_number":701,"context_line":"fixes. The drivers will not be removed until Nova supports resizing"},{"line_number":702,"context_line":"instances with Cyborg-managed devices, ensuring operators have a"},{"line_number":703,"context_line":"migration path. Upgrade documentation will include examples of how to"},{"line_number":704,"context_line":"manage Inspur devices with the new generic driver."},{"line_number":705,"context_line":""},{"line_number":706,"context_line":"Cyborg does not support rolling upgrades today. All services must be"},{"line_number":707,"context_line":"upgraded together: conductor and API first, then agents. If an old"}],"source_content_type":"text/x-rst","patch_set":11,"id":"e62a24d9_192cb956","line":704,"range":{"start_line":703,"start_character":15,"end_line":704,"end_character":42},"updated":"2026-06-17 15:27:52.000000000","message":"we can decided where that goes but im thinking this is more applicable to the deriver docs","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":false,"context_lines":[{"line_number":700,"context_line":"development will be done on the vendor-specific drivers outside of bug"},{"line_number":701,"context_line":"fixes. The drivers will not be removed until Nova supports resizing"},{"line_number":702,"context_line":"instances with Cyborg-managed devices, ensuring operators have a"},{"line_number":703,"context_line":"migration path. Upgrade documentation will include examples of how to"},{"line_number":704,"context_line":"manage Inspur devices with the new generic driver."},{"line_number":705,"context_line":""},{"line_number":706,"context_line":"Cyborg does not support rolling upgrades today. All services must be"},{"line_number":707,"context_line":"upgraded together: conductor and API first, then agents. If an old"}],"source_content_type":"text/x-rst","patch_set":11,"id":"bd5db9fa_c3f43575","line":704,"range":{"start_line":703,"start_character":15,"end_line":704,"end_character":42},"in_reply_to":"e62a24d9_192cb956","updated":"2026-06-19 04:57:23.000000000","message":"Done, migration examples will be in the driver documentation","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":true,"context_lines":[{"line_number":708,"context_line":"agent receives the new ``cleanup_device`` RPC cast, the call fails"},{"line_number":709,"context_line":"silently and the device stays in ``pending_cleaning`` until the agent"},{"line_number":710,"context_line":"is upgraded. On restart, ``init_host()`` detects the device and moves"},{"line_number":711,"context_line":"it to ``error`` for operator remediation."},{"line_number":712,"context_line":""},{"line_number":713,"context_line":""},{"line_number":714,"context_line":"Implementation"}],"source_content_type":"text/x-rst","patch_set":11,"id":"081da43d_b788c161","line":711,"updated":"2026-06-17 15:27:52.000000000","message":"so while this is technically true we want to make progress towards fixing that","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":false,"context_lines":[{"line_number":708,"context_line":"agent receives the new ``cleanup_device`` RPC cast, the call fails"},{"line_number":709,"context_line":"silently and the device stays in ``pending_cleaning`` until the agent"},{"line_number":710,"context_line":"is upgraded. On restart, ``init_host()`` detects the device and moves"},{"line_number":711,"context_line":"it to ``error`` for operator remediation."},{"line_number":712,"context_line":""},{"line_number":713,"context_line":""},{"line_number":714,"context_line":"Implementation"}],"source_content_type":"text/x-rst","patch_set":11,"id":"a243a8c3_3f1e4fac","line":711,"in_reply_to":"081da43d_b788c161","updated":"2026-06-19 04:57:23.000000000","message":"I have improved the wording.\n\nCyborg does not yet have full rolling upgrade support, though grenade testing has been added recently and work is planned for the 2027.1 cycle to support N-1 agent compatibility.","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"6ec310a2e8f810269a651f783e1f386837663403","unresolved":true,"context_lines":[{"line_number":718,"context_line":"-----------"},{"line_number":719,"context_line":""},{"line_number":720,"context_line":"Primary assignee:"},{"line_number":721,"context_line":"  Chandan Kumar (chkumar246)"},{"line_number":722,"context_line":""},{"line_number":723,"context_line":"Other contributors:"},{"line_number":724,"context_line":"  None"}],"source_content_type":"text/x-rst","patch_set":11,"id":"e38723e6_2cc805e0","line":721,"range":{"start_line":721,"start_character":17,"end_line":721,"end_character":27},"updated":"2026-06-17 15:27:52.000000000","message":"nit: that should be your irc nic normally\n\nits not super imporant but that is the convention","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"efeff99ab500f2f0bd0184246af56ce4959f4c19","unresolved":false,"context_lines":[{"line_number":718,"context_line":"-----------"},{"line_number":719,"context_line":""},{"line_number":720,"context_line":"Primary assignee:"},{"line_number":721,"context_line":"  Chandan Kumar (chkumar246)"},{"line_number":722,"context_line":""},{"line_number":723,"context_line":"Other contributors:"},{"line_number":724,"context_line":"  None"}],"source_content_type":"text/x-rst","patch_set":11,"id":"5ab77f2e_9b2ec38a","line":721,"range":{"start_line":721,"start_character":17,"end_line":721,"end_character":27},"in_reply_to":"e38723e6_2cc805e0","updated":"2026-06-19 04:57:23.000000000","message":"Done","commit_id":"105dd4f23db7a7d92fbaffe58e7e92aaa76107f5"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"4dc771e9eef187a82999ba8682dbea8e05971713","unresolved":true,"context_lines":[{"line_number":366,"context_line":"per-device locking (``@utils.synchronized()``) to prevent concurrent"},{"line_number":367,"context_line":"cleanup on the same device."},{"line_number":368,"context_line":""},{"line_number":369,"context_line":"The agent selects the best available sanitize method based on the"},{"line_number":370,"context_line":"device capabilities discovered at enumeration time. If the expected"},{"line_number":371,"context_line":"method fails, the device moves to ``error`` state. There is no silent"},{"line_number":372,"context_line":"fallback to a weaker method; operators can investigate and re-trigger"},{"line_number":373,"context_line":"cleanup via ``POST /v2/devices/{uuid}/clean``."},{"line_number":374,"context_line":""},{"line_number":375,"context_line":"The fallback strategy for devices that do not support any ``nvme"},{"line_number":376,"context_line":"sanitize`` variant (CES, BES, or OWS) or where ``nvme format`` also"},{"line_number":377,"context_line":"fails is under discussion. Host-side tools such as ``shred`` or"},{"line_number":378,"context_line":"``dd`` operate at the logical block level, but it is unclear how much"},{"line_number":379,"context_line":"data they can effectively erase on NVMe media given the Flash"},{"line_number":380,"context_line":"Translation Layer, wear leveling, and over-provisioned areas. Open"},{"line_number":381,"context_line":"questions:"},{"line_number":382,"context_line":""},{"line_number":383,"context_line":"1. Can ``shred`` or ``dd`` guarantee secure erasure on NVMe devices,"},{"line_number":384,"context_line":"   or do they leave residual data in remapped and over-provisioned"},{"line_number":385,"context_line":"   flash cells?"},{"line_number":386,"context_line":""},{"line_number":387,"context_line":"2. Should devices without sanitize support simply move to ``error``"},{"line_number":388,"context_line":"   state and let the operator decide, rather than providing a fallback"},{"line_number":389,"context_line":"   that may give a false sense of security?"},{"line_number":390,"context_line":""},{"line_number":391,"context_line":"3. If a \"best effort\" fallback is offered, should it require explicit"},{"line_number":392,"context_line":"   operator opt-in via configuration?"},{"line_number":393,"context_line":""},{"line_number":394,"context_line":"Device State Machine"},{"line_number":395,"context_line":"--------------------"}],"source_content_type":"text/x-rst","patch_set":12,"id":"b8cd5ee5_c7b85fee","line":392,"range":{"start_line":369,"start_character":0,"end_line":392,"end_character":37},"updated":"2026-06-19 05:08:34.000000000","message":"I will be updating this section based on final decision.","commit_id":"800d6bb1e9f8fb955e648798f49a1b9445e57406"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b65481c2244c0e91192130e709bbffab6db0e09f","unresolved":false,"context_lines":[{"line_number":366,"context_line":"per-device locking (``@utils.synchronized()``) to prevent concurrent"},{"line_number":367,"context_line":"cleanup on the same device."},{"line_number":368,"context_line":""},{"line_number":369,"context_line":"The agent selects the best available sanitize method based on the"},{"line_number":370,"context_line":"device capabilities discovered at enumeration time. If the expected"},{"line_number":371,"context_line":"method fails, the device moves to ``error`` state. There is no silent"},{"line_number":372,"context_line":"fallback to a weaker method; operators can investigate and re-trigger"},{"line_number":373,"context_line":"cleanup via ``POST /v2/devices/{uuid}/clean``."},{"line_number":374,"context_line":""},{"line_number":375,"context_line":"The fallback strategy for devices that do not support any ``nvme"},{"line_number":376,"context_line":"sanitize`` variant (CES, BES, or OWS) or where ``nvme format`` also"},{"line_number":377,"context_line":"fails is under discussion. Host-side tools such as ``shred`` or"},{"line_number":378,"context_line":"``dd`` operate at the logical block level, but it is unclear how much"},{"line_number":379,"context_line":"data they can effectively erase on NVMe media given the Flash"},{"line_number":380,"context_line":"Translation Layer, wear leveling, and over-provisioned areas. Open"},{"line_number":381,"context_line":"questions:"},{"line_number":382,"context_line":""},{"line_number":383,"context_line":"1. Can ``shred`` or ``dd`` guarantee secure erasure on NVMe devices,"},{"line_number":384,"context_line":"   or do they leave residual data in remapped and over-provisioned"},{"line_number":385,"context_line":"   flash cells?"},{"line_number":386,"context_line":""},{"line_number":387,"context_line":"2. Should devices without sanitize support simply move to ``error``"},{"line_number":388,"context_line":"   state and let the operator decide, rather than providing a fallback"},{"line_number":389,"context_line":"   that may give a false sense of security?"},{"line_number":390,"context_line":""},{"line_number":391,"context_line":"3. If a \"best effort\" fallback is offered, should it require explicit"},{"line_number":392,"context_line":"   operator opt-in via configuration?"},{"line_number":393,"context_line":""},{"line_number":394,"context_line":"Device State Machine"},{"line_number":395,"context_line":"--------------------"}],"source_content_type":"text/x-rst","patch_set":12,"id":"d634cd07_8ac21cff","line":392,"range":{"start_line":369,"start_character":0,"end_line":392,"end_character":37},"in_reply_to":"b8cd5ee5_c7b85fee","updated":"2026-06-24 06:03:25.000000000","message":"Done","commit_id":"800d6bb1e9f8fb955e648798f49a1b9445e57406"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"60b13a0bb7da8c04abfe6e28b00458819a26d152","unresolved":false,"context_lines":[{"line_number":280,"context_line":"The traits are defined in a new ``os_traits/hw/nvme/__init__.py``"},{"line_number":281,"context_line":"module::"},{"line_number":282,"context_line":""},{"line_number":283,"context_line":"    TRAITS \u003d ["},{"line_number":284,"context_line":"        \u0027CES\u0027,   # Crypto Erase Sanitize Supported (sanicap bit 0)"},{"line_number":285,"context_line":"        \u0027BES\u0027,   # Block Erase Sanitize Supported (sanicap bit 1)"},{"line_number":286,"context_line":"        \u0027WZS\u0027,   # Write Zeroes Supported (ONCS bit 3)"}],"source_content_type":"text/x-rst","patch_set":15,"id":"fbf94f10_fbfa2bd7","line":283,"updated":"2026-06-23 14:44:38.000000000","message":"The trait names in the os-traits code block (CES, BES, WZS) are shown as bare abbreviations but the Placement integration example uses HW_NVME_CES, HW_NVME_BES, HW_NVME_WZS prefixed names. The module should show full trait names.\n\n**Severity**: SUGGESTION | **Confidence**: 0.9\n\n**Benefit**: Implementers reading the os-traits code block will know the exact trait string to register in Placement, avoiding a mismatch between the module listing and the Placement names used later.\n\n**Recommendation**:\nUpdate the TRAITS list to show the full os-traits standard names (e.g., HW_NVME_CES) matching what appears in the Placement integration example at line 406.","commit_id":"dceca2a5eaabb579f37a88c0c08898cd09e87764"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"60b13a0bb7da8c04abfe6e28b00458819a26d152","unresolved":false,"context_lines":[{"line_number":335,"context_line":"    | zero           | crypto      | invalid configuration                    |"},{"line_number":336,"context_line":"    +----------------+-------------+------------------------------------------+"},{"line_number":337,"context_line":""},{"line_number":338,"context_line":"The resolution chain within the ``zero`` method prefers ``nvme"},{"line_number":339,"context_line":"write-zeroes`` (controller-side, if ``WZS`` is supported) over"},{"line_number":340,"context_line":"host-side zeroing via ``shred``, following Nova\u0027s ``volume_clear``"},{"line_number":341,"context_line":"pattern for LVM volumes."}],"source_content_type":"text/x-rst","patch_set":15,"id":"40be132d_11a8a3f2","line":338,"updated":"2026-06-23 14:44:38.000000000","message":"The \u0027no silent fallback to weaker erase methods\u0027 principle is contradicted by the zero-path fallback chain: nvme write-zeroes else shred. shred is host-side user-space overwrite weaker than controller-side write-zeroes, with no guarantee against wear-leveling remapping.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: Operators selecting clear_method\u003dzero on hardware lacking WZS support silently get a weaker shred-based erase. This contradicts the spec\u0027s stated design principle of no silent fallback.\n\n**Suggestion**:\nEither treat shred as a distinct explicitly-named clear_mode operators must opt into, or add a clear statement in the Security impact section that the zero-path shred fallback is weaker than controller-side erase and document the residual risk.","commit_id":"dceca2a5eaabb579f37a88c0c08898cd09e87764"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"60b13a0bb7da8c04abfe6e28b00458819a26d152","unresolved":false,"context_lines":[{"line_number":504,"context_line":"``status`` field which remains exclusively for the enabled/disabled"},{"line_number":505,"context_line":"scheduling control."},{"line_number":506,"context_line":""},{"line_number":507,"context_line":"The state transitions are::"},{"line_number":508,"context_line":""},{"line_number":509,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":510,"context_line":"    allocated (reserved\u003dtotal) ──(agent receives cleanup RPC)──► pending_cleaning (reserved\u003dtotal)"}],"source_content_type":"text/x-rst","patch_set":15,"id":"e0403793_fb540c2a","line":507,"updated":"2026-06-23 14:44:38.000000000","message":"The state machine does not define a transition from pending_cleaning on timeout. cleanup_timeout is only enforced via future.result(timeout) after cleanup starts (cleaning state). If the dispatch to the thread pool blocks, no timeout transitions pending_cleaning to error.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: A device stuck in pending_cleaning due to thread pool exhaustion or pre-execution failure stays indefinitely (reserved\u003dtotal) until the next init_host crash recovery cycle.\n\n**Suggestion**:\nAdd an explicit timeout or periodic sweep for devices in pending_cleaning longer than a threshold, transitioning them to error if cleanup has not progressed to cleaning within a bounded interval.","commit_id":"dceca2a5eaabb579f37a88c0c08898cd09e87764"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"60b13a0bb7da8c04abfe6e28b00458819a26d152","unresolved":false,"context_lines":[{"line_number":540,"context_line":"``init_host()`` so operators can investigate and manually correct the"},{"line_number":541,"context_line":"state."},{"line_number":542,"context_line":""},{"line_number":543,"context_line":"Device Bind"},{"line_number":544,"context_line":"^^^^^^^^^^^"},{"line_number":545,"context_line":""},{"line_number":546,"context_line":"When the Cyborg conductor binds an NVMe device to an instance, it sets"}],"source_content_type":"text/x-rst","patch_set":15,"id":"5d43e3d3_14ad432c","line":543,"updated":"2026-06-23 14:44:38.000000000","message":"The bind guard (device_state\u003d\u003davailable) runs in the conductor while all device_state transitions are managed by the agent in a separate process. This creates a TOCTOU race between the check and the reserved\u003dtotal update with no atomicity guarantee specified.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: A device could be allocated to two instances simultaneously, or a device still in cleaning could pass the stale check and be bound with residual tenant data, defeating the core security guarantee.\n\n**Priority**: Before merge\n**Why This Matters**: The entire security model depends on a device never being allocated while not in available state. Without atomicity, the cross-process guard is advisory and can be bypassed under concurrency.\n\n**Recommendation**:\nSpecify that the bind path uses an atomic conditional DB update (e.g., UPDATE devices SET device_state\u003d\u0027allocated\u0027 WHERE uuid\u003d? AND device_state\u003d\u0027available\u0027) and only proceeds if rowcount is 1. Alternatively document a distributed lock held across the bind-and-reserve sequence.","commit_id":"dceca2a5eaabb579f37a88c0c08898cd09e87764"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"60b13a0bb7da8c04abfe6e28b00458819a26d152","unresolved":false,"context_lines":[{"line_number":643,"context_line":""},{"line_number":644,"context_line":""},{"line_number":645,"context_line":"REST API impact"},{"line_number":646,"context_line":"---------------"},{"line_number":647,"context_line":""},{"line_number":648,"context_line":"A new microversion is required. The exact version number depends on"},{"line_number":649,"context_line":"merge ordering with other in-flight specs; this spec claims the next"}],"source_content_type":"text/x-rst","patch_set":15,"id":"7bf3770a_bf781904","line":646,"updated":"2026-06-23 14:44:38.000000000","message":"The microversion is left unspecified (\u0027the next available microversion\u0027). The commit message references microversion 2.4, but the REST API section avoids committing to a number, creating ambiguity for cross-spec coordination.\n\n**Severity**: SUGGESTION | **Confidence**: 0.8\n\n**Benefit**: A concrete or tentatively reserved microversion helps coordinate with other in-flight API changes and gives reviewers a concrete reference point for the device_state field introduction.\n\n**Recommendation**:\nState a target microversion (e.g., \u00272.4, subject to change based on merge ordering\u0027) to anchor the discussion, consistent with how other Cyborg specs specify their microversion.","commit_id":"dceca2a5eaabb579f37a88c0c08898cd09e87764"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"60b13a0bb7da8c04abfe6e28b00458819a26d152","unresolved":false,"context_lines":[{"line_number":703,"context_line":"cleanup after a failure."},{"line_number":704,"context_line":""},{"line_number":705,"context_line":"Normal response code: ``202 Accepted``"},{"line_number":706,"context_line":""},{"line_number":707,"context_line":"Error response codes:"},{"line_number":708,"context_line":""},{"line_number":709,"context_line":"* ``404 Not Found`` — device UUID does not exist"}],"source_content_type":"text/x-rst","patch_set":15,"id":"0b8aef34_9e1e8388","line":706,"updated":"2026-06-23 14:44:38.000000000","message":"The spec does not specify a 409 Conflict response for POST /clean when the device is in \u0027available\u0027 state. The error codes list 409 for allocated, cleaning, and pending_cleaning but omit available, leaving its behavior undefined.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: An operator hitting POST /clean on a device that is already available gets undefined behavior. The endpoint is described as a recovery mechanism for error state.\n\n**Suggestion**:\nAdd explicit handling for device_state\u003d\u003davailable: either return 409 Conflict (\u0027device is already available\u0027) or document as a 204 No Content no-op. Also clarify the primary success case (error to pending_cleaning).","commit_id":"dceca2a5eaabb579f37a88c0c08898cd09e87764"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"472574265830d0304543a1f0919d14757cd13b2e","unresolved":false,"context_lines":[{"line_number":280,"context_line":"The traits are defined in a new ``os_traits/hw/nvme/__init__.py``"},{"line_number":281,"context_line":"module::"},{"line_number":282,"context_line":""},{"line_number":283,"context_line":"    TRAITS \u003d ["},{"line_number":284,"context_line":"        \u0027CES\u0027,   # Crypto Erase Sanitize Supported (sanicap bit 0)"},{"line_number":285,"context_line":"        \u0027BES\u0027,   # Block Erase Sanitize Supported (sanicap bit 1)"},{"line_number":286,"context_line":"        \u0027WZS\u0027,   # Write Zeroes Supported (ONCS bit 3)"}],"source_content_type":"text/x-rst","patch_set":16,"id":"5ab83d70_9e588dea","line":283,"updated":"2026-06-23 15:25:48.000000000","message":"os-traits TRAITS list uses bare names \u0027CES\u0027, \u0027BES\u0027, \u0027WZS\u0027 but the Placement example shows fully qualified \u0027HW_NVME_CES\u0027, \u0027HW_NVME_BES\u0027, \u0027HW_NVME_WZS\u0027. The os-traits convention requires fully qualified trait names in the TRAITS constant.\n\n**Severity**: HIGH | **Confidence**: 0.9\n\n**Risk**: If implemented as written, the TRAITS constant would register bare names that collide with the os-traits namespace convention and would not match the HW_NVME_ prefixed traits reported to Placement, causing a Placement trait mismatch.\n\n**Priority**: Before merge\n**Why This Matters**: os-traits uses a hierarchical naming scheme (e.g., HW_NIC_SRIOV, HW_CPU_X86_AVX2). The TRAITS list must match the actual trait names used in Placement. The code example as written would produce incorrect trait registration if implemented literally.\n\n**Recommendation**:\nUpdate the TRAITS list to use fully qualified names matching the Placement example: \u0027HW_NVME_CES\u0027, \u0027HW_NVME_BES\u0027, \u0027HW_NVME_WZS\u0027. Alternatively, clarify that the list shows short labels and the actual os-traits module maps them to HW_NVME_ prefixed names.","commit_id":"43ead47d9364497a3395d37fa8261ac03ec38f11"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"163347f132ba133539bf8a61e7b19b724445a666","unresolved":false,"context_lines":[{"line_number":280,"context_line":"The traits are defined in a new ``os_traits/hw/nvme/__init__.py``"},{"line_number":281,"context_line":"module::"},{"line_number":282,"context_line":""},{"line_number":283,"context_line":"    TRAITS \u003d ["},{"line_number":284,"context_line":"        \u0027CES\u0027,   # Crypto Erase Sanitize Supported (sanicap bit 0)"},{"line_number":285,"context_line":"        \u0027BES\u0027,   # Block Erase Sanitize Supported (sanicap bit 1)"},{"line_number":286,"context_line":"        \u0027WZS\u0027,   # Write Zeroes Supported (ONCS bit 3)"}],"source_content_type":"text/x-rst","patch_set":16,"id":"e790587d_4dac57a2","line":283,"in_reply_to":"5ab83d70_9e588dea","updated":"2026-06-23 16:02:44.000000000","message":"no this is already correct\n\nin os-traits the constnat are build by walking the module stucture recursivly","commit_id":"43ead47d9364497a3395d37fa8261ac03ec38f11"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"472574265830d0304543a1f0919d14757cd13b2e","unresolved":false,"context_lines":[{"line_number":356,"context_line":"------------------"},{"line_number":357,"context_line":""},{"line_number":358,"context_line":"The existing SSD and Inspur drivers will be deprecated in this release"},{"line_number":359,"context_line":"and will be removed in 2027.2 in favour of the generic NVMe driver. The"},{"line_number":360,"context_line":"deprecated drivers and the new generic NVMe driver can coexist on the"},{"line_number":361,"context_line":"same system as long as they manage independent devices. For any given"},{"line_number":362,"context_line":"device, managing it with more than one driver simultaneously is"}],"source_content_type":"text/x-rst","patch_set":16,"id":"1ca31a9d_195e88d9","line":359,"updated":"2026-06-23 15:25:48.000000000","message":"Deprecation removal timeline is contradictory: \u0027will be removed in 2027.2\u0027 vs \u0027will not be removed until Nova supports resizing instances with Cyborg-managed devices\u0027. These are two independent removal conditions that may never align.\n\n**Severity**: HIGH | **Confidence**: 0.9\n\n**Risk**: Operators and reviewers cannot determine the actual deprecation commitment. If 2027.2 arrives but Nova does not yet support resize with Cyborg devices, the two statements conflict, creating ambiguity for deployment planning.\n\n**Priority**: Before merge\n**Why This Matters**: A spec is the contractual design document. Contradictory deprecation timelines will propagate into release notes, operator docs, and deprecation warnings in code, creating confusion and potential upgrade-path breakage for operators using the Inspur or SSD drivers.\n\n**Recommendation**:\nPick one removal criterion and use it consistently. Either state \u0027removed in 2027.2\u0027 in both sections, or state \u0027removed when Nova supports resizing with Cyborg-managed devices (target: 2027.2)\u0027 in both. The Driver Deprecation and Upgrade impact sections must agree.","commit_id":"43ead47d9364497a3395d37fa8261ac03ec38f11"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"163347f132ba133539bf8a61e7b19b724445a666","unresolved":true,"context_lines":[{"line_number":356,"context_line":"------------------"},{"line_number":357,"context_line":""},{"line_number":358,"context_line":"The existing SSD and Inspur drivers will be deprecated in this release"},{"line_number":359,"context_line":"and will be removed in 2027.2 in favour of the generic NVMe driver. The"},{"line_number":360,"context_line":"deprecated drivers and the new generic NVMe driver can coexist on the"},{"line_number":361,"context_line":"same system as long as they manage independent devices. For any given"},{"line_number":362,"context_line":"device, managing it with more than one driver simultaneously is"}],"source_content_type":"text/x-rst","patch_set":16,"id":"6b1d6913_eaa31122","line":359,"in_reply_to":"1ca31a9d_195e88d9","updated":"2026-06-23 16:02:44.000000000","message":"\u0027will not be removed until Nova supports resizing instances with Cyborg-managed devices\u0027.\n \n is the critira and 2027.2 is the earlist releasea we will do the removal in\n its also the target release but yes we shgoudl clarity this","commit_id":"43ead47d9364497a3395d37fa8261ac03ec38f11"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b65481c2244c0e91192130e709bbffab6db0e09f","unresolved":false,"context_lines":[{"line_number":356,"context_line":"------------------"},{"line_number":357,"context_line":""},{"line_number":358,"context_line":"The existing SSD and Inspur drivers will be deprecated in this release"},{"line_number":359,"context_line":"and will be removed in 2027.2 in favour of the generic NVMe driver. The"},{"line_number":360,"context_line":"deprecated drivers and the new generic NVMe driver can coexist on the"},{"line_number":361,"context_line":"same system as long as they manage independent devices. For any given"},{"line_number":362,"context_line":"device, managing it with more than one driver simultaneously is"}],"source_content_type":"text/x-rst","patch_set":16,"id":"13a447d3_f1ebcbe0","line":359,"in_reply_to":"6b1d6913_eaa31122","updated":"2026-06-24 06:03:25.000000000","message":"Done","commit_id":"43ead47d9364497a3395d37fa8261ac03ec38f11"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"472574265830d0304543a1f0919d14757cd13b2e","unresolved":false,"context_lines":[{"line_number":514,"context_line":"    cleaning (reserved\u003dtotal) ──(success)──► available (reserved\u003d0)"},{"line_number":515,"context_line":"    cleaning (reserved\u003dtotal) ──(failure / timeout)──► error (reserved\u003dtotal)"},{"line_number":516,"context_line":"    cleaning (reserved\u003dtotal) ──(init_host crash recovery)──► error (reserved\u003dtotal)"},{"line_number":517,"context_line":"    error (reserved\u003dtotal) ──(operator POST /clean)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":518,"context_line":""},{"line_number":519,"context_line":"A device starts in ``available`` state with ``reserved\u003d0``, meaning it"},{"line_number":520,"context_line":"is clean and ready for allocation. When bound to an instance,"}],"source_content_type":"text/x-rst","patch_set":16,"id":"66f8afd0_5c056f5a","line":517,"updated":"2026-06-23 15:25:48.000000000","message":"State machine lacks an \u0027error\u0027 to \u0027available\u0027 transition for the case where an operator manually verifies a device is clean and wants to return it to the pool without running cleanup again.\n\n**Severity**: SUGGESTION | **Confidence**: 0.8\n\n**Benefit**: Provides a defined escape hatch for operators who have manually sanitized a device outside Cyborg (e.g., via direct nvme-cli) and want to return it to service without triggering another cleanup cycle. Without this, a device in error requires a full cleanup retry.\n\n**Recommendation**:\nConsider adding an admin-only \u0027reset to available\u0027 transition or documenting that operators must use the clean endpoint to re-run cleanup as the only recovery path. At minimum, acknowledge this limitation in the state machine section.","commit_id":"43ead47d9364497a3395d37fa8261ac03ec38f11"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"163347f132ba133539bf8a61e7b19b724445a666","unresolved":false,"context_lines":[{"line_number":514,"context_line":"    cleaning (reserved\u003dtotal) ──(success)──► available (reserved\u003d0)"},{"line_number":515,"context_line":"    cleaning (reserved\u003dtotal) ──(failure / timeout)──► error (reserved\u003dtotal)"},{"line_number":516,"context_line":"    cleaning (reserved\u003dtotal) ──(init_host crash recovery)──► error (reserved\u003dtotal)"},{"line_number":517,"context_line":"    error (reserved\u003dtotal) ──(operator POST /clean)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":518,"context_line":""},{"line_number":519,"context_line":"A device starts in ``available`` state with ``reserved\u003d0``, meaning it"},{"line_number":520,"context_line":"is clean and ready for allocation. When bound to an instance,"}],"source_content_type":"text/x-rst","patch_set":16,"id":"5b877e0a_31e3ff5b","line":517,"in_reply_to":"66f8afd0_5c056f5a","updated":"2026-06-23 16:02:44.000000000","message":"that is intentonall we are not going to supprot that","commit_id":"43ead47d9364497a3395d37fa8261ac03ec38f11"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"472574265830d0304543a1f0919d14757cd13b2e","unresolved":false,"context_lines":[{"line_number":868,"context_line":"Cyborg does not yet have full rolling upgrade support, though grenade"},{"line_number":869,"context_line":"testing has been added recently and work is planned for the 2027.1"},{"line_number":870,"context_line":"cycle to support N-1 agent compatibility. All services must be"},{"line_number":871,"context_line":"upgraded together: conductor and API first, then agents. If an old"},{"line_number":872,"context_line":"agent receives the new ``cleanup_device`` RPC cast, the call fails"},{"line_number":873,"context_line":"silently and the device stays in ``pending_cleaning`` until the agent"},{"line_number":874,"context_line":"is upgraded. On restart, ``init_host()`` detects the device and moves"}],"source_content_type":"text/x-rst","patch_set":16,"id":"b795a2aa_540d4db8","line":871,"updated":"2026-06-23 15:25:48.000000000","message":"Upgrade impact section is internally contradictory: it states \u0027All services must be upgraded together: conductor and API first, then agents\u0027 which describes an ordered sequence, not a simultaneous upgrade, yet also says Cyborg does not have full rolling upgrade support.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: Reviewers and operators may be confused about whether a live rolling upgrade is supported or whether downtime is required. The phrase \u0027upgraded together\u0027 implies simultaneous restart, while the ordering describes a sequential upgrade.\n\n**Suggestion**:\nClarify the upgrade model. If it is a sequential (N then N-1) rolling upgrade where old agents temporarily coexist with upgraded conductor/API, state that explicitly. If it requires full downtime, say so. Replace \u0027upgraded together\u0027 with the precise ordered sequence and what guarantees hold during the transition window.","commit_id":"43ead47d9364497a3395d37fa8261ac03ec38f11"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b65481c2244c0e91192130e709bbffab6db0e09f","unresolved":false,"context_lines":[{"line_number":868,"context_line":"Cyborg does not yet have full rolling upgrade support, though grenade"},{"line_number":869,"context_line":"testing has been added recently and work is planned for the 2027.1"},{"line_number":870,"context_line":"cycle to support N-1 agent compatibility. All services must be"},{"line_number":871,"context_line":"upgraded together: conductor and API first, then agents. If an old"},{"line_number":872,"context_line":"agent receives the new ``cleanup_device`` RPC cast, the call fails"},{"line_number":873,"context_line":"silently and the device stays in ``pending_cleaning`` until the agent"},{"line_number":874,"context_line":"is upgraded. On restart, ``init_host()`` detects the device and moves"}],"source_content_type":"text/x-rst","patch_set":16,"id":"95bc4c35_13a1c879","line":871,"in_reply_to":"230fc57e_6c645660","updated":"2026-06-24 06:03:25.000000000","message":"Acknowledged","commit_id":"43ead47d9364497a3395d37fa8261ac03ec38f11"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"163347f132ba133539bf8a61e7b19b724445a666","unresolved":true,"context_lines":[{"line_number":868,"context_line":"Cyborg does not yet have full rolling upgrade support, though grenade"},{"line_number":869,"context_line":"testing has been added recently and work is planned for the 2027.1"},{"line_number":870,"context_line":"cycle to support N-1 agent compatibility. All services must be"},{"line_number":871,"context_line":"upgraded together: conductor and API first, then agents. If an old"},{"line_number":872,"context_line":"agent receives the new ``cleanup_device`` RPC cast, the call fails"},{"line_number":873,"context_line":"silently and the device stays in ``pending_cleaning`` until the agent"},{"line_number":874,"context_line":"is upgraded. On restart, ``init_host()`` detects the device and moves"}],"source_content_type":"text/x-rst","patch_set":16,"id":"230fc57e_6c645660","line":871,"in_reply_to":"b795a2aa_540d4db8","updated":"2026-06-23 16:02:44.000000000","message":"right i think we can likely just cover this here and remove the previos deprecation secton.","commit_id":"43ead47d9364497a3395d37fa8261ac03ec38f11"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"31578484e635ad0071270e48a7a8fe51c36b9d69","unresolved":true,"context_lines":[{"line_number":507,"context_line":""},{"line_number":508,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":509,"context_line":"    allocated (reserved\u003dtotal) ──(agent receives cleanup RPC)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":510,"context_line":"    allocated (reserved\u003dtotal) ──(init_host: no active ARQ)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":511,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(agent starts cleanup)──► cleaning (reserved\u003dtotal)"},{"line_number":512,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(init_host crash recovery)──► error (reserved\u003dtotal)"},{"line_number":513,"context_line":"    cleaning (reserved\u003dtotal) ──(success)──► available (reserved\u003d0)"}],"source_content_type":"text/x-rst","patch_set":17,"id":"77f9a35b_c68e0eae","line":510,"updated":"2026-06-26 10:18:08.000000000","message":"I don\u0027t understand this state transition. The device starts as allocated and reserved, and then after `init_host: no active ARQ` (does that mean an agent restart?) it goes to cleaning. What situation is supposed to model? An ARQ delete while the agent is down?","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"32fd3dac549912b5ed2d6c1281d466ee023f8972","unresolved":false,"context_lines":[{"line_number":507,"context_line":""},{"line_number":508,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":509,"context_line":"    allocated (reserved\u003dtotal) ──(agent receives cleanup RPC)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":510,"context_line":"    allocated (reserved\u003dtotal) ──(init_host: no active ARQ)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":511,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(agent starts cleanup)──► cleaning (reserved\u003dtotal)"},{"line_number":512,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(init_host crash recovery)──► error (reserved\u003dtotal)"},{"line_number":513,"context_line":"    cleaning (reserved\u003dtotal) ──(success)──► available (reserved\u003d0)"}],"source_content_type":"text/x-rst","patch_set":17,"id":"96650bc6_4e488fbe","line":510,"in_reply_to":"384092a3_3bc9101c","updated":"2026-06-29 15:54:39.000000000","message":"thanks Chandan, that is clearer now, the case makes sense","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"21ad868201ec441db039643b4a2840cedf8807bf","unresolved":true,"context_lines":[{"line_number":507,"context_line":""},{"line_number":508,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":509,"context_line":"    allocated (reserved\u003dtotal) ──(agent receives cleanup RPC)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":510,"context_line":"    allocated (reserved\u003dtotal) ──(init_host: no active ARQ)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":511,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(agent starts cleanup)──► cleaning (reserved\u003dtotal)"},{"line_number":512,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(init_host crash recovery)──► error (reserved\u003dtotal)"},{"line_number":513,"context_line":"    cleaning (reserved\u003dtotal) ──(success)──► available (reserved\u003d0)"}],"source_content_type":"text/x-rst","patch_set":17,"id":"384092a3_3bc9101c","line":510,"in_reply_to":"77f9a35b_c68e0eae","updated":"2026-06-26 13:47:12.000000000","message":"Good catch — the label was not clear. Here is the scenario: \n the cyborg-agent was down when Nova deleted the instance, The instance and its ARQ are gone, but the device is still marked allocated in the database because nobody cleaned it up.\n \n When the agent restarts, init_host() sees a device in allocated state with no ARQ referencing it \u003d that\u0027s the \"instance gone, cleanup RPC lost\" case.\n \n I will update the state machine diagram to make it more clear once I update the spec.","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"31578484e635ad0071270e48a7a8fe51c36b9d69","unresolved":true,"context_lines":[{"line_number":548,"context_line":"(``cyborg/common/placement_client.py``) on the device\u0027s resource"},{"line_number":549,"context_line":"provider. This prevents the scheduler from allocating the same device"},{"line_number":550,"context_line":"to another instance while it is in use or awaiting cleanup. The bind"},{"line_number":551,"context_line":"path in ``ExtARQ.bind()`` (``cyborg/objects/ext_arq.py``) also checks"},{"line_number":552,"context_line":"that ``device_state`` is ``available`` before proceeding; if the"},{"line_number":553,"context_line":"device is in any other state the bind is rejected. This guard ensures"},{"line_number":554,"context_line":"that a device still undergoing cleanup or in error cannot be"}],"source_content_type":"text/x-rst","patch_set":17,"id":"666d94dd_e7ca01c7","line":551,"updated":"2026-06-26 10:18:08.000000000","message":"this means that the bind/unbind flow will set the `device_state` field for all devices, not only nvmes right? Otherwise we would need type checks which should be avoided IMO","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"21ad868201ec441db039643b4a2840cedf8807bf","unresolved":true,"context_lines":[{"line_number":548,"context_line":"(``cyborg/common/placement_client.py``) on the device\u0027s resource"},{"line_number":549,"context_line":"provider. This prevents the scheduler from allocating the same device"},{"line_number":550,"context_line":"to another instance while it is in use or awaiting cleanup. The bind"},{"line_number":551,"context_line":"path in ``ExtARQ.bind()`` (``cyborg/objects/ext_arq.py``) also checks"},{"line_number":552,"context_line":"that ``device_state`` is ``available`` before proceeding; if the"},{"line_number":553,"context_line":"device is in any other state the bind is rejected. This guard ensures"},{"line_number":554,"context_line":"that a device still undergoing cleanup or in error cannot be"}],"source_content_type":"text/x-rst","patch_set":17,"id":"bc68c959_8932fed7","line":551,"in_reply_to":"666d94dd_e7ca01c7","updated":"2026-06-26 13:47:12.000000000","message":"Yes — device_state applies to all devices regardless of type, with no type-specific checks anywhere.\n\nThe bind guard (device_state \u003d\u003d available) and the Placement reserved\u003dtotal update both go into the ExtARQ.bind() path for all device types.\n\n\nNon-NVMe drivers never call cleanup, so their devices stay in available permanently — the guard always passes, bind works as before.\n\n\nI will add a developer impact note about the non-nvme driver no-op cleanup method.","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"32fd3dac549912b5ed2d6c1281d466ee023f8972","unresolved":false,"context_lines":[{"line_number":548,"context_line":"(``cyborg/common/placement_client.py``) on the device\u0027s resource"},{"line_number":549,"context_line":"provider. This prevents the scheduler from allocating the same device"},{"line_number":550,"context_line":"to another instance while it is in use or awaiting cleanup. The bind"},{"line_number":551,"context_line":"path in ``ExtARQ.bind()`` (``cyborg/objects/ext_arq.py``) also checks"},{"line_number":552,"context_line":"that ``device_state`` is ``available`` before proceeding; if the"},{"line_number":553,"context_line":"device is in any other state the bind is rejected. This guard ensures"},{"line_number":554,"context_line":"that a device still undergoing cleanup or in error cannot be"}],"source_content_type":"text/x-rst","patch_set":17,"id":"572869f1_87d6399c","line":551,"in_reply_to":"bc68c959_8932fed7","updated":"2026-06-29 15:54:39.000000000","message":"Done","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"31578484e635ad0071270e48a7a8fe51c36b9d69","unresolved":true,"context_lines":[{"line_number":623,"context_line":""},{"line_number":624,"context_line":"One additive Alembic migration is needed. A new ``device_state``"},{"line_number":625,"context_line":"column is added to the ``devices`` table as an Enum column over the"},{"line_number":626,"context_line":"values ``available``, ``allocated``, ``pending_cleaning``,"},{"line_number":627,"context_line":"``cleaning``, and ``error``, following the same pattern as the"},{"line_number":628,"context_line":"existing ``status`` column (``cyborg/db/sqlalchemy/models.py``). The"},{"line_number":629,"context_line":"column defaults to ``available`` and is ``NOT NULL``. The ``NVME``"}],"source_content_type":"text/x-rst","patch_set":17,"id":"1ec471b5_d646b9b8","line":626,"updated":"2026-06-26 10:18:08.000000000","message":"how will this migration work for existing devices, will it set `available` by default or will it check for existing ARQs to set allocated when appropiate? Also, if a device has been disabled by an operator, what would be the right state to set, `available`?","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"21ad868201ec441db039643b4a2840cedf8807bf","unresolved":true,"context_lines":[{"line_number":623,"context_line":""},{"line_number":624,"context_line":"One additive Alembic migration is needed. A new ``device_state``"},{"line_number":625,"context_line":"column is added to the ``devices`` table as an Enum column over the"},{"line_number":626,"context_line":"values ``available``, ``allocated``, ``pending_cleaning``,"},{"line_number":627,"context_line":"``cleaning``, and ``error``, following the same pattern as the"},{"line_number":628,"context_line":"existing ``status`` column (``cyborg/db/sqlalchemy/models.py``). The"},{"line_number":629,"context_line":"column defaults to ``available`` and is ``NOT NULL``. The ``NVME``"}],"source_content_type":"text/x-rst","patch_set":17,"id":"4085480b_2bc4239d","line":626,"in_reply_to":"1ec471b5_d646b9b8","updated":"2026-06-26 13:47:12.000000000","message":"It will set to available for all existing device. \n\nit does not check which devices are currently in use. \n\nOn next agent restart, init_host() detects devices that are attached to an instance and corrects them to allocated, and moves any stale cleanup states to error.\n\nIf a device has been disabled by an operator, available is correct. \n\ndevice_state tracks the device\u0027s lifecycle — available, allocated, cleaning, or error. \n\nstatus controls whether the device can be scheduled.\n\nThey are independent — a disabled device that isn\u0027t allocated to any instance is status\u003dmaintaining, device_state\u003davailable.","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"32fd3dac549912b5ed2d6c1281d466ee023f8972","unresolved":false,"context_lines":[{"line_number":623,"context_line":""},{"line_number":624,"context_line":"One additive Alembic migration is needed. A new ``device_state``"},{"line_number":625,"context_line":"column is added to the ``devices`` table as an Enum column over the"},{"line_number":626,"context_line":"values ``available``, ``allocated``, ``pending_cleaning``,"},{"line_number":627,"context_line":"``cleaning``, and ``error``, following the same pattern as the"},{"line_number":628,"context_line":"existing ``status`` column (``cyborg/db/sqlalchemy/models.py``). The"},{"line_number":629,"context_line":"column defaults to ``available`` and is ``NOT NULL``. The ``NVME``"}],"source_content_type":"text/x-rst","patch_set":17,"id":"930bb44b_f64b7c4e","line":626,"in_reply_to":"4085480b_2bc4239d","updated":"2026-06-29 15:54:39.000000000","message":"thanks, that sounds correct","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"31578484e635ad0071270e48a7a8fe51c36b9d69","unresolved":true,"context_lines":[{"line_number":645,"context_line":"available microversion after its dependencies land. The microversion"},{"line_number":646,"context_line":"is driven by the new ``device_state`` field in device responses and the"},{"line_number":647,"context_line":"new ``POST /v2/devices/{uuid}/clean`` endpoint. Adding ``NVME`` to the"},{"line_number":648,"context_line":"``type`` field does not by itself require a microversion because the"},{"line_number":649,"context_line":"type field is extensible for out-of-tree drivers without an API change."},{"line_number":650,"context_line":""},{"line_number":651,"context_line":"For API microversions lower than the new version, ``GET /v2/devices``"}],"source_content_type":"text/x-rst","patch_set":17,"id":"18b02686_a2d1338c","line":648,"updated":"2026-06-26 10:18:08.000000000","message":"I was under the impression that changing an enum does require an API microversion, is that wrong?","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"21ad868201ec441db039643b4a2840cedf8807bf","unresolved":true,"context_lines":[{"line_number":645,"context_line":"available microversion after its dependencies land. The microversion"},{"line_number":646,"context_line":"is driven by the new ``device_state`` field in device responses and the"},{"line_number":647,"context_line":"new ``POST /v2/devices/{uuid}/clean`` endpoint. Adding ``NVME`` to the"},{"line_number":648,"context_line":"``type`` field does not by itself require a microversion because the"},{"line_number":649,"context_line":"type field is extensible for out-of-tree drivers without an API change."},{"line_number":650,"context_line":""},{"line_number":651,"context_line":"For API microversions lower than the new version, ``GET /v2/devices``"}],"source_content_type":"text/x-rst","patch_set":17,"id":"bb358b3f_fa7656e1","line":648,"in_reply_to":"18b02686_a2d1338c","updated":"2026-06-26 13:47:12.000000000","message":"No. The type field in the API response is wtypes.text https://github.com/openstack/cyborg/blob/master/cyborg/api/controllers/v2/devices.py#L49 - free form text\n\nmicroversion is needed for \"the allowed values of non free form fields\" check \"https://docs.openstack.org/cyborg/latest/contributor/microversions.html#when-do-i-need-a-new-microversion\" section\n\nSince it a free text, no api version is needed.\n\nI will update this line to make it clear.","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"32fd3dac549912b5ed2d6c1281d466ee023f8972","unresolved":false,"context_lines":[{"line_number":645,"context_line":"available microversion after its dependencies land. The microversion"},{"line_number":646,"context_line":"is driven by the new ``device_state`` field in device responses and the"},{"line_number":647,"context_line":"new ``POST /v2/devices/{uuid}/clean`` endpoint. Adding ``NVME`` to the"},{"line_number":648,"context_line":"``type`` field does not by itself require a microversion because the"},{"line_number":649,"context_line":"type field is extensible for out-of-tree drivers without an API change."},{"line_number":650,"context_line":""},{"line_number":651,"context_line":"For API microversions lower than the new version, ``GET /v2/devices``"}],"source_content_type":"text/x-rst","patch_set":17,"id":"276fe0ca_4a6557f2","line":648,"in_reply_to":"bb358b3f_fa7656e1","updated":"2026-06-29 15:54:39.000000000","message":"thanks, I had indeed confused the api object with the db one","commit_id":"7234886a4ac989d550800bf0ff439e9af13ea3b7"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"8391b9248ff291e4d8a361240ef294b969188e24","unresolved":false,"context_lines":[{"line_number":480,"context_line":"and the ``devices`` database table. ``device_state`` is added to all"},{"line_number":481,"context_line":"devices regardless of type. Drivers that do not implement ``cleanup()``"},{"line_number":482,"context_line":"leave their devices in ``available`` permanently. The bind guard and"},{"line_number":483,"context_line":"Placement ``reserved`` updates apply uniformly to all device types"},{"line_number":484,"context_line":"with no type-specific checks. This is separate from the existing"},{"line_number":485,"context_line":"``status`` field which remains exclusively for the enabled/disabled"},{"line_number":486,"context_line":"scheduling control."}],"source_content_type":"text/x-rst","patch_set":18,"id":"30aa86e7_52c4bf45","line":483,"updated":"2026-06-26 14:57:56.000000000","message":"device_state is added to all device types and the bind guard requires device_state \u003d\u003d available, but existing non-NVMe drivers never set lifecycle states. The spec relies on the migration defaulting rows to \u0027available\u0027, with no documented behavior if a device is left non-available after upgrade.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: Devices managed by drivers without cleanup() that ever enter a non-available state have no path back to available except manual DB editing, since no cleanup RPC applies to them. This could strand non-NVMe devices after an upgrade anomaly.\n\n**Suggestion**:\nDocument that for drivers without a cleanup() implementation device_state is permanently \u0027available\u0027 and that the bind guard/reconciler should treat such drivers as always-available, or provide an admin reset path for device_state independent of POST /clean.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"8391b9248ff291e4d8a361240ef294b969188e24","unresolved":false,"context_lines":[{"line_number":488,"context_line":"The state transitions are::"},{"line_number":489,"context_line":""},{"line_number":490,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":491,"context_line":"    allocated (reserved\u003dtotal) ──(unbind: no cleanup)──► available (reserved\u003d0)"},{"line_number":492,"context_line":"    allocated (reserved\u003dtotal) ──(unbind)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":493,"context_line":"    allocated (reserved\u003dtotal) ──(agent restart: instance deleted while agent was down)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":494,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(agent starts cleanup)──► cleaning (reserved\u003dtotal)"}],"source_content_type":"text/x-rst","patch_set":18,"id":"d9fe299c_e3310ac5","line":491,"updated":"2026-06-26 14:57:56.000000000","message":"State machine shows an \u0027unbind: no cleanup\u0027 transition from allocated directly to available, but the spec states cleanup_device RPC is dispatched for ALL device types during unbind. This \u0027no cleanup\u0027 path is never defined or triggered, creating an internal contradiction in the core lifecycle model.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: An implementer cannot determine whether the allocated-\u003eavailable (no cleanup) transition is reachable. If implemented literally it bypasses sanitization and leaks tenant data; if omitted the state machine is incomplete. The documented invariant is ambiguous.\n\n**Priority**: Before merge\n**Why This Matters**: This is the central security guarantee: devices must be cleaned before returning to available. An undefined \u0027no cleanup\u0027 escape hatch undermines that contract and makes implementation behavior nondeterministic for reviewers and implementers.\n\n**Recommendation**:\nRemove the \u0027allocated -\u003e available (unbind: no cleanup)\u0027 transition, or explicitly define its trigger (e.g., a driver whose cleanup() is a no-op still flows pending_cleaning-\u003ecleaning-\u003eavailable rather than skipping the lifecycle).","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"e07ba7c0ead7cb9cfbb7aadabacafc92d668d3b0","unresolved":true,"context_lines":[{"line_number":488,"context_line":"The state transitions are::"},{"line_number":489,"context_line":""},{"line_number":490,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":491,"context_line":"    allocated (reserved\u003dtotal) ──(unbind: no cleanup)──► available (reserved\u003d0)"},{"line_number":492,"context_line":"    allocated (reserved\u003dtotal) ──(unbind)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":493,"context_line":"    allocated (reserved\u003dtotal) ──(agent restart: instance deleted while agent was down)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":494,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(agent starts cleanup)──► cleaning (reserved\u003dtotal)"}],"source_content_type":"text/x-rst","patch_set":18,"id":"59b238c4_de9bbcd4","line":491,"in_reply_to":"0a02f8ee_5b3dfb5c","updated":"2026-06-30 13:40:39.000000000","message":"Yes, removed intentionally. \n\nAll devices follow the same state transitions. I have re-created the diagram for all cases\n```\n● # Normal lifecycle\n  available ──(bind)──► allocated\n  allocated ──(unbind)──► pending_cleaning\n  pending_cleaning ──(agent starts cleanup)──► cleaning\n  cleaning ──(success)──► available\n  # Non-NVMe: cleanup() is a no-op\n  allocated ──(unbind)──► pending_cleaning ──► cleaning ──► available\n\n  # Failure paths\n  cleaning ──(failure / timeout)──► error\n\n  # Agent restart reconciliation\n  allocated ──(init_host: missed cleanup)──► pending_cleaning\n  pending_cleaning ──(init_host: crash recovery)──► error\n  cleaning ──(init_host: crash recovery)──► error\n\n  # Operator recovery\n  error ──(operator POST /clean)──► pending_cleaning\n```\nI hope it will make the transition clear.\n\nI have improves the wording for non-nvme driver cleaning state diagram in prose in this section(since pending_cleaning/cleaning happens so fast, Just added no-cleanup in the diagram): https://review.opendev.org/c/openstack/cyborg-specs/+/985349/18..21/specs/2026.2/approved/generic-nvme-driver-with-secure-cleanup.rst\n\nDo let me know if it is not clear. Thank you.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":false,"context_lines":[{"line_number":488,"context_line":"The state transitions are::"},{"line_number":489,"context_line":""},{"line_number":490,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":491,"context_line":"    allocated (reserved\u003dtotal) ──(unbind: no cleanup)──► available (reserved\u003d0)"},{"line_number":492,"context_line":"    allocated (reserved\u003dtotal) ──(unbind)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":493,"context_line":"    allocated (reserved\u003dtotal) ──(agent restart: instance deleted while agent was down)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":494,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(agent starts cleanup)──► cleaning (reserved\u003dtotal)"}],"source_content_type":"text/x-rst","patch_set":18,"id":"754b6a5b_b2daafb2","line":491,"in_reply_to":"59b238c4_de9bbcd4","updated":"2026-06-30 19:46:58.000000000","message":"Acknowledged","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"b42cb2d4e2f69f928932dd1e3a1110f6609a8765","unresolved":false,"context_lines":[{"line_number":488,"context_line":"The state transitions are::"},{"line_number":489,"context_line":""},{"line_number":490,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":491,"context_line":"    allocated (reserved\u003dtotal) ──(unbind: no cleanup)──► available (reserved\u003d0)"},{"line_number":492,"context_line":"    allocated (reserved\u003dtotal) ──(unbind)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":493,"context_line":"    allocated (reserved\u003dtotal) ──(agent restart: instance deleted while agent was down)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":494,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(agent starts cleanup)──► cleaning (reserved\u003dtotal)"}],"source_content_type":"text/x-rst","patch_set":18,"id":"bfce8299_e64c8b87","line":491,"in_reply_to":"59b238c4_de9bbcd4","updated":"2026-06-30 15:39:37.000000000","message":"thanks Chandan, it\u0027s clear now.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"6436309281cd5c117b281242713f5f7e1ee520a3","unresolved":true,"context_lines":[{"line_number":488,"context_line":"The state transitions are::"},{"line_number":489,"context_line":""},{"line_number":490,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":491,"context_line":"    allocated (reserved\u003dtotal) ──(unbind: no cleanup)──► available (reserved\u003d0)"},{"line_number":492,"context_line":"    allocated (reserved\u003dtotal) ──(unbind)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":493,"context_line":"    allocated (reserved\u003dtotal) ──(agent restart: instance deleted while agent was down)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":494,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(agent starts cleanup)──► cleaning (reserved\u003dtotal)"}],"source_content_type":"text/x-rst","patch_set":18,"id":"efe6c34d_ab97ce71","line":491,"in_reply_to":"6f48e8fd_fc2ebacf","updated":"2026-06-30 12:51:54.000000000","message":"Added a paragraph in the state machine section explaining\n  init_host() reconciliation — missed cleanup (allocated → pending_cleaning) \n  \n  \n and crash recovery (pending_cleaning/cleaning → error) with a cross-reference to the Timeout and Crash Recovery section for implementation details.\n \n I hope this will help, thank you.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"32fd3dac549912b5ed2d6c1281d466ee023f8972","unresolved":true,"context_lines":[{"line_number":488,"context_line":"The state transitions are::"},{"line_number":489,"context_line":""},{"line_number":490,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":491,"context_line":"    allocated (reserved\u003dtotal) ──(unbind: no cleanup)──► available (reserved\u003d0)"},{"line_number":492,"context_line":"    allocated (reserved\u003dtotal) ──(unbind)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":493,"context_line":"    allocated (reserved\u003dtotal) ──(agent restart: instance deleted while agent was down)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":494,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(agent starts cleanup)──► cleaning (reserved\u003dtotal)"}],"source_content_type":"text/x-rst","patch_set":18,"id":"6f48e8fd_fc2ebacf","line":491,"in_reply_to":"d9fe299c_e3310ac5","updated":"2026-06-29 15:54:39.000000000","message":"I agree, I\u0027m not sure under which conditions this transition could occur. Overall, I feel like it would be helpful to spell out in prose the less common transitions below (like this one or the ones involving an agent restart)","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":34452,"name":"Joan Gilabert","display_name":"jgilaber","email":"jgilaber@redhat.com","username":"jgilaber"},"change_message_id":"dd93736119a6db222ebfacb51a1c5f646249db44","unresolved":true,"context_lines":[{"line_number":488,"context_line":"The state transitions are::"},{"line_number":489,"context_line":""},{"line_number":490,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":491,"context_line":"    allocated (reserved\u003dtotal) ──(unbind: no cleanup)──► available (reserved\u003d0)"},{"line_number":492,"context_line":"    allocated (reserved\u003dtotal) ──(unbind)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":493,"context_line":"    allocated (reserved\u003dtotal) ──(agent restart: instance deleted while agent was down)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":494,"context_line":"    pending_cleaning (reserved\u003dtotal) ──(agent starts cleanup)──► cleaning (reserved\u003dtotal)"}],"source_content_type":"text/x-rst","patch_set":18,"id":"0a02f8ee_5b3dfb5c","line":491,"in_reply_to":"efe6c34d_ab97ce71","updated":"2026-06-30 13:01:49.000000000","message":"the ` allocated (reserved\u003dtotal) ──(unbind: no cleanup)──► available (reserved\u003d0)` is not on the list anymore, it\u0027s no longer relevant?","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"8391b9248ff291e4d8a361240ef294b969188e24","unresolved":false,"context_lines":[{"line_number":516,"context_line":"``POST /v2/devices/{uuid}/clean``, which transitions the device back"},{"line_number":517,"context_line":"to ``pending_cleaning``."},{"line_number":518,"context_line":""},{"line_number":519,"context_line":"The invariant ``reserved\u003dtotal`` in Placement is always consistent with"},{"line_number":520,"context_line":"``device_state`` not being ``available``. A mismatch indicates a defect"},{"line_number":521,"context_line":"in the bind or cleanup path; the agent logs a warning during"},{"line_number":522,"context_line":"``init_host()`` so operators can investigate and manually correct the"}],"source_content_type":"text/x-rst","patch_set":18,"id":"ea1c68b6_64cd0b8e","line":519,"updated":"2026-06-26 14:57:56.000000000","message":"The device_state/reserved invariant is enforced only by a warning log in init_host() when a mismatch is detected; there is no described mechanism to automatically correct or fence a mismatched device.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: A reserved/total vs available mismatch indicates a defect that could allow a stale-data device to be scheduled. Logging-only means a device in the most dangerous state may continue operating until a human intervenes.\n\n**Suggestion**:\nSpecify that a detected invariant mismatch quarantines the device (e.g., force reserved\u003dtotal and device_state\u003derror) rather than only logging, so the defective device cannot be allocated while the operator investigates.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"8391b9248ff291e4d8a361240ef294b969188e24","unresolved":false,"context_lines":[{"line_number":559,"context_line":"database for devices in ``cleaning`` or ``pending_cleaning`` state on"},{"line_number":560,"context_line":"its host. For each such device, the agent moves it to ``error`` state"},{"line_number":561,"context_line":"so the operator can investigate and re-trigger cleanup via"},{"line_number":562,"context_line":"``POST /v2/devices/{uuid}/clean``. The agent also checks for devices"},{"line_number":563,"context_line":"in ``allocated`` state that have no active ARQ allocation, which"},{"line_number":564,"context_line":"indicates a missed cleanup RPC. For each such device, the agent"},{"line_number":565,"context_line":"transitions it to ``pending_cleaning`` and triggers cleanup. The"}],"source_content_type":"text/x-rst","patch_set":18,"id":"e87873a6_e3689c3c","line":562,"updated":"2026-06-26 14:57:56.000000000","message":"The init_host() reconciliation detects \u0027devices in allocated state that have no active ARQ allocation\u0027 as a missed-cleanup signal, but this same condition is used in the data-model reconciliation for genuinely allocated devices. The two reconciliation rules could double-trigger.\n\n**Severity**: SUGGESTION | **Confidence**: 0.8\n\n**Benefit**: Disambiguating the two reconciliation passes prevents a legitimately allocated device from being erroneously sent to pending_cleaning (data loss of a bound tenant) or a missed-cleanup device from being left allocated.\n\n**Recommendation**:\nClarify the ordering and mutual exclusivity of the two reconciliation rules: first reconcile allocated devices with bound ARQs (keep allocated), then treat remaining allocated devices with no ARQ as missed cleanups.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"8391b9248ff291e4d8a361240ef294b969188e24","unresolved":false,"context_lines":[{"line_number":617,"context_line":"currently contains ``GPU``, ``FPGA``, ``AICHIP``, ``QAT``, ``NIC``,"},{"line_number":618,"context_line":"and ``SSD``."},{"line_number":619,"context_line":""},{"line_number":620,"context_line":"This is expand-only with no contract phase. The ``Device``"},{"line_number":621,"context_line":"oslo.versionedobjects definition (``cyborg/objects/device.py``,"},{"line_number":622,"context_line":"currently version ``1.2``) is bumped to include the ``device_state``"},{"line_number":623,"context_line":"field. The existing ``status`` field remains unchanged and continues"}],"source_content_type":"text/x-rst","patch_set":18,"id":"8ddd1915_aa198731","line":620,"updated":"2026-06-26 14:57:56.000000000","message":"Data model section calls the migration \u0027expand-only with no contract phase\u0027, but the Device oslo.versionedobjects version is bumped to add device_state. A versioned-object bump is the rolling-upgrade contract surface; calling it expand-only obscures the real compatibility constraint.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: Implementers may treat this as trivially additive and skip the versioned-object compatibility handling (obj_make_compatible) needed so older conductors/APIs receiving a Device with the new field do not break. With the admitted lack of N-1 agent support this raises real rolling-upgrade breakage risk.\n\n**Priority**: Before merge\n**Why This Matters**: The Upgrade section states Cyborg lacks full rolling-upgrade support and all services must be upgraded together. Mislabeling the object bump as expand-only contradicts that and under-specifies the versioning handling an implementer must write.\n\n**Recommendation**:\nRecharacterize as an additive schema column plus a versioned-object minor bump; state that obj_make_compatible must drop device_state for older object versions, and note mixed-version clusters are unsupported until the 2027.1 N-1 work lands.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"8391b9248ff291e4d8a361240ef294b969188e24","unresolved":false,"context_lines":[{"line_number":649,"context_line":"    \"type\": \"NVME\","},{"line_number":650,"context_line":"    \"vendor\": \"8086\","},{"line_number":651,"context_line":"    \"model\": \"0001\","},{"line_number":652,"context_line":"    \"std_board_info\": \"{\\\"product_id\\\": \\\"0001\\\","},{"line_number":653,"context_line":"        \\\"pci_address\\\": \\\"0000:01:00.0\\\"}\","},{"line_number":654,"context_line":"    \"vendor_board_info\": null,"},{"line_number":655,"context_line":"    \"hostname\": \"compute-1\","}],"source_content_type":"text/x-rst","patch_set":18,"id":"78a50f0d_7a597a6a","line":652,"updated":"2026-06-26 14:57:56.000000000","message":"The GET response examples embed std_board_info as a JSON-encoded string nested inside a JSON response. Flattening or documenting the expected shape would aid SDK and client consumers.\n\n**Severity**: SUGGESTION | **Confidence**: 0.8\n\n**Benefit**: A documented, machine-validatable response schema reduces SDK guesswork about whether std_board_info is a string or object and how device_state integrates into the existing Device response envelope.\n\n**Recommendation**:\nAdd a formal response JSON schema for the Device object at the new microversion (with device_state as a required string field), consistent with the template\u0027s schema requirement, rather than only free-form examples.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"8391b9248ff291e4d8a361240ef294b969188e24","unresolved":false,"context_lines":[{"line_number":676,"context_line":"    \"updated_at\": \"2026-05-15T16:30:15Z\""},{"line_number":677,"context_line":"  }"},{"line_number":678,"context_line":""},{"line_number":679,"context_line":"**POST /v2/devices/{uuid}/clean (new endpoint, new microversion):**"},{"line_number":680,"context_line":""},{"line_number":681,"context_line":"This is an admin-only endpoint that triggers device cleanup. It"},{"line_number":682,"context_line":"dispatches the ``cleanup_device`` RPC to the agent, which transitions"}],"source_content_type":"text/x-rst","patch_set":18,"id":"9ec35a6a_063b38da","line":679,"updated":"2026-06-26 14:57:56.000000000","message":"REST API section provides examples for POST /clean but omits the JSON schema definitions for the request and response body that the template explicitly requires. It is unclear whether the endpoint accepts any request body.\n\n**Severity**: WARNING | **Confidence**: 0.9\n\n**Impact**: API changes are held to the highest scrutiny per the template; a missing schema leaves request validation (e.g., whether a cleanup method override is allowed) undefined, inviting incompatible implementations.\n\n**Suggestion**:\nAdd an explicit (empty or minimal) request body JSON schema and a response body schema (202 with no body). State that no request parameters are accepted and the retry reuses the locked-in action. Note additionalProperties: False as the template recommends.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"8391b9248ff291e4d8a361240ef294b969188e24","unresolved":false,"context_lines":[{"line_number":690,"context_line":""},{"line_number":691,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":692,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"},{"line_number":693,"context_line":"* ``409 Conflict`` — device is in ``allocated`` state (bound to an"},{"line_number":694,"context_line":"  instance)"},{"line_number":695,"context_line":"* ``409 Conflict`` — device is already in ``cleaning`` or"},{"line_number":696,"context_line":"  ``pending_cleaning`` state"}],"source_content_type":"text/x-rst","patch_set":18,"id":"9c73eef6_d80f8cd3","line":693,"updated":"2026-06-26 14:57:56.000000000","message":"The conductor dispatches cleanup_device via an async cast but never writes device_state. Between unbind and agent receipt the device stays \u0027allocated\u0027 though the instance is gone. POST /clean returns 409 for \u0027allocated\u0027 devices, so an operator cannot force cleanup if the RPC is lost.\n\n**Severity**: HIGH | **Confidence**: 0.9\n\n**Risk**: If the cleanup RPC is dropped (old agent, network partition, agent crash before receipt), the device is stuck in \u0027allocated\u0027 and the operator\u0027s only documented recovery tool (POST /clean) is rejected with 409. Recovery then depends solely on the next agent restart.\n\n**Priority**: Before merge\n**Why This Matters**: The spec\u0027s own use case states operators need manual recovery tools for stuck cleanups. A 409 on \u0027allocated\u0027 defeats that for the most common stuck scenario (lost/undelivered cleanup RPC).\n\n**Recommendation**:\nEither allow POST /clean on \u0027allocated\u0027 devices that have no active ARQ, or have the conductor transition device_state to pending_cleaning before the cast so the endpoint\u0027s accepted states cover the recovery path. Document the exact set of states POST /clean accepts.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"8391b9248ff291e4d8a361240ef294b969188e24","unresolved":false,"context_lines":[{"line_number":701,"context_line":""},{"line_number":702,"context_line":"  HTTP/1.1 202 Accepted"},{"line_number":703,"context_line":""},{"line_number":704,"context_line":"Policy is unchanged for existing endpoints. The new ``POST"},{"line_number":705,"context_line":"/v2/devices/{uuid}/clean`` endpoint requires admin credentials."},{"line_number":706,"context_line":""},{"line_number":707,"context_line":"RPC API impact"}],"source_content_type":"text/x-rst","patch_set":18,"id":"ea04cbcf_d5ffde1f","line":704,"updated":"2026-06-26 14:57:56.000000000","message":"The policy for the new admin-only POST /clean endpoint is described only as \u0027requires admin credentials\u0027; the concrete policy rule name and its mapping in policy.yaml are not specified.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: Without a named policy rule, deployers cannot scope the endpoint under consistent RBAC (e.g., role:admin vs system:admin) and the SDK/CLI cannot document the required persona. This matters given Cyborg\u0027s secure-RBAC direction.\n\n**Suggestion**:\nName the policy rule (e.g., \u0027cyborg_api:device:clean\u0027) and state the default role/persona, cross-referencing the existing Cyborg secure-RBAC spec for consistency.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"8391b9248ff291e4d8a361240ef294b969188e24","unresolved":false,"context_lines":[{"line_number":928,"context_line":"Dependencies"},{"line_number":929,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":930,"context_line":""},{"line_number":931,"context_line":"``nvme-cli`` must be installed on compute nodes. An os-traits companion"},{"line_number":932,"context_line":"patch adding ``os_traits/hw/nvme/__init__.py`` with traits ``CES``,"},{"line_number":933,"context_line":"``BES``, and ``WZS`` must land before or alongside the Cyborg"},{"line_number":934,"context_line":"implementation. ``futurist`` is added as a new Python dependency."}],"source_content_type":"text/x-rst","patch_set":18,"id":"96e8fad9_a80ff410","line":931,"updated":"2026-06-26 14:57:56.000000000","message":"The os-traits companion patch (CES/BES/WZS) is a hard dependency but the spec does not reference a blueprint or change for it, only a generic os-traits HW namespace link.\n\n**Severity**: SUGGESTION | **Confidence**: 0.8\n\n**Benefit**: An explicit cross-project dependency reference lets the Cyborg gate and reviewers track the prerequisite and avoids the implementation landing before the traits exist.\n\n**Recommendation**:\nAdd a concrete dependency reference (os-traits blueprint or change URL) for the NVMe trait additions, and note the gating implication if it has not merged.","commit_id":"a974972a29cfa54e3de6922b0423221f16171d66"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"bf60856f00e91a6ca2bf3ef9cdbbb645c45b247b","unresolved":false,"context_lines":[{"line_number":112,"context_line":""},{"line_number":113,"context_line":"::"},{"line_number":114,"context_line":""},{"line_number":115,"context_line":"    ┌─────────────────────────────────────────────────┐"},{"line_number":116,"context_line":"    │  Operator configures cyborg.conf                │"},{"line_number":117,"context_line":"    │  [agent] enabled_drivers \u003d nvme_driver          │"},{"line_number":118,"context_line":"    │  [nvme]  device_spec \u003d {vendor_id, product_id,  │"}],"source_content_type":"text/x-rst","patch_set":19,"id":"e85ee72a_615102c8","line":115,"updated":"2026-06-28 07:47:41.000000000","message":"The ASCII lifecycle-flow diagram is ~157 characters wide, exceeding 79 columns on ~20 lines. doc8 D001 in the pep8 environment checks all lines including literal blocks, so this will fail the gate.\n\n**Severity**: WARNING | **Confidence**: 0.9\n\n**Impact**: CI failure in the pep8/doc8 check. While ASCII diagrams are required by the template and cannot always fit 79 columns, doc8 does not exempt them by default.\n\n**Suggestion**:\nNarrow the diagram boxes to fit within 79 columns (shorten interior text, use abbreviations, or split into a vertical flow). Alternatively, align the tox.ini \u0027doc8 specs/\u0027 invocation with the \u0027docs\u0027 environment pattern (doc8 --ignore D001 doc/) via a project-wide decision, but that requires a separate tox.ini change.","commit_id":"9436c6cb877ab12f9e42609448a37e6d94d8b8b8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"bf60856f00e91a6ca2bf3ef9cdbbb645c45b247b","unresolved":false,"context_lines":[{"line_number":306,"context_line":"* ``block``: require a non-cryptographic block/media erase or block"},{"line_number":307,"context_line":"  clear action."},{"line_number":308,"context_line":""},{"line_number":309,"context_line":"The policy matrix is::"},{"line_number":310,"context_line":""},{"line_number":311,"context_line":"    +----------------+-------------+------------------------------------------+"},{"line_number":312,"context_line":"    | clear_method   | clear_mode  | selected cleanup action                  |"}],"source_content_type":"text/x-rst","patch_set":19,"id":"890c1abd_7cb123dd","line":309,"updated":"2026-06-28 07:47:41.000000000","message":"The policy matrix lists \u0027shred\u0027 as a fallback action without noting in the table that shred is a host-side Linux software command, distinct from the NVMe hardware commands.\n\n**Severity**: SUGGESTION | **Confidence**: 0.8\n\n**Benefit**: Makes the policy matrix self-documenting for operators who may not read the surrounding prose at lines 333-336.\n\n**Recommendation**:\nAdd a footnote or annotation in the table clarifying that shred is a host-side software fallback (not an nvme-cli command), referencing Nova\u0027s volume_clear pattern already cited.","commit_id":"9436c6cb877ab12f9e42609448a37e6d94d8b8b8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"bf60856f00e91a6ca2bf3ef9cdbbb645c45b247b","unresolved":false,"context_lines":[{"line_number":477,"context_line":""},{"line_number":478,"context_line":"The device lifecycle is tracked using a new ``device_state`` field"},{"line_number":479,"context_line":"added to the ``Device`` versioned object (``cyborg/objects/device.py``)"},{"line_number":480,"context_line":"and the ``devices`` database table. ``device_state`` is added to all"},{"line_number":481,"context_line":"devices regardless of type. Drivers that do not implement ``cleanup()``"},{"line_number":482,"context_line":"leave their devices in ``available`` permanently. The bind guard and"},{"line_number":483,"context_line":"Placement ``reserved`` updates apply uniformly to all device types"}],"source_content_type":"text/x-rst","patch_set":19,"id":"5de6a665_ba5f29a0","line":480,"updated":"2026-06-28 07:47:41.000000000","message":"The spec applies device_state, the bind guard, and Placement reserved\u003dtotal updates uniformly to ALL device types with no type-specific checks, but does not analyze migration safety for existing GPU/FPGA/AICHIP/QAT/NIC/SSD drivers that have never seen device_state.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: A behavioral change to the shared bind/unbind path affecting every device type could block device allocation after upgrade if any existing driver assumes no state guard or if a device is mid-bind when the DB migration runs.\n\n**Priority**: Before merge\n**Why This Matters**: This is a cross-cutting change beyond the NVMe scope. Operators upgrading will have all devices default to available, and the conductor will dispatch a new cleanup_device RPC on unbind for all device types, adding an RPC round-trip to existing unbind paths.\n\n**Recommendation**:\nAdd explicit confirmation that the migration default (available) is the only bind-allowing state so existing devices remain allocatable, that no-op cleanup() for non-NVMe drivers means the unbind RPC completes immediately, and that a device mid-bind at migration time is not left in a non-available state.","commit_id":"9436c6cb877ab12f9e42609448a37e6d94d8b8b8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"bf60856f00e91a6ca2bf3ef9cdbbb645c45b247b","unresolved":false,"context_lines":[{"line_number":486,"context_line":"scheduling control."},{"line_number":487,"context_line":""},{"line_number":488,"context_line":"The state transitions are::"},{"line_number":489,"context_line":""},{"line_number":490,"context_line":"    available (reserved\u003d0) ──(bind)──► allocated (reserved\u003dtotal)"},{"line_number":491,"context_line":"    allocated (reserved\u003dtotal) ──(unbind)──► pending_cleaning (reserved\u003dtotal)"},{"line_number":492,"context_line":"    allocated (reserved\u003dtotal) ──(agent restart: instance deleted while agent was down)──► pending_cleaning (reserved\u003dtotal)"}],"source_content_type":"text/x-rst","patch_set":19,"id":"ffde4d7d_3af86374","line":489,"updated":"2026-06-28 07:47:41.000000000","message":"The device state-machine transition list (a literal block) exceeds 79 columns on 7 lines (e.g. line 492 is 134 chars). The pep8 tox environment runs \u0027doc8 specs/\u0027 without --ignore D001, so D001 (line-too-long) is active and will fail CI.\n\n**Severity**: WARNING | **Confidence**: 1.0\n\n**Impact**: The Zuul pep8 gate job will fail on these lines, blocking merge. The template explicitly states \u0027Please wrap text at 79 columns.\u0027\n\n**Suggestion**:\nReformat the state-machine transitions to fit within 79 columns (abbreviate labels, e.g. \u0027res\u003dtotal\u0027 for \u0027reserved\u003dtotal\u0027, or break long transitions across lines). Unlike bare URLs, these text lines are reformattable.","commit_id":"9436c6cb877ab12f9e42609448a37e6d94d8b8b8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"bf60856f00e91a6ca2bf3ef9cdbbb645c45b247b","unresolved":false,"context_lines":[{"line_number":639,"context_line":"microversion and later, device responses include the ``device_state``"},{"line_number":640,"context_line":"field."},{"line_number":641,"context_line":""},{"line_number":642,"context_line":"**Example GET /v2/devices/{uuid} response (new microversion):**"},{"line_number":643,"context_line":""},{"line_number":644,"context_line":"Device in available state after successful cleanup::"},{"line_number":645,"context_line":""}],"source_content_type":"text/x-rst","patch_set":19,"id":"e616dafa_7bc885cd","line":642,"updated":"2026-06-28 07:47:41.000000000","message":"REST API impact section provides illustrative JSON examples but no formal JSON schema definitions for the device response or the POST /clean request body, which the template explicitly requires (with additionalProperties: false).\n\n**Severity**: HIGH | **Confidence**: 0.9\n\n**Risk**: Merged API changes must be supported forever; an ambiguous contract for device_state enum values and the POST body creates implementation risk and reviewer uncertainty.\n\n**Priority**: Before merge\n**Why This Matters**: The spec template states API changes are held to a much higher level of scrutiny and require restrictive JSON schema definitions. The device_state field\u0027s allowed values and the POST endpoint\u0027s empty-body contract are currently only implied by examples.\n\n**Recommendation**:\nAdd a formal JSON schema block for the device response (device_state as enum of available|allocated|pending_cleaning|cleaning|error, additionalProperties: false) and explicitly state that POST /clean accepts no request body (Content-Length: 0); sending a body returns 400.","commit_id":"9436c6cb877ab12f9e42609448a37e6d94d8b8b8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"bf60856f00e91a6ca2bf3ef9cdbbb645c45b247b","unresolved":false,"context_lines":[{"line_number":643,"context_line":""},{"line_number":644,"context_line":"Device in available state after successful cleanup::"},{"line_number":645,"context_line":""},{"line_number":646,"context_line":"  {"},{"line_number":647,"context_line":"    \"uuid\": \"7c8a5f3b-2d4e-4a9c-b1e7-9f8d3c2a1b0e\","},{"line_number":648,"context_line":"    \"type\": \"NVME\","},{"line_number":649,"context_line":"    \"vendor\": \"8086\","}],"source_content_type":"text/x-rst","patch_set":19,"id":"68976235_587e703a","line":646,"updated":"2026-06-28 07:47:41.000000000","message":"The example GET response shows std_board_info as a JSON-encoded string with a raw line break mid-string (lines 651-652 and 668-669), which is invalid JSON if copy-pasted.\n\n**Severity**: WARNING | **Confidence**: 0.9\n\n**Impact**: API examples are frequently used as references for API sample tests and documentation; invalid JSON examples create confusion for implementers and doc writers.\n\n**Suggestion**:\nPut the entire std_board_info value on a single line, or note that it is a JSON-encoded string field and render it as a properly escaped single-line value with explicit \\n escapes rather than raw line breaks.","commit_id":"9436c6cb877ab12f9e42609448a37e6d94d8b8b8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"bf60856f00e91a6ca2bf3ef9cdbbb645c45b247b","unresolved":false,"context_lines":[{"line_number":675,"context_line":"    \"updated_at\": \"2026-05-15T16:30:15Z\""},{"line_number":676,"context_line":"  }"},{"line_number":677,"context_line":""},{"line_number":678,"context_line":"**POST /v2/devices/{uuid}/clean (new endpoint, new microversion):**"},{"line_number":679,"context_line":""},{"line_number":680,"context_line":"This is an admin-only endpoint that triggers device cleanup. It"},{"line_number":681,"context_line":"dispatches the ``cleanup_device`` RPC to the agent, which transitions"}],"source_content_type":"text/x-rst","patch_set":19,"id":"09eeb129_f1fa43fc","line":678,"updated":"2026-06-28 07:47:41.000000000","message":"The POST /clean error-response list (404 + three 409 variants) omits explicit handling semantics: it does not state that 202 is only valid for the error state, nor what happens for an unexpected request body (400).\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: API ambiguity about valid preconditions and syntactic error handling; the template requires a description for each possible error code including semantic causes.\n\n**Suggestion**:\nExplicitly state the endpoint accepts devices in error state (202) and returns 409 for all other states, and that an unexpected request body results in 400 Bad Request. This also resolves the missing JSON-schema requirement for the request body.","commit_id":"9436c6cb877ab12f9e42609448a37e6d94d8b8b8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"bf60856f00e91a6ca2bf3ef9cdbbb645c45b247b","unresolved":false,"context_lines":[{"line_number":859,"context_line":"Cyborg does not yet have full rolling upgrade support, though grenade"},{"line_number":860,"context_line":"testing has been added recently and work is planned for the 2027.1"},{"line_number":861,"context_line":"cycle to support N-1 agent compatibility. All services must be"},{"line_number":862,"context_line":"upgraded together: conductor and API first, then agents. If an old"},{"line_number":863,"context_line":"agent receives the new ``cleanup_device`` RPC cast, the call fails"},{"line_number":864,"context_line":"silently and the device stays in ``allocated`` until the agent"},{"line_number":865,"context_line":"is upgraded. On restart, ``init_host()`` detects the device and moves"}],"source_content_type":"text/x-rst","patch_set":19,"id":"fe381f22_46885d30","line":862,"updated":"2026-06-28 07:47:41.000000000","message":"The upgrade section states an old agent receiving the new cleanup_device RPC cast \u0027fails silently\u0027, which mischaracterizes oslo.messaging behavior: a version-pinned cast (version\u003d\u00271.1\u0027) is rejected by version negotiation and logged.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: Operators troubleshooting stuck allocated devices during a partial upgrade would be misled by \u0027fails silently\u0027 into expecting no log trace, delaying diagnosis.\n\n**Priority**: Before merge\n**Why This Matters**: The safety outcome (device stays allocated, later moved to error by init_host) is correct, but the described mechanism is inaccurate. Implementers and operators need the true failure mode for upgrade debugging.\n\n**Recommendation**:\nReword to: oslo.messaging rejects the cast due to the version pin (1.1); since cast is asynchronous the conductor does not detect the rejection. The device remains allocated until the agent is upgraded; on restart init_host reconciles it to error.","commit_id":"9436c6cb877ab12f9e42609448a37e6d94d8b8b8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"996d94b699b731074d6015c3bc443bcda9cca1df","unresolved":false,"context_lines":[{"line_number":93,"context_line":"        NVMeDriver (new, cyborg/accelerator/drivers/nvme.py,"},{"line_number":94,"context_line":"                    standalone driver)"},{"line_number":95,"context_line":"        ├── init_host()             → validates nvme-cli is installed"},{"line_number":96,"context_line":"        ├── discover()              → sysfs PCI enumeration filtered by device_spec,"},{"line_number":97,"context_line":"        │                              NVMe capability detection via nvme id-ctrl"},{"line_number":98,"context_line":"        └── cleanup(device)         → nvme-cli sanitize/zero"},{"line_number":99,"context_line":""}],"source_content_type":"text/x-rst","patch_set":21,"id":"bf18ca3a_4f28ecca","line":96,"updated":"2026-06-30 14:07:11.000000000","message":"Two lines in the class-hierarchy code block exceed the 79-column limit enforced by \u0027doc8 specs/\u0027 in the pep8 tox env: line 96 (84 chars) and line 97 (81 chars). These are not a documented exception (unlike ASCII diagrams and URLs) and would be flagged by doc8 D001.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: The pep8 gate (tox -e pep8 runs \u0027doc8 specs/\u0027 without --ignore D001) will fail on these lines, blocking merge of the spec.\n\n**Suggestion**:\nShorten the discover() annotation: move \u0027NVMe capability detection via nvme id-ctrl\u0027 to its own aligned row and shorten \u0027sysfs PCI enumeration filtered by device_spec\u0027, keeping each line at or under 79 characters.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":false,"context_lines":[{"line_number":259,"context_line":""},{"line_number":260,"context_line":"``clear_method`` and ``clear_mode`` default to ``auto`` when omitted."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":"Discovery and Cleanup Resolution"},{"line_number":263,"context_line":"---------------------------------"},{"line_number":264,"context_line":""},{"line_number":265,"context_line":"During discovery, the NVMe driver resolves the NVMe device path from the"}],"source_content_type":"text/x-rst","patch_set":21,"id":"51988d6e_5c1fcc98","line":262,"updated":"2026-06-30 19:46:58.000000000","message":"this section is now quite clear thanks","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"996d94b699b731074d6015c3bc443bcda9cca1df","unresolved":false,"context_lines":[{"line_number":260,"context_line":"``clear_method`` and ``clear_mode`` default to ``auto`` when omitted."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":"Discovery and Cleanup Resolution"},{"line_number":263,"context_line":"---------------------------------"},{"line_number":264,"context_line":""},{"line_number":265,"context_line":"During discovery, the NVMe driver resolves the NVMe device path from the"},{"line_number":266,"context_line":"PCI address via sysfs and runs ``nvme id-ctrl`` under privsep for each"}],"source_content_type":"text/x-rst","patch_set":21,"id":"8877b4f1_c5774c11","line":263,"updated":"2026-06-30 14:07:11.000000000","message":"Three section underlines are one character longer than their titles (lines 263, 412, 590). docutils tolerates over-long underlines and doc8 does not flag them, but the template convention is exact-length underlines.\n\n**Severity**: SUGGESTION | **Confidence**: 0.7\n\n**Benefit**: Consistent underline lengths match every other section in the spec and the template, avoiding reviewer nitpicks.\n\n**Recommendation**:\nTrim the trailing character on the underlines at lines 263 (33-\u003e32), 412 (29-\u003e28), and 590 (31-\u003e30) to exactly match the title length.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":true,"context_lines":[{"line_number":260,"context_line":"``clear_method`` and ``clear_mode`` default to ``auto`` when omitted."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":"Discovery and Cleanup Resolution"},{"line_number":263,"context_line":"---------------------------------"},{"line_number":264,"context_line":""},{"line_number":265,"context_line":"During discovery, the NVMe driver resolves the NVMe device path from the"},{"line_number":266,"context_line":"PCI address via sysfs and runs ``nvme id-ctrl`` under privsep for each"}],"source_content_type":"text/x-rst","patch_set":21,"id":"caab6787_fccdd134","line":263,"in_reply_to":"8877b4f1_c5774c11","updated":"2026-06-30 19:46:58.000000000","message":"i guess that is true","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":false,"context_lines":[{"line_number":260,"context_line":"``clear_method`` and ``clear_mode`` default to ``auto`` when omitted."},{"line_number":261,"context_line":""},{"line_number":262,"context_line":"Discovery and Cleanup Resolution"},{"line_number":263,"context_line":"---------------------------------"},{"line_number":264,"context_line":""},{"line_number":265,"context_line":"During discovery, the NVMe driver resolves the NVMe device path from the"},{"line_number":266,"context_line":"PCI address via sysfs and runs ``nvme id-ctrl`` under privsep for each"}],"source_content_type":"text/x-rst","patch_set":21,"id":"c36726b2_ad59cac6","line":263,"in_reply_to":"caab6787_fccdd134","updated":"2026-07-01 07:56:58.000000000","message":"Marked as resolved.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":4690,"name":"melanie witt","display_name":"melwitt","email":"melwittt@gmail.com","username":"melwitt"},"change_message_id":"632b308da857eec01c8fa8306095d056b78ca864","unresolved":true,"context_lines":[{"line_number":299,"context_line":"  full device."},{"line_number":300,"context_line":""},{"line_number":301,"context_line":"``clear_mode`` selects the erase action within the mechanism family:"},{"line_number":302,"context_line":""},{"line_number":303,"context_line":"* ``auto`` (default): choose the strongest supported action for the"},{"line_number":304,"context_line":"  selected method."},{"line_number":305,"context_line":"* ``crypto``: require cryptographic erase."}],"source_content_type":"text/x-rst","patch_set":21,"id":"56b6615f_577b6c5e","line":302,"updated":"2026-06-30 23:38:40.000000000","message":"As a newcomer to this spec, I will say that the terms `clear_method` and `clear_mode` are very similar and unclear the difference of their meaning ... and I expect people will transpose these constantly.\n\nIt is not necessary to figure this out in the spec but in the implementation I strongly suggest to not make them both be named \"clear_\" + a word that starts with \"m\".\n\nA few ideas:\n\n* `erase_strategy` and `erase_action`\n* `clear_strategy` and `clear_action`\n* `clear_method` and `clear_action`\n\nJust MHO of course.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":false,"context_lines":[{"line_number":299,"context_line":"  full device."},{"line_number":300,"context_line":""},{"line_number":301,"context_line":"``clear_mode`` selects the erase action within the mechanism family:"},{"line_number":302,"context_line":""},{"line_number":303,"context_line":"* ``auto`` (default): choose the strongest supported action for the"},{"line_number":304,"context_line":"  selected method."},{"line_number":305,"context_line":"* ``crypto``: require cryptographic erase."}],"source_content_type":"text/x-rst","patch_set":21,"id":"1f476fec_84fc4725","line":302,"in_reply_to":"56b6615f_577b6c5e","updated":"2026-07-01 07:56:58.000000000","message":"went with clear_strategy and clear_action","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"996d94b699b731074d6015c3bc443bcda9cca1df","unresolved":false,"context_lines":[{"line_number":335,"context_line":"host-side zeroing via ``shred``, following Nova\u0027s ``volume_clear``"},{"line_number":336,"context_line":"pattern for LVM volumes."},{"line_number":337,"context_line":""},{"line_number":338,"context_line":"The resolved action is stored in ``std_board_info`` alongside existing"},{"line_number":339,"context_line":"device metadata. The selected cleanup action is fixed for the device"},{"line_number":340,"context_line":"until the next discovery cycle or agent restart. There is no runtime"},{"line_number":341,"context_line":"fallback. If the"}],"source_content_type":"text/x-rst","patch_set":21,"id":"24b0d5d0_98a844d8","line":338,"updated":"2026-06-30 14:07:11.000000000","message":"The locked-in cleanup action is stored in std_board_info (free-form JSON metadata). The operator retry (POST /clean) reads it from there. If a discovery cycle re-runs before cleanup, update_available_resource could overwrite std_board_info and change the locked-in action mid-cleanup window.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: A device in pending_cleaning/cleaning whose std_board_info is refreshed by discovery could have its clear_action silently replaced, so the retry performs a different (possibly weaker) erase than originally selected, undermining the \u0027no runtime fallback\u0027 guarantee.\n\n**Suggestion**:\nEither state that discover() must not overwrite the cleanup-action portion of std_board_info while device_state is non-available, or store the locked-in clear_action in a dedicated column. Add this invariant to the State Machine section.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":true,"context_lines":[{"line_number":335,"context_line":"host-side zeroing via ``shred``, following Nova\u0027s ``volume_clear``"},{"line_number":336,"context_line":"pattern for LVM volumes."},{"line_number":337,"context_line":""},{"line_number":338,"context_line":"The resolved action is stored in ``std_board_info`` alongside existing"},{"line_number":339,"context_line":"device metadata. The selected cleanup action is fixed for the device"},{"line_number":340,"context_line":"until the next discovery cycle or agent restart. There is no runtime"},{"line_number":341,"context_line":"fallback. If the"}],"source_content_type":"text/x-rst","patch_set":21,"id":"daddf321_63e95280","line":338,"in_reply_to":"24b0d5d0_98a844d8","updated":"2026-06-30 19:46:58.000000000","message":"hum that an imporant implation detail\n\nthere is a tradeoff here.  it woudl be godo to now allow this to change while its bound, to a vm however if its unbond and in error i think we shoudl allow this  to change in the evnet that falling back to zero is required to enabel cleanign to compelte.\n\ni woudl be ok defering this detail to the impleation however.\nbut yes we dont want this to change  while a device is allcoated to an isntance as it coudl invalidate the schduling request if they reqeusted a specific cleanign mdoe via a trait","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":false,"context_lines":[{"line_number":335,"context_line":"host-side zeroing via ``shred``, following Nova\u0027s ``volume_clear``"},{"line_number":336,"context_line":"pattern for LVM volumes."},{"line_number":337,"context_line":""},{"line_number":338,"context_line":"The resolved action is stored in ``std_board_info`` alongside existing"},{"line_number":339,"context_line":"device metadata. The selected cleanup action is fixed for the device"},{"line_number":340,"context_line":"until the next discovery cycle or agent restart. There is no runtime"},{"line_number":341,"context_line":"fallback. If the"}],"source_content_type":"text/x-rst","patch_set":21,"id":"4be953ea_44667288","line":338,"in_reply_to":"daddf321_63e95280","updated":"2026-07-01 07:56:58.000000000","message":"Done, \n\nAdded rule: discover() must not overwrite cleanup action in std_board_info while device_state is not available; in error state the action may be updated.\n\nWe will discuss more during implementation.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"996d94b699b731074d6015c3bc443bcda9cca1df","unresolved":false,"context_lines":[{"line_number":479,"context_line":"added to the ``Device`` versioned object (``cyborg/objects/device.py``)"},{"line_number":480,"context_line":"and the ``devices`` database table. ``device_state`` is added to all"},{"line_number":481,"context_line":"devices regardless of type; every device follows the same state"},{"line_number":482,"context_line":"transitions with no type-specific checks. The bind guard and"},{"line_number":483,"context_line":"Placement ``reserved`` updates also apply uniformly. This is"},{"line_number":484,"context_line":"separate from the existing ``status`` field which remains"},{"line_number":485,"context_line":"exclusively for the enabled/disabled scheduling control."}],"source_content_type":"text/x-rst","patch_set":21,"id":"7b5d89b0_d81576f2","line":482,"updated":"2026-06-30 14:07:11.000000000","message":"The device_state bind guard, Placement reserved changes, and init_host() reconciliation are applied to ALL device types (not just NVMe), but the spec only validates the NVMe cleanup path. The bind guard rejects bind when device_state !\u003d available for every device.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: If an existing driver, ARQ allocation pattern, or a DB row left by a pre-migration agent yields a device_state that is not \u0027available\u0027 at bind time, device allocation for that type will be silently rejected after upgrade. GPU/FPGA passthrough could break in production.\n\n**Priority**: Before merge\n**Why This Matters**: init_host() reconciles all devices on restart: devices with bound ARQs move to \u0027allocated\u0027, stale states move to \u0027error\u0027. If reconciliation miscounts ARQs for non-NVMe types, those devices land in \u0027error\u0027 and become unallocatable. The spec does not analyse the ARQ query for non-NVMe types.\n\n**Recommendation**:\nAdd a subsection clarifying: (1) the exact ARQ-state query used to distinguish \u0027allocated\u0027 from \u0027missed cleanup\u0027, confirmed for all device types; (2) that the migration default\u003d\u0027available\u0027 means no healthy device is rejected on first boot; (3) a rollback if reconciliation wrongly fences off non-NVMe devices. Consider gating the bind guard behind the new microversion initially.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":true,"context_lines":[{"line_number":487,"context_line":"The state transitions are (``reserved\u003dtotal`` for all states"},{"line_number":488,"context_line":"except ``available`` which has ``reserved\u003d0``)::"},{"line_number":489,"context_line":""},{"line_number":490,"context_line":"    # Normal lifecycle"},{"line_number":491,"context_line":"    available ──(bind)──► allocated"},{"line_number":492,"context_line":"    allocated ──(unbind)──► pending_cleaning"},{"line_number":493,"context_line":"    pending_cleaning ──(agent starts cleanup)──► cleaning"},{"line_number":494,"context_line":"    cleaning ──(success)──► available"},{"line_number":495,"context_line":"    # Non-NVMe: cleanup() is a no-op"},{"line_number":496,"context_line":"    allocated ──(unbind)──► pending_cleaning ──► cleaning ──► available"},{"line_number":497,"context_line":""},{"line_number":498,"context_line":"    # Failure paths"},{"line_number":499,"context_line":"    cleaning ──(failure / timeout)──► error"},{"line_number":500,"context_line":""},{"line_number":501,"context_line":"    # Agent restart reconciliation"},{"line_number":502,"context_line":"    allocated ──(init_host: missed cleanup)──► pending_cleaning"},{"line_number":503,"context_line":"    pending_cleaning ──(init_host: crash recovery)──► error"},{"line_number":504,"context_line":"    cleaning ──(init_host: crash recovery)──► error"},{"line_number":505,"context_line":""},{"line_number":506,"context_line":"    # Operator recovery"},{"line_number":507,"context_line":"    error ──(operator POST /clean)──► pending_cleaning"},{"line_number":508,"context_line":""},{"line_number":509,"context_line":"A device starts in ``available`` state with ``reserved\u003d0``, meaning it"},{"line_number":510,"context_line":"is clean and ready for allocation. When bound to an instance,"},{"line_number":511,"context_line":"``reserved`` is set to ``total`` and the device moves to ``allocated``."}],"source_content_type":"text/x-rst","patch_set":21,"id":"c8b95de3_8d35e008","line":508,"range":{"start_line":490,"start_character":0,"end_line":508,"end_character":1},"updated":"2026-06-30 19:46:58.000000000","message":"```suggestion\n  +-----------+       bind       +-----------+\n  | available |-----------------\u003e| allocated |\n  +-----------+                  +-----------+\n        ^                              |\n        |                              | unbind /\n        | success                      | init_host: missed cleanup\n        |                              v\n  +-----------+   agent starts   +------------------+\n  |  cleaning |\u003c-----------------| pending_cleaning |\u003c---------+\n  +-----------+   / no-op        +------------------+          |\n        |                              |                       |\n        | failure / timeout /          | init_host:            |\n        | init_host: crash rec.        | crash rec.            | operator\n        v                              v                       | POST /clean\n  +------------------------------------------------------------+\n  |                           error                            |\n  +------------------------------------------------------------+\n```\n\nthis is a bit cleaner but what you yhave is also fine\n\nwe used to use https://asciiflow.com/#/ to create these but you can just ask gmini to create them its pretty good at taht.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":false,"context_lines":[{"line_number":487,"context_line":"The state transitions are (``reserved\u003dtotal`` for all states"},{"line_number":488,"context_line":"except ``available`` which has ``reserved\u003d0``)::"},{"line_number":489,"context_line":""},{"line_number":490,"context_line":"    # Normal lifecycle"},{"line_number":491,"context_line":"    available ──(bind)──► allocated"},{"line_number":492,"context_line":"    allocated ──(unbind)──► pending_cleaning"},{"line_number":493,"context_line":"    pending_cleaning ──(agent starts cleanup)──► cleaning"},{"line_number":494,"context_line":"    cleaning ──(success)──► available"},{"line_number":495,"context_line":"    # Non-NVMe: cleanup() is a no-op"},{"line_number":496,"context_line":"    allocated ──(unbind)──► pending_cleaning ──► cleaning ──► available"},{"line_number":497,"context_line":""},{"line_number":498,"context_line":"    # Failure paths"},{"line_number":499,"context_line":"    cleaning ──(failure / timeout)──► error"},{"line_number":500,"context_line":""},{"line_number":501,"context_line":"    # Agent restart reconciliation"},{"line_number":502,"context_line":"    allocated ──(init_host: missed cleanup)──► pending_cleaning"},{"line_number":503,"context_line":"    pending_cleaning ──(init_host: crash recovery)──► error"},{"line_number":504,"context_line":"    cleaning ──(init_host: crash recovery)──► error"},{"line_number":505,"context_line":""},{"line_number":506,"context_line":"    # Operator recovery"},{"line_number":507,"context_line":"    error ──(operator POST /clean)──► pending_cleaning"},{"line_number":508,"context_line":""},{"line_number":509,"context_line":"A device starts in ``available`` state with ``reserved\u003d0``, meaning it"},{"line_number":510,"context_line":"is clean and ready for allocation. When bound to an instance,"},{"line_number":511,"context_line":"``reserved`` is set to ``total`` and the device moves to ``allocated``."}],"source_content_type":"text/x-rst","patch_set":21,"id":"a5e31d43_9fef0acb","line":508,"range":{"start_line":490,"start_character":0,"end_line":508,"end_character":1},"in_reply_to":"c8b95de3_8d35e008","updated":"2026-07-01 07:56:58.000000000","message":"Done","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":true,"context_lines":[{"line_number":613,"context_line":""},{"line_number":614,"context_line":"Data model impact"},{"line_number":615,"context_line":"-----------------"},{"line_number":616,"context_line":""},{"line_number":617,"context_line":"One additive Alembic migration is needed. A new ``device_state``"},{"line_number":618,"context_line":"column is added to the ``devices`` table as an Enum column over the"},{"line_number":619,"context_line":"values ``available``, ``allocated``, ``pending_cleaning``,"},{"line_number":620,"context_line":"``cleaning``, and ``error``, following the same pattern as the"},{"line_number":621,"context_line":"existing ``status`` column migration"},{"line_number":622,"context_line":"(``cyborg/db/sqlalchemy/models.py``). The column defaults to"},{"line_number":623,"context_line":"``available`` and is ``NOT NULL``. All existing device rows receive"},{"line_number":624,"context_line":"``available`` as the default value; no data-dependent backfill is"},{"line_number":625,"context_line":"performed. On next agent restart, ``init_host()`` reconciles"},{"line_number":626,"context_line":"``device_state`` with actual ARQ allocations — devices with bound"},{"line_number":627,"context_line":"ARQs transition to ``allocated``, and devices in stale cleanup states"},{"line_number":628,"context_line":"move to ``error``."},{"line_number":629,"context_line":""},{"line_number":630,"context_line":"``device_state`` and ``status`` are orthogonal. A device with"},{"line_number":631,"context_line":"``status\u003dmaintaining`` can be in any ``device_state`` —"},{"line_number":632,"context_line":"``maintaining`` prevents new scheduling, ``device_state`` tracks"}],"source_content_type":"text/x-rst","patch_set":21,"id":"8af414eb_baaa4c83","line":629,"range":{"start_line":616,"start_character":1,"end_line":629,"end_character":1},"updated":"2026-06-30 19:46:58.000000000","message":"this is not correct.\n\nfor any device that is currently bound we will need to backfil\n\nthe db column with the correct state. (allocated or available)\n\nand because we dont want to break upgrades we will need to intially allow the collum to be null and either fix it on conductor start of with an online data migration the same way we did for the project id.\n\nso the schema migration will add the new colme with NULL for all rows\nan online data migration will need to be added here\nhttps://github.com/openstack/cyborg/blob/master/cyborg/cmd/dbsync.py#L47-L49\nhttps://github.com/openstack/cyborg/blob/master/cyborg/common/data_migrations.py\nand it need to also be added here\nhttps://github.com/openstack/cyborg/blob/master/cyborg/conductor/manager.py#L51\nand a cybrog status check\nhere https://github.com/openstack/cyborg/blob/master/cyborg/cmd/status.py\nto check that all row are backfiled\n\nin teh future (2027.2) we will be able to add a contract migration to remove the nullablity from the column.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"cccfbd3a0270cec494709953de9b5bcf06843730","unresolved":false,"context_lines":[{"line_number":613,"context_line":""},{"line_number":614,"context_line":"Data model impact"},{"line_number":615,"context_line":"-----------------"},{"line_number":616,"context_line":""},{"line_number":617,"context_line":"One additive Alembic migration is needed. A new ``device_state``"},{"line_number":618,"context_line":"column is added to the ``devices`` table as an Enum column over the"},{"line_number":619,"context_line":"values ``available``, ``allocated``, ``pending_cleaning``,"},{"line_number":620,"context_line":"``cleaning``, and ``error``, following the same pattern as the"},{"line_number":621,"context_line":"existing ``status`` column migration"},{"line_number":622,"context_line":"(``cyborg/db/sqlalchemy/models.py``). The column defaults to"},{"line_number":623,"context_line":"``available`` and is ``NOT NULL``. All existing device rows receive"},{"line_number":624,"context_line":"``available`` as the default value; no data-dependent backfill is"},{"line_number":625,"context_line":"performed. On next agent restart, ``init_host()`` reconciles"},{"line_number":626,"context_line":"``device_state`` with actual ARQ allocations — devices with bound"},{"line_number":627,"context_line":"ARQs transition to ``allocated``, and devices in stale cleanup states"},{"line_number":628,"context_line":"move to ``error``."},{"line_number":629,"context_line":""},{"line_number":630,"context_line":"``device_state`` and ``status`` are orthogonal. A device with"},{"line_number":631,"context_line":"``status\u003dmaintaining`` can be in any ``device_state`` —"},{"line_number":632,"context_line":"``maintaining`` prevents new scheduling, ``device_state`` tracks"}],"source_content_type":"text/x-rst","patch_set":21,"id":"c5621819_5dfba7c7","line":629,"range":{"start_line":616,"start_character":1,"end_line":629,"end_character":1},"in_reply_to":"5c351f73_e3038d2c","updated":"2026-07-01 09:15:11.000000000","message":"so im a little conflicted on \n\nConductor startup via ``init_host()``\n  (``cyborg/conductor/manager.py``)\nAgent restart via ``init_host()`` for state reconciliation\n\n\nreally we want the conductors to do as little work as possible.\n\nthere role is to provide access to the db to the agents and to offload any long running operation form teh api.\n\nso while it would be  better to have the agents manage this the way they have to do it is a diffent impleaiton then how we woudl do it in the cli and conductor as that can simply update the db directly.\n\nso this is fien for now but we may revisit this when we get to the implementation review","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":4690,"name":"melanie witt","display_name":"melwitt","email":"melwittt@gmail.com","username":"melwitt"},"change_message_id":"922b11619a94ac898f9fe9546e5ab76a65536422","unresolved":true,"context_lines":[{"line_number":613,"context_line":""},{"line_number":614,"context_line":"Data model impact"},{"line_number":615,"context_line":"-----------------"},{"line_number":616,"context_line":""},{"line_number":617,"context_line":"One additive Alembic migration is needed. A new ``device_state``"},{"line_number":618,"context_line":"column is added to the ``devices`` table as an Enum column over the"},{"line_number":619,"context_line":"values ``available``, ``allocated``, ``pending_cleaning``,"},{"line_number":620,"context_line":"``cleaning``, and ``error``, following the same pattern as the"},{"line_number":621,"context_line":"existing ``status`` column migration"},{"line_number":622,"context_line":"(``cyborg/db/sqlalchemy/models.py``). The column defaults to"},{"line_number":623,"context_line":"``available`` and is ``NOT NULL``. All existing device rows receive"},{"line_number":624,"context_line":"``available`` as the default value; no data-dependent backfill is"},{"line_number":625,"context_line":"performed. On next agent restart, ``init_host()`` reconciles"},{"line_number":626,"context_line":"``device_state`` with actual ARQ allocations — devices with bound"},{"line_number":627,"context_line":"ARQs transition to ``allocated``, and devices in stale cleanup states"},{"line_number":628,"context_line":"move to ``error``."},{"line_number":629,"context_line":""},{"line_number":630,"context_line":"``device_state`` and ``status`` are orthogonal. A device with"},{"line_number":631,"context_line":"``status\u003dmaintaining`` can be in any ``device_state`` —"},{"line_number":632,"context_line":"``maintaining`` prevents new scheduling, ``device_state`` tracks"}],"source_content_type":"text/x-rst","patch_set":21,"id":"89131793_de65ab86","line":629,"range":{"start_line":616,"start_character":1,"end_line":629,"end_character":1},"in_reply_to":"73e3eb67_506021d5","updated":"2026-06-30 23:40:23.000000000","message":"*cannot default existing bound devices to \"available\"","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":true,"context_lines":[{"line_number":613,"context_line":""},{"line_number":614,"context_line":"Data model impact"},{"line_number":615,"context_line":"-----------------"},{"line_number":616,"context_line":""},{"line_number":617,"context_line":"One additive Alembic migration is needed. A new ``device_state``"},{"line_number":618,"context_line":"column is added to the ``devices`` table as an Enum column over the"},{"line_number":619,"context_line":"values ``available``, ``allocated``, ``pending_cleaning``,"},{"line_number":620,"context_line":"``cleaning``, and ``error``, following the same pattern as the"},{"line_number":621,"context_line":"existing ``status`` column migration"},{"line_number":622,"context_line":"(``cyborg/db/sqlalchemy/models.py``). The column defaults to"},{"line_number":623,"context_line":"``available`` and is ``NOT NULL``. All existing device rows receive"},{"line_number":624,"context_line":"``available`` as the default value; no data-dependent backfill is"},{"line_number":625,"context_line":"performed. On next agent restart, ``init_host()`` reconciles"},{"line_number":626,"context_line":"``device_state`` with actual ARQ allocations — devices with bound"},{"line_number":627,"context_line":"ARQs transition to ``allocated``, and devices in stale cleanup states"},{"line_number":628,"context_line":"move to ``error``."},{"line_number":629,"context_line":""},{"line_number":630,"context_line":"``device_state`` and ``status`` are orthogonal. A device with"},{"line_number":631,"context_line":"``status\u003dmaintaining`` can be in any ``device_state`` —"},{"line_number":632,"context_line":"``maintaining`` prevents new scheduling, ``device_state`` tracks"}],"source_content_type":"text/x-rst","patch_set":21,"id":"5c351f73_e3038d2c","line":629,"range":{"start_line":616,"start_character":1,"end_line":629,"end_character":1},"in_reply_to":"89131793_de65ab86","updated":"2026-07-01 07:56:58.000000000","message":"Done, I have rewritten that section as follows:\n-  column is nullable, existing rows start as NULL\n- online data migration follows heal_arq_project_ids pattern across all four files you mentioned\n-  cyborg-status upgrade check added.\n\nAby thing else we need here?","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":4690,"name":"melanie witt","display_name":"melwitt","email":"melwittt@gmail.com","username":"melwitt"},"change_message_id":"632b308da857eec01c8fa8306095d056b78ca864","unresolved":true,"context_lines":[{"line_number":613,"context_line":""},{"line_number":614,"context_line":"Data model impact"},{"line_number":615,"context_line":"-----------------"},{"line_number":616,"context_line":""},{"line_number":617,"context_line":"One additive Alembic migration is needed. A new ``device_state``"},{"line_number":618,"context_line":"column is added to the ``devices`` table as an Enum column over the"},{"line_number":619,"context_line":"values ``available``, ``allocated``, ``pending_cleaning``,"},{"line_number":620,"context_line":"``cleaning``, and ``error``, following the same pattern as the"},{"line_number":621,"context_line":"existing ``status`` column migration"},{"line_number":622,"context_line":"(``cyborg/db/sqlalchemy/models.py``). The column defaults to"},{"line_number":623,"context_line":"``available`` and is ``NOT NULL``. All existing device rows receive"},{"line_number":624,"context_line":"``available`` as the default value; no data-dependent backfill is"},{"line_number":625,"context_line":"performed. On next agent restart, ``init_host()`` reconciles"},{"line_number":626,"context_line":"``device_state`` with actual ARQ allocations — devices with bound"},{"line_number":627,"context_line":"ARQs transition to ``allocated``, and devices in stale cleanup states"},{"line_number":628,"context_line":"move to ``error``."},{"line_number":629,"context_line":""},{"line_number":630,"context_line":"``device_state`` and ``status`` are orthogonal. A device with"},{"line_number":631,"context_line":"``status\u003dmaintaining`` can be in any ``device_state`` —"},{"line_number":632,"context_line":"``maintaining`` prevents new scheduling, ``device_state`` tracks"}],"source_content_type":"text/x-rst","patch_set":21,"id":"73e3eb67_506021d5","line":629,"range":{"start_line":616,"start_character":1,"end_line":629,"end_character":1},"in_reply_to":"8af414eb_baaa4c83","updated":"2026-06-30 23:38:40.000000000","message":"+1 we cannot default existing devices state to \"available\". Will need the data migration etc.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"996d94b699b731074d6015c3bc443bcda9cca1df","unresolved":false,"context_lines":[{"line_number":636,"context_line":"currently contains ``GPU``, ``FPGA``, ``AICHIP``, ``QAT``, ``NIC``,"},{"line_number":637,"context_line":"and ``SSD``."},{"line_number":638,"context_line":""},{"line_number":639,"context_line":"This is expand-only with no contract phase. The ``Device``"},{"line_number":640,"context_line":"oslo.versionedobjects definition (``cyborg/objects/device.py``,"},{"line_number":641,"context_line":"currently version ``1.2``) is bumped to include the ``device_state``"},{"line_number":642,"context_line":"field. The existing ``status`` field remains unchanged and continues"}],"source_content_type":"text/x-rst","patch_set":21,"id":"762b9aa1_13becf72","line":639,"updated":"2026-06-30 14:07:11.000000000","message":"The data model section says the migration is expand-only with no contract phase, but does not state the target Device oslo.versionedobjects version (currently 1.2) or how N-1 services handle the new field.\n\n**Severity**: SUGGESTION | **Confidence**: 0.8\n\n**Benefit**: An explicit object version bump target (1.2 -\u003e 1.3) and N-1 read behaviour avoids a contract/expand ambiguity that Cyborg\u0027s partial rolling-upgrade support makes risky.\n\n**Recommendation**:\nState the target Device object version, confirm the new field is optional on read for older services, and cross-reference the Upgrade impact note that all services must move together.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":true,"context_lines":[{"line_number":637,"context_line":"and ``SSD``."},{"line_number":638,"context_line":""},{"line_number":639,"context_line":"This is expand-only with no contract phase. The ``Device``"},{"line_number":640,"context_line":"oslo.versionedobjects definition (``cyborg/objects/device.py``,"},{"line_number":641,"context_line":"currently version ``1.2``) is bumped to include the ``device_state``"},{"line_number":642,"context_line":"field. The existing ``status`` field remains unchanged and continues"},{"line_number":643,"context_line":"to serve its enabled/disabled purpose."},{"line_number":644,"context_line":""},{"line_number":645,"context_line":""},{"line_number":646,"context_line":"REST API impact"}],"source_content_type":"text/x-rst","patch_set":21,"id":"a07f63a7_3969af60","line":643,"range":{"start_line":640,"start_character":0,"end_line":643,"end_character":38},"updated":"2026-06-30 19:46:58.000000000","message":"so yes we need a new object version but we also need to provide a make compatible method to downlevel the responce to supprot cybrog agent that have not been upgraded yet.\n\nthis is a built in capablityof oslo.versioned object but cyborg has not been using it so we just need to make sure we do it\n\nit looks like this \nhttps://github.com/openstack/nova/blob/master/nova/objects/image_meta.py#L214\n\nso it just an if chain and poping the filed when downlevling.\n\nby the way i think we are also going to want to know if a device device supprot cleaning\n\nwe coudl start with a property that just returns self.type \u003d\u003d \"NVME\"\n\nthat woudl be enought to knwo if the condoctor shoudl call clean and allow use to shrotcut non nvme device form allocatd directly to aviable on unbind.\n\nlong term\n\ni think we shoudl likely add a device_metadata json blob weher we can store capablities like `SUPPORTS_CLEANING` or `SUPPORTS_PROGRAMING`\n\n```\ndevice_metadata \u003d {\"capablities\":[SUPPORTS_CLEANING]}\n```\nso we can make these types of decision in a driver independent way.\n\nthe reaons im hesitent to do that right now is because fo the atibutes api\n\nso for right now i guess we coudl store the capablites as atibute on the device in the atribtus tables and just skip reportign thos as traits\n\nso CAP_SUPPORTS_CLEANING\u003dTrue|False\n\nfor this spec i think we can do the device tyep check but in the future we shoudl factor this in to the new deriver framework spec.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":false,"context_lines":[{"line_number":637,"context_line":"and ``SSD``."},{"line_number":638,"context_line":""},{"line_number":639,"context_line":"This is expand-only with no contract phase. The ``Device``"},{"line_number":640,"context_line":"oslo.versionedobjects definition (``cyborg/objects/device.py``,"},{"line_number":641,"context_line":"currently version ``1.2``) is bumped to include the ``device_state``"},{"line_number":642,"context_line":"field. The existing ``status`` field remains unchanged and continues"},{"line_number":643,"context_line":"to serve its enabled/disabled purpose."},{"line_number":644,"context_line":""},{"line_number":645,"context_line":""},{"line_number":646,"context_line":"REST API impact"}],"source_content_type":"text/x-rst","patch_set":21,"id":"3aa9e23e_35064d32","line":643,"range":{"start_line":640,"start_character":0,"end_line":643,"end_character":38},"in_reply_to":"a07f63a7_3969af60","updated":"2026-07-01 07:56:58.000000000","message":"Done,\n\nI have added obj_make_compatible() (1.2 → 1.3),  supports_cleaning property.\n\nI have also mived driver-independent capability model to out-of-scope section with driver metadata json and attibutes.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"996d94b699b731074d6015c3bc443bcda9cca1df","unresolved":false,"context_lines":[{"line_number":646,"context_line":"REST API impact"},{"line_number":647,"context_line":"---------------"},{"line_number":648,"context_line":""},{"line_number":649,"context_line":"A new microversion is required. The exact version number depends on"},{"line_number":650,"context_line":"merge ordering with other in-flight specs; this spec claims the next"},{"line_number":651,"context_line":"available microversion after its dependencies land. The microversion"},{"line_number":652,"context_line":"is driven by the new ``device_state`` field in device responses and the"}],"source_content_type":"text/x-rst","patch_set":21,"id":"f4f584ae_d821f7df","line":649,"updated":"2026-06-30 14:07:11.000000000","message":"The REST API section defers the exact microversion number (\u0027the exact version number depends on merge ordering\u0027) with no placeholder, no claim range, and no list of conflicting in-flight specs. Phase 5 (SDK/client) implementers cannot be ordered without this.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: Cross-spec microversion collisions become likely at merge time. The APIImpact commit flag promises a concrete API change but the spec leaves its key identifier unresolved, blocking the SDK/client work items.\n\n**Suggestion**:\nState the current maximum Cyborg microversion and claim the next available number (e.g. \u0027X.42\u0027) with a note that it will be renumbered if another spec merges first, mirroring how Nova specs handle ordering.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":false,"context_lines":[{"line_number":646,"context_line":"REST API impact"},{"line_number":647,"context_line":"---------------"},{"line_number":648,"context_line":""},{"line_number":649,"context_line":"A new microversion is required. The exact version number depends on"},{"line_number":650,"context_line":"merge ordering with other in-flight specs; this spec claims the next"},{"line_number":651,"context_line":"available microversion after its dependencies land. The microversion"},{"line_number":652,"context_line":"is driven by the new ``device_state`` field in device responses and the"}],"source_content_type":"text/x-rst","patch_set":21,"id":"e79e1474_1d13c137","line":649,"in_reply_to":"e1127f81_067cb701","updated":"2026-07-01 07:56:58.000000000","message":"Acknowledged","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":true,"context_lines":[{"line_number":646,"context_line":"REST API impact"},{"line_number":647,"context_line":"---------------"},{"line_number":648,"context_line":""},{"line_number":649,"context_line":"A new microversion is required. The exact version number depends on"},{"line_number":650,"context_line":"merge ordering with other in-flight specs; this spec claims the next"},{"line_number":651,"context_line":"available microversion after its dependencies land. The microversion"},{"line_number":652,"context_line":"is driven by the new ``device_state`` field in device responses and the"}],"source_content_type":"text/x-rst","patch_set":21,"id":"e1127f81_067cb701","line":649,"in_reply_to":"f4f584ae_d821f7df","updated":"2026-06-30 19:46:58.000000000","message":"no this is fine\n\nit will likely be 2.5 but we do not need to allcoate the microvsion that will be used here.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":false,"context_lines":[{"line_number":652,"context_line":"is driven by the new ``device_state`` field in device responses and the"},{"line_number":653,"context_line":"new ``POST /v2/devices/{uuid}/clean`` endpoint. Adding ``NVME`` to the"},{"line_number":654,"context_line":"``type`` field does not require a microversion because the type field"},{"line_number":655,"context_line":"is ``wtypes.text`` (free-form string) at the API layer."},{"line_number":656,"context_line":""},{"line_number":657,"context_line":"For API microversions lower than the new version, ``GET /v2/devices``"},{"line_number":658,"context_line":"and ``GET /v2/devices/{uuid}`` responses remain as today. For the new"}],"source_content_type":"text/x-rst","patch_set":21,"id":"ff01439c_d4554c7a","line":655,"updated":"2026-06-30 19:46:58.000000000","message":"more preciesly beause its docuemnted as a sting in the api ref but yes\n\nwe can make this an enum in the future at the api level if we want too but at that point we also need to filter devices base don microversion","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":true,"context_lines":[{"line_number":705,"context_line":""},{"line_number":706,"context_line":"Normal response code: ``202 Accepted``"},{"line_number":707,"context_line":""},{"line_number":708,"context_line":"Error response codes:"},{"line_number":709,"context_line":""},{"line_number":710,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":711,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"},{"line_number":712,"context_line":"* ``409 Conflict`` — device is in ``allocated`` state (bound to an"},{"line_number":713,"context_line":"  instance)"},{"line_number":714,"context_line":"* ``409 Conflict`` — device is already in ``cleaning`` or"},{"line_number":715,"context_line":"  ``pending_cleaning`` state"},{"line_number":716,"context_line":""},{"line_number":717,"context_line":"Example::"},{"line_number":718,"context_line":""}],"source_content_type":"text/x-rst","patch_set":21,"id":"41f41c5c_310a9baf","line":715,"range":{"start_line":708,"start_character":0,"end_line":715,"end_character":28},"updated":"2026-06-30 19:46:58.000000000","message":"so we shoudl decide what the return code is if a device does not support cleaning\n\ni think that shoudl be a 400 bade request\n\ni.e. if you call this on a device manged by the pci driver.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":false,"context_lines":[{"line_number":705,"context_line":""},{"line_number":706,"context_line":"Normal response code: ``202 Accepted``"},{"line_number":707,"context_line":""},{"line_number":708,"context_line":"Error response codes:"},{"line_number":709,"context_line":""},{"line_number":710,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":711,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"},{"line_number":712,"context_line":"* ``409 Conflict`` — device is in ``allocated`` state (bound to an"},{"line_number":713,"context_line":"  instance)"},{"line_number":714,"context_line":"* ``409 Conflict`` — device is already in ``cleaning`` or"},{"line_number":715,"context_line":"  ``pending_cleaning`` state"},{"line_number":716,"context_line":""},{"line_number":717,"context_line":"Example::"},{"line_number":718,"context_line":""}],"source_content_type":"text/x-rst","patch_set":21,"id":"d12347c2_8c1cca9c","line":715,"range":{"start_line":708,"start_character":0,"end_line":715,"end_character":28},"in_reply_to":"03ab3298_575d1c79","updated":"2026-07-01 07:56:58.000000000","message":"Done. Added 400 Bad Request — device does not support cleaning","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":4690,"name":"melanie witt","display_name":"melwitt","email":"melwittt@gmail.com","username":"melwitt"},"change_message_id":"632b308da857eec01c8fa8306095d056b78ca864","unresolved":true,"context_lines":[{"line_number":705,"context_line":""},{"line_number":706,"context_line":"Normal response code: ``202 Accepted``"},{"line_number":707,"context_line":""},{"line_number":708,"context_line":"Error response codes:"},{"line_number":709,"context_line":""},{"line_number":710,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":711,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"},{"line_number":712,"context_line":"* ``409 Conflict`` — device is in ``allocated`` state (bound to an"},{"line_number":713,"context_line":"  instance)"},{"line_number":714,"context_line":"* ``409 Conflict`` — device is already in ``cleaning`` or"},{"line_number":715,"context_line":"  ``pending_cleaning`` state"},{"line_number":716,"context_line":""},{"line_number":717,"context_line":"Example::"},{"line_number":718,"context_line":""}],"source_content_type":"text/x-rst","patch_set":21,"id":"03ab3298_575d1c79","line":715,"range":{"start_line":708,"start_character":0,"end_line":715,"end_character":28},"in_reply_to":"41f41c5c_310a9baf","updated":"2026-06-30 23:38:40.000000000","message":"+1 I think 400 Bad Request would be appropriate for POST /clean to a device that does not support cleaning.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":false,"context_lines":[{"line_number":718,"context_line":""},{"line_number":719,"context_line":"  POST /v2/devices/3f2a8b9c-1e5d-4c7b-a3f6-8d2e9c1b4a7f/clean"},{"line_number":720,"context_line":""},{"line_number":721,"context_line":"  HTTP/1.1 202 Accepted"},{"line_number":722,"context_line":""},{"line_number":723,"context_line":"Policy is unchanged for existing endpoints. The new ``POST"},{"line_number":724,"context_line":"/v2/devices/{uuid}/clean`` endpoint is governed by the"}],"source_content_type":"text/x-rst","patch_set":21,"id":"ab4ced8d_5a07a2ca","line":721,"range":{"start_line":721,"start_character":11,"end_line":721,"end_character":23},"updated":"2026-06-30 19:46:58.000000000","message":"yes 202 is correct as this is asyc","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":true,"context_lines":[{"line_number":744,"context_line":"The conductor calls this using ``cctxt.cast()`` (async,"},{"line_number":745,"context_line":"fire-and-forget)::"},{"line_number":746,"context_line":""},{"line_number":747,"context_line":"    cctxt \u003d self.agent.client.prepare("},{"line_number":748,"context_line":"        server\u003dhostname, version\u003d\u00271.1\u0027)"},{"line_number":749,"context_line":"    cctxt.cast(context, \u0027cleanup_device\u0027, device\u003ddevice)"},{"line_number":750,"context_line":""},{"line_number":751,"context_line":"The RPC returns immediately so Nova\u0027s instance deletion completes"},{"line_number":752,"context_line":"without waiting for cleanup. The agent manages all ``device_state``"}],"source_content_type":"text/x-rst","patch_set":21,"id":"a1d26a16_d9d8d0d2","line":749,"range":{"start_line":747,"start_character":1,"end_line":749,"end_character":56},"updated":"2026-06-30 19:46:58.000000000","message":"i belive one of those two can fail fi the target system does not support that rpc verison\n\nhttps://github.com/openstack/nova/blob/e020a01835754c09719a6616a1859f3d01f8cc5f/nova/scheduler/rpcapi.py#L132-L160\n\nso you actully need to backlevel the rpc call like this\n\n```\n  version \u003d \u00274.5\u0027\n        msg_args \u003d {\u0027instance_uuids\u0027: instance_uuids,\n                    \u0027spec_obj\u0027: spec_obj,\n                    \u0027return_objects\u0027: return_objects,\n                    \u0027return_alternates\u0027: return_alternates}\n        if not self.client.can_send_version(version):\n            if msg_args[\u0027return_objects\u0027] or msg_args[\u0027return_alternates\u0027]:\n                # The client is requesting an RPC version we can\u0027t support.\n                raise exc.SelectionObjectsWithOldRPCVersionNotSupported(\n                        version\u003dself.client.version_cap)\n            del msg_args[\u0027return_objects\u0027]\n            del msg_args[\u0027return_alternates\u0027]\n            version \u003d \u00274.4\u0027\n```\n\nin this specific case we know that a cyborg agent that does not support 1.1 also has no cleanabel devices so we can set the device state directly to avaible\n\n```\n if not self.client.can_send_version(version):\n   device.device_sate\u003d\"Avialable\"\n   device.save()\n   return\n cctxt.cast(context, \u0027cleanup_device\u0027, device\u003ddevice)\n````\nin general whoever we would normally fail the api request.\n\nif this was nova we woudl use hte compute service version to detect if the api call was valid for the given hsot at the api but we cant do it here until we get to the conductor.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":true,"context_lines":[{"line_number":744,"context_line":"The conductor calls this using ``cctxt.cast()`` (async,"},{"line_number":745,"context_line":"fire-and-forget)::"},{"line_number":746,"context_line":""},{"line_number":747,"context_line":"    cctxt \u003d self.agent.client.prepare("},{"line_number":748,"context_line":"        server\u003dhostname, version\u003d\u00271.1\u0027)"},{"line_number":749,"context_line":"    cctxt.cast(context, \u0027cleanup_device\u0027, device\u003ddevice)"},{"line_number":750,"context_line":""},{"line_number":751,"context_line":"The RPC returns immediately so Nova\u0027s instance deletion completes"},{"line_number":752,"context_line":"without waiting for cleanup. The agent manages all ``device_state``"}],"source_content_type":"text/x-rst","patch_set":21,"id":"bcd398f0_741cefaa","line":749,"range":{"start_line":747,"start_character":1,"end_line":749,"end_character":56},"in_reply_to":"a1d26a16_d9d8d0d2","updated":"2026-07-01 07:56:58.000000000","message":"Thank you for providing code\nI have added can_send_version guard following Nova\u0027s scheduler/rpcapi.py pattern.\n\nCan you check that section one more time?","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"cccfbd3a0270cec494709953de9b5bcf06843730","unresolved":false,"context_lines":[{"line_number":744,"context_line":"The conductor calls this using ``cctxt.cast()`` (async,"},{"line_number":745,"context_line":"fire-and-forget)::"},{"line_number":746,"context_line":""},{"line_number":747,"context_line":"    cctxt \u003d self.agent.client.prepare("},{"line_number":748,"context_line":"        server\u003dhostname, version\u003d\u00271.1\u0027)"},{"line_number":749,"context_line":"    cctxt.cast(context, \u0027cleanup_device\u0027, device\u003ddevice)"},{"line_number":750,"context_line":""},{"line_number":751,"context_line":"The RPC returns immediately so Nova\u0027s instance deletion completes"},{"line_number":752,"context_line":"without waiting for cleanup. The agent manages all ``device_state``"}],"source_content_type":"text/x-rst","patch_set":21,"id":"4a0f0b91_3e1ae07d","line":749,"range":{"start_line":747,"start_character":1,"end_line":749,"end_character":56},"in_reply_to":"bcd398f0_741cefaa","updated":"2026-07-01 09:15:11.000000000","message":"looks good\n\nwe can refine in the implemetion review if needed but this is correct at a high level","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":true,"context_lines":[{"line_number":829,"context_line":"Other deployer impact"},{"line_number":830,"context_line":"---------------------"},{"line_number":831,"context_line":""},{"line_number":832,"context_line":"Operators must install ``nvme-cli`` on compute nodes and set"},{"line_number":833,"context_line":"``[agent] enabled_drivers`` and ``[nvme] device_spec`` in"},{"line_number":834,"context_line":"``cyborg.conf``. The ``device_spec`` JSON accepts optional"},{"line_number":835,"context_line":"``clear_method`` (auto|sanitize|zero) and ``clear_mode``"}],"source_content_type":"text/x-rst","patch_set":21,"id":"564b997c_0a65abf8","line":832,"range":{"start_line":832,"start_character":25,"end_line":832,"end_character":33},"updated":"2026-06-30 19:46:58.000000000","message":"we shoudl remember to add this to bindeps.txt","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":false,"context_lines":[{"line_number":829,"context_line":"Other deployer impact"},{"line_number":830,"context_line":"---------------------"},{"line_number":831,"context_line":""},{"line_number":832,"context_line":"Operators must install ``nvme-cli`` on compute nodes and set"},{"line_number":833,"context_line":"``[agent] enabled_drivers`` and ``[nvme] device_spec`` in"},{"line_number":834,"context_line":"``cyborg.conf``. The ``device_spec`` JSON accepts optional"},{"line_number":835,"context_line":"``clear_method`` (auto|sanitize|zero) and ``clear_mode``"}],"source_content_type":"text/x-rst","patch_set":21,"id":"f656efc5_c0db6261","line":832,"range":{"start_line":832,"start_character":25,"end_line":832,"end_character":33},"in_reply_to":"564b997c_0a65abf8","updated":"2026-07-01 07:56:58.000000000","message":"Done","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"3f453a291e4c9614e315c3599569ffe42fc30a1e","unresolved":true,"context_lines":[{"line_number":878,"context_line":""},{"line_number":879,"context_line":"Cyborg does not yet have full rolling upgrade support, though grenade"},{"line_number":880,"context_line":"testing has been added recently and work is planned for the 2027.1"},{"line_number":881,"context_line":"cycle to support N-1 agent compatibility. All services must be"},{"line_number":882,"context_line":"upgraded together: conductor and API first, then agents. If an old"},{"line_number":883,"context_line":"agent receives the new ``cleanup_device`` RPC cast, the call fails"},{"line_number":884,"context_line":"silently and the device stays in ``allocated`` until the agent"},{"line_number":885,"context_line":"is upgraded. On restart, ``init_host()`` detects the device and moves"},{"line_number":886,"context_line":"it to ``error`` for operator remediation."},{"line_number":887,"context_line":""},{"line_number":888,"context_line":""},{"line_number":889,"context_line":"Implementation"}],"source_content_type":"text/x-rst","patch_set":21,"id":"2dcb22f1_8d29a7f2","line":886,"range":{"start_line":881,"start_character":42,"end_line":886,"end_character":41},"updated":"2026-06-30 19:46:58.000000000","message":"so this is not quite good enough\n\nfor non nvme devices we need to go directly to aviaable and ideally not\nmake the rpc by doign the can_send_version check.\n\nwe shoudl put the device in error fi the device.type is somehow NVME and can_send_version fails.\n\nwe need to start buildign rooling upgrade supprot as n+1 was the minitum and n+2 is not expected for all serivces so for this release we shodul supprot a mix of 2026.1 agents with 2026.2 api\u0027s/conductors.\n\ntrying to call clean on a host that is runing 2026.1 shoudl be a 400 bad request\n\nfor 2 reasons one no device supproted by 2026.1 support cleanign and secodn becase the can_send_version will fail wehn we try to inovke clean.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"b8b2acc50645b0c0426690441d611d65635cd607","unresolved":true,"context_lines":[{"line_number":878,"context_line":""},{"line_number":879,"context_line":"Cyborg does not yet have full rolling upgrade support, though grenade"},{"line_number":880,"context_line":"testing has been added recently and work is planned for the 2027.1"},{"line_number":881,"context_line":"cycle to support N-1 agent compatibility. All services must be"},{"line_number":882,"context_line":"upgraded together: conductor and API first, then agents. If an old"},{"line_number":883,"context_line":"agent receives the new ``cleanup_device`` RPC cast, the call fails"},{"line_number":884,"context_line":"silently and the device stays in ``allocated`` until the agent"},{"line_number":885,"context_line":"is upgraded. On restart, ``init_host()`` detects the device and moves"},{"line_number":886,"context_line":"it to ``error`` for operator remediation."},{"line_number":887,"context_line":""},{"line_number":888,"context_line":""},{"line_number":889,"context_line":"Implementation"}],"source_content_type":"text/x-rst","patch_set":21,"id":"eee0d10a_9f6779dd","line":886,"range":{"start_line":881,"start_character":42,"end_line":886,"end_character":41},"in_reply_to":"2dcb22f1_8d29a7f2","updated":"2026-07-01 07:56:58.000000000","message":"Done, I have rewritten the upgrade section based on above suggestions.\n\nIt includes: \n- N-1 agent compatibility\n- non-NVMe direct to available without RPC\n- NVMe + can_send_version failure to error\n- POST /clean on 2026.1 agent\n- 400 Bad Request\n\nDo let me know if more update is needed here. thank you!","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"cccfbd3a0270cec494709953de9b5bcf06843730","unresolved":false,"context_lines":[{"line_number":878,"context_line":""},{"line_number":879,"context_line":"Cyborg does not yet have full rolling upgrade support, though grenade"},{"line_number":880,"context_line":"testing has been added recently and work is planned for the 2027.1"},{"line_number":881,"context_line":"cycle to support N-1 agent compatibility. All services must be"},{"line_number":882,"context_line":"upgraded together: conductor and API first, then agents. If an old"},{"line_number":883,"context_line":"agent receives the new ``cleanup_device`` RPC cast, the call fails"},{"line_number":884,"context_line":"silently and the device stays in ``allocated`` until the agent"},{"line_number":885,"context_line":"is upgraded. On restart, ``init_host()`` detects the device and moves"},{"line_number":886,"context_line":"it to ``error`` for operator remediation."},{"line_number":887,"context_line":""},{"line_number":888,"context_line":""},{"line_number":889,"context_line":"Implementation"}],"source_content_type":"text/x-rst","patch_set":21,"id":"b5dc2da0_ed0d1abb","line":886,"range":{"start_line":881,"start_character":42,"end_line":886,"end_character":41},"in_reply_to":"eee0d10a_9f6779dd","updated":"2026-07-01 09:15:11.000000000","message":"Done","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"996d94b699b731074d6015c3bc443bcda9cca1df","unresolved":false,"context_lines":[{"line_number":923,"context_line":""},{"line_number":924,"context_line":"* Add generic ``cleanup_device`` RPC with ``cctxt.cast()``."},{"line_number":925,"context_line":"* Implement ``NVMeDriver.cleanup()`` with futurist thread pool, state"},{"line_number":926,"context_line":"  transitions, and Placement reserved handling on bind and unbind."},{"line_number":927,"context_line":"* For zero path: if more than one namespace exists, delete all and"},{"line_number":928,"context_line":"  create a single namespace (conditional on OACS bit 3)."},{"line_number":929,"context_line":"* Implement zero path: prefer ``nvme write-zeroes`` (ONCS bit 3), fall"}],"source_content_type":"text/x-rst","patch_set":21,"id":"17c051b5_3001ae74","line":926,"updated":"2026-06-30 14:07:11.000000000","message":"The Phase 3 work item says \u0027Placement reserved handling on bind and unbind\u0027, but unbind itself does not change reserved. The device stays reserved through pending_cleaning and cleaning; it is the cleanup-success path that unreserves (reserved\u003d0 only on transition to available).\n\n**Severity**: SUGGESTION | **Confidence**: 0.9\n\n**Benefit**: Removing the word \u0027unbind\u0027 prevents an implementer from adding an unreserve call in the unbind path, which would briefly make a device schedulable while it still holds tenant data.\n\n**Recommendation**:\nRephrase to \u0027Placement reserved handling on bind and on cleanup completion\u0027 to match the state machine invariants.","commit_id":"2f9710187ba66cd6696cb7725b97106c0e76e2c8"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"ce8cf7166e056b875ba7a2e847e81264102e44f4","unresolved":false,"context_lines":[{"line_number":285,"context_line":"The traits are defined in a new ``os_traits/hw/nvme/__init__.py``"},{"line_number":286,"context_line":"module::"},{"line_number":287,"context_line":""},{"line_number":288,"context_line":"    TRAITS \u003d ["},{"line_number":289,"context_line":"        \u0027CES\u0027,   # Crypto Erase Sanitize Supported (sanicap bit 0)"},{"line_number":290,"context_line":"        \u0027BES\u0027,   # Block Erase Sanitize Supported (sanicap bit 1)"},{"line_number":291,"context_line":"        \u0027WZS\u0027,   # Write Zeroes Supported (ONCS bit 3)"}],"source_content_type":"text/x-rst","patch_set":22,"id":"431d8cc7_74d06170","line":288,"updated":"2026-07-01 08:07:24.000000000","message":"TRAITS lists bare short names (\u0027CES\u0027,\u0027BES\u0027,\u0027WZS\u0027) but the Placement example on line 411 shows fully-qualified traits (HW_NVME_CES, HW_NVME_BES, HW_NVME_WZS). In os-traits constants map short names to fully-qualified strings. The list registers bare names, inconsistent with the example.\n\n**Severity**: HIGH | **Confidence**: 0.9\n\n**Risk**: An implementer following this literally registers CES/BES/WZS instead of HW_NVME_*, mismatching the documented example and breaking cross-project trait standardization os-traits enforces.\n\n**Priority**: Before merge\n**Why This Matters**: Placement traits are permanent once used. Non-standard short names would require a future cleanup migration and break scheduling interop with consumers expecting the HW_NVME_* namespace.\n\n**Recommendation**:\nShow fully-qualified values (TRAITS\u003d[\u0027HW_NVME_CES\u0027,\u0027HW_NVME_BES\u0027,\u0027HW_NVME_WZS\u0027]) or the constant-to-string mapping (CES\u003d\u0027HW_NVME_CES\u0027) os-traits modules use. Make the snippet consistent with the Placement example on line 411.","commit_id":"8909064b10ee6b305e0b90e1ff03348b4a91187c"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"ce8cf7166e056b875ba7a2e847e81264102e44f4","unresolved":false,"context_lines":[{"line_number":343,"context_line":"host-side zeroing via ``shred``, following Nova\u0027s ``volume_clear``"},{"line_number":344,"context_line":"pattern for LVM volumes."},{"line_number":345,"context_line":""},{"line_number":346,"context_line":"The resolved action is stored in ``std_board_info`` alongside existing"},{"line_number":347,"context_line":"device metadata. The selected cleanup action is fixed for the device"},{"line_number":348,"context_line":"until the next discovery cycle or agent restart. There is no runtime"},{"line_number":349,"context_line":"fallback. The cleanup action portion of ``std_board_info`` must not be"}],"source_content_type":"text/x-rst","patch_set":22,"id":"2977187b_63e9c3e5","line":346,"updated":"2026-07-01 08:07:24.000000000","message":"The cleanup action is locked in at discovery based on hardware capabilities (lines 346-354). If capabilities change between cycles (firmware update, transient nvme id-ctrl failure), a locked-in action could become unsatisfiable. The spec does not address this beyond moving to error state.\n\n**Severity**: SUGGESTION | **Confidence**: 0.9\n\n**Benefit**: Documenting capability-drift handling makes the error-state recovery path explicit and tells operators what to expect when capabilities change under a locked-in action.\n\n**Recommendation**:\nAdd a sentence: if the locked-in action references a capability the device no longer reports on the next discovery, the device is excluded or moved to error, and the error-state action-update allowance (line 351) is the remedy.","commit_id":"8909064b10ee6b305e0b90e1ff03348b4a91187c"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"ce8cf7166e056b875ba7a2e847e81264102e44f4","unresolved":false,"context_lines":[{"line_number":410,"context_line":"    Inventory:         total\u003d1"},{"line_number":411,"context_line":"    Traits:            OWNER_CYBORG, HW_NVME_CES, HW_NVME_BES, HW_NVME_WZS"},{"line_number":412,"context_line":""},{"line_number":413,"context_line":"During discovery, if the NVMe driver finds that a resource provider"},{"line_number":414,"context_line":"already exists for a PCI address but does not have the ``OWNER_CYBORG``"},{"line_number":415,"context_line":"trait, the driver logs an error and skips that device. This prevents"},{"line_number":416,"context_line":"conflicts when the same device is also managed by Nova."}],"source_content_type":"text/x-rst","patch_set":22,"id":"d511475a_bc0b241e","line":413,"updated":"2026-07-01 08:07:24.000000000","message":"The spec references Nova\u0027s \u003chostname\u003e_\u003cpci_address\u003e RP naming and OWNER_CYBORG to avoid conflicts (lines 413-420). It does not describe what happens if both Cyborg and Nova claim the same PCI address into RPs with the same name, causing a Placement RP-name collision.\n\n**Severity**: SUGGESTION | **Confidence**: 0.8\n\n**Benefit**: Documenting RP-name collision behavior makes the ownership-detection contract precise for operators running both Cyborg and Nova PCI management.\n\n**Recommendation**:\nAdd a note clarifying whether the OWNER_CYBORG check occurs before RP creation (to avoid collision) or whether Cyborg and Nova must never manage the same PCI address by configuration. State the operator-visible error on collision.","commit_id":"8909064b10ee6b305e0b90e1ff03348b4a91187c"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"ce8cf7166e056b875ba7a2e847e81264102e44f4","unresolved":false,"context_lines":[{"line_number":473,"context_line":"   failure"},{"line_number":474,"context_line":""},{"line_number":475,"context_line":"For zero-based cleanup:"},{"line_number":476,"context_line":""},{"line_number":477,"context_line":"#. Resolve the NVMe controller device from the PCI address"},{"line_number":478,"context_line":"#. If more than one namespace exists, delete all namespaces and"},{"line_number":479,"context_line":"   create a single namespace covering the full device (so the"}],"source_content_type":"text/x-rst","patch_set":22,"id":"99093f0a_d86e20c3","line":476,"updated":"2026-07-01 08:07:24.000000000","message":"The zero-path cleanup order (lines 476-484) deletes all namespaces and creates one namespace before zeroing. The timeout scope is unclear: whether namespace deletion is bounded by cleanup_timeout or counted separately, and whether a consolidation failure moves the device to error before zeroing.\n\n**Severity**: SUGGESTION | **Confidence**: 0.9\n\n**Benefit**: Clarifying timeout scope and failure semantics prevents double-counting the timeout or silently proceeding to zero a partially-consolidated device.\n\n**Recommendation**:\nState that namespace consolidation is part of the same cleanup_timeout window, and a failure at any sub-step (consolidation, zero-write, shred) moves the device to error without proceeding to the next sub-step.","commit_id":"8909064b10ee6b305e0b90e1ff03348b4a91187c"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"ce8cf7166e056b875ba7a2e847e81264102e44f4","unresolved":false,"context_lines":[{"line_number":505,"context_line":"    | available |-----------------\u003e| allocated |"},{"line_number":506,"context_line":"    +-----------+                  +-----------+"},{"line_number":507,"context_line":"          ^                              |"},{"line_number":508,"context_line":"          |                              | unbind /"},{"line_number":509,"context_line":"          | success                      | init_host: missed cleanup"},{"line_number":510,"context_line":"          |                              v"},{"line_number":511,"context_line":"    +-----------+   agent starts   +------------------+"}],"source_content_type":"text/x-rst","patch_set":22,"id":"48a66214_e509a7ee","line":508,"updated":"2026-07-01 08:07:24.000000000","message":"The diagram labels allocated-\u003epending_cleaning as \u0027unbind / init_host: missed cleanup\u0027, conflating two paths. The conductor dispatches a cleanup_device RPC on unbind (lines 526-533). init_host reconciles allocated devices with no active ARQ (lines 543-546). Distinct triggers in one edge label.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: An implementer reading the diagram alone may not realize the conductor path and init_host reconciliation are separate code paths with different preconditions (ARQ presence), risking a misplaced guard.\n\n**Suggestion**:\nSplit the label or add a footnote clarifying two triggers: (1) conductor cleanup_device RPC on unbind, (2) init_host reconciliation for allocated devices with no active ARQ.","commit_id":"8909064b10ee6b305e0b90e1ff03348b4a91187c"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"ce8cf7166e056b875ba7a2e847e81264102e44f4","unresolved":false,"context_lines":[{"line_number":707,"context_line":"microversion and later, device responses include the ``device_state``"},{"line_number":708,"context_line":"field."},{"line_number":709,"context_line":""},{"line_number":710,"context_line":"**Example GET /v2/devices/{uuid} response (new microversion):**"},{"line_number":711,"context_line":""},{"line_number":712,"context_line":"Device in available state after successful cleanup::"},{"line_number":713,"context_line":""}],"source_content_type":"text/x-rst","patch_set":22,"id":"1556d8aa_d10eaa0d","line":710,"updated":"2026-07-01 08:07:24.000000000","message":"REST API section gives JSON examples but no formal schemas with additionalProperties: false. The template (lines 162-182) requires restrictive JSON schemas for request and response bodies. POST /clean also lacks any statement of its request body schema.\n\n**Severity**: HIGH | **Confidence**: 0.9\n\n**Risk**: Without a restrictive schema the device_state field type, enum values, and nullability are ambiguous. Since API changes are supported forever, underspecified schemas risk divergent implementations across Cyborg API, SDK, and client.\n\n**Priority**: Before merge\n**Why This Matters**: This spec adds a new microversion and endpoint. The template REST API section is the most scrutinized part of a spec; missing schemas mean the contract is defined only by examples.\n\n**Recommendation**:\nAdd a formal JSON schema for the device response (type\u003dobject, properties with types, device_state as enum of the five states, additionalProperties: false). State POST /clean accepts no request body. Define the POST /clean response.","commit_id":"8909064b10ee6b305e0b90e1ff03348b4a91187c"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"ce8cf7166e056b875ba7a2e847e81264102e44f4","unresolved":false,"context_lines":[{"line_number":753,"context_line":""},{"line_number":754,"context_line":"Normal response code: ``202 Accepted``"},{"line_number":755,"context_line":""},{"line_number":756,"context_line":"Error response codes:"},{"line_number":757,"context_line":""},{"line_number":758,"context_line":"* ``400 Bad Request`` — device does not support cleaning"},{"line_number":759,"context_line":"* ``404 Not Found`` — device UUID does not exist"}],"source_content_type":"text/x-rst","patch_set":22,"id":"d3156b79_c60e5c0d","line":756,"updated":"2026-07-01 08:07:24.000000000","message":"POST /clean is governed by cyborg:device:clean defaulting to role:admin (line 774), and error codes list 400/404/409 but omit 403 Forbidden. A non-admin caller receives 403 before any state validation, so it should be documented per the template requirement to describe each error code.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: Operators and SDK/client implementers building against this endpoint would not know from the spec that a 403 is possible.\n\n**Suggestion**:\nAdd \u0027403 Forbidden - caller lacks the cyborg:device:clean (role:admin) policy\u0027 to the error response codes for POST /clean.","commit_id":"8909064b10ee6b305e0b90e1ff03348b4a91187c"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"ce8cf7166e056b875ba7a2e847e81264102e44f4","unresolved":false,"context_lines":[{"line_number":793,"context_line":"The conductor guards the RPC dispatch with two checks, following"},{"line_number":794,"context_line":"the pattern in Nova\u0027s ``scheduler/rpcapi.py``::"},{"line_number":795,"context_line":""},{"line_number":796,"context_line":"    if not device.supports_cleaning:"},{"line_number":797,"context_line":"        # non-NVMe -\u003e available"},{"line_number":798,"context_line":"        device.device_state \u003d \u0027available\u0027"},{"line_number":799,"context_line":"        device.save()"}],"source_content_type":"text/x-rst","patch_set":22,"id":"8d33f901_77d8ff7d","line":796,"updated":"2026-07-01 08:07:24.000000000","message":"The cleanup_device RPC passes a Device object (line 786); the conductor guards with can_send_version(\u00271.1\u0027) (line 801). The spec omits that the conductor must serialize the Device at object version 1.2 to an old (2026.1) agent. obj_make_compatible pops device_state; version-targeting is undescribed.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: If a Device 1.3 object reaches a 2026.1 agent that only knows 1.2, deserialization could fail. The N-1 compatibility claim is a permanent contract and must be unambiguous.\n\n**Suggestion**:\nState that the conductor serializes the Device at the max version the target agent understands (1.2 for 2026.1 agents). can_send_version and object downgrade together guarantee an old agent never sees device_state.","commit_id":"8909064b10ee6b305e0b90e1ff03348b4a91187c"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":285,"context_line":"The traits are defined in a new ``os_traits/hw/nvme/__init__.py``"},{"line_number":286,"context_line":"module::"},{"line_number":287,"context_line":""},{"line_number":288,"context_line":"    TRAITS \u003d ["},{"line_number":289,"context_line":"        \u0027CES\u0027,   # Crypto Erase Sanitize Supported (sanicap bit 0)"},{"line_number":290,"context_line":"        \u0027BES\u0027,   # Block Erase Sanitize Supported (sanicap bit 1)"},{"line_number":291,"context_line":"        \u0027WZS\u0027,   # Write Zeroes Supported (ONCS bit 3)"}],"source_content_type":"text/x-rst","patch_set":23,"id":"6fbd2098_4b6961b2","line":288,"updated":"2026-07-01 08:33:42.000000000","message":"Capability trait naming is internally inconsistent. The TRAITS list and os-traits patch define bare \u0027CES\u0027,\u0027BES\u0027,\u0027WZS\u0027, but the Placement example registers \u0027HW_NVME_CES\u0027,\u0027HW_NVME_BES\u0027,\u0027HW_NVME_WZS\u0027. The bare form is not a valid Placement standard trait name.\n\n**Severity**: HIGH | **Confidence**: 0.9\n\n**Risk**: If os-traits lands the bare names and Cyborg reports HW_NVME_*, traits never match and capability-based verification silently breaks. Mismatched companion patches cause cross-project confusion.\n\n**Priority**: Before merge\n**Why This Matters**: os-traits standard names are namespaced (e.g. HW_NIC_SRIOV, matching the cited os_traits/hw/nic module and OWNER_CYBORG). A spec must fix the exact trait strings because Cyborg and os-traits are separate projects that must agree, and trait strings are immutable once shipped.\n\n**Recommendation**:\nUse the fully-qualified HW_NVME_CES/BES/WZS form everywhere (matches line 411, the os-traits hw namespace, OWNER_CYBORG). Update the TRAITS list (lines 288-292), Work Items (line 983) and Dependencies (lines 1025-1026) to state the canonical prefixed names explicitly.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"cccfbd3a0270cec494709953de9b5bcf06843730","unresolved":true,"context_lines":[{"line_number":317,"context_line":"The policy matrix is::"},{"line_number":318,"context_line":""},{"line_number":319,"context_line":"    +----------------+--------------+-----------------------------------------+"},{"line_number":320,"context_line":"    | clear_strategy | clear_action | selected cleanup action                 |"},{"line_number":321,"context_line":"    +\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+"},{"line_number":322,"context_line":"    | auto           | auto         | CES sanitize, else BES sanitize, else   |"},{"line_number":323,"context_line":"    |                |              | write zeroes, else shred                |"}],"source_content_type":"text/x-rst","patch_set":23,"id":"d97a18fd_aec56580","line":320,"range":{"start_line":320,"start_character":5,"end_line":320,"end_character":35},"updated":"2026-07-01 09:15:11.000000000","message":"these are backword so me.\n\nsantaize or zero are the action aka clear_method\n\ncyrpto and block are the stragy aka clear_mode  to implment the action\n\nthe way to think about this is method and actoin are a verb sanitize,zero\nmode and strategy take nouns block,crypto\n\nthat was the logic behind the orgial clear_method and clear_mode split\n\nthe \"what to do\" is the method/action the \"how to do it\" is that mode/strategy\n\ni dont really ind if we use mode/method vs stragey/action as long as we done break that logic\n\nthis is the reason for the -1\n\nonce this is fixed im +2","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"feb684ade0a8457ea063ea6603a60a30db01ba4f","unresolved":false,"context_lines":[{"line_number":317,"context_line":"The policy matrix is::"},{"line_number":318,"context_line":""},{"line_number":319,"context_line":"    +----------------+--------------+-----------------------------------------+"},{"line_number":320,"context_line":"    | clear_strategy | clear_action | selected cleanup action                 |"},{"line_number":321,"context_line":"    +\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+"},{"line_number":322,"context_line":"    | auto           | auto         | CES sanitize, else BES sanitize, else   |"},{"line_number":323,"context_line":"    |                |              | write zeroes, else shred                |"}],"source_content_type":"text/x-rst","patch_set":23,"id":"b34d61ef_167235cc","line":320,"range":{"start_line":320,"start_character":5,"end_line":320,"end_character":35},"in_reply_to":"59be9be2_d05e0b99","updated":"2026-07-02 11:08:44.000000000","message":"we may want to consider including that explaintionin the docs somewher but we can explain that in the implemetion","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":4690,"name":"melanie witt","display_name":"melwitt","email":"melwittt@gmail.com","username":"melwitt"},"change_message_id":"4f8fb4cb3e872599868ea52bfc612d6328c196ca","unresolved":true,"context_lines":[{"line_number":317,"context_line":"The policy matrix is::"},{"line_number":318,"context_line":""},{"line_number":319,"context_line":"    +----------------+--------------+-----------------------------------------+"},{"line_number":320,"context_line":"    | clear_strategy | clear_action | selected cleanup action                 |"},{"line_number":321,"context_line":"    +\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+"},{"line_number":322,"context_line":"    | auto           | auto         | CES sanitize, else BES sanitize, else   |"},{"line_number":323,"context_line":"    |                |              | write zeroes, else shred                |"}],"source_content_type":"text/x-rst","patch_set":23,"id":"59be9be2_d05e0b99","line":320,"range":{"start_line":320,"start_character":5,"end_line":320,"end_character":35},"in_reply_to":"b64f6be6_80e69bdf","updated":"2026-07-01 18:41:31.000000000","message":"This actually helped my understanding more than expected 🙂\n\nI didn\u0027t realize which set of choices was the \"approach\" vs the \"action\" and this clears that up for me. And the new names make more sense, thank you.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"1a1c3ba2dc2640129f2de76f8065e66408d45902","unresolved":true,"context_lines":[{"line_number":317,"context_line":"The policy matrix is::"},{"line_number":318,"context_line":""},{"line_number":319,"context_line":"    +----------------+--------------+-----------------------------------------+"},{"line_number":320,"context_line":"    | clear_strategy | clear_action | selected cleanup action                 |"},{"line_number":321,"context_line":"    +\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+"},{"line_number":322,"context_line":"    | auto           | auto         | CES sanitize, else BES sanitize, else   |"},{"line_number":323,"context_line":"    |                |              | write zeroes, else shred                |"}],"source_content_type":"text/x-rst","patch_set":23,"id":"b64f6be6_80e69bdf","line":320,"range":{"start_line":320,"start_character":5,"end_line":320,"end_character":35},"in_reply_to":"d97a18fd_aec56580","updated":"2026-07-01 11:42:18.000000000","message":"Done. \n\nI have swapped the semantics based on above suggestion.\n\nclear_action now selects the operation (auto|sanitize|zero) and clear_strategy selects the erase approach (auto|crypto|block). \n\nAdded one line explainer also for the same.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":344,"context_line":"pattern for LVM volumes."},{"line_number":345,"context_line":""},{"line_number":346,"context_line":"The resolved action is stored in ``std_board_info`` alongside existing"},{"line_number":347,"context_line":"device metadata. The selected cleanup action is fixed for the device"},{"line_number":348,"context_line":"until the next discovery cycle or agent restart. There is no runtime"},{"line_number":349,"context_line":"fallback. The cleanup action portion of ``std_board_info`` must not be"},{"line_number":350,"context_line":"overwritten by ``discover()`` while ``device_state`` is not"}],"source_content_type":"text/x-rst","patch_set":23,"id":"40572816_85589adf","line":347,"updated":"2026-07-01 08:33:42.000000000","message":"std_board_info is a free-form JSON string and the spec stores the locked-in cleanup action there (lines 347-349, 687-691). Parsing a JSON blob to recover the action at retry time is fragile and unqueryable, while the spec elsewhere favours a structured capability model.\n\n**Severity**: SUGGESTION | **Confidence**: 0.7\n\n**Benefit**: A typed column (or the deferred device_metadata capability model) would make the locked-in action first-class, operator-queryable, and resilient to std_board_info shape changes across discover() runs.\n\n**Recommendation**:\nModel the locked-in action as a typed attribute/column rather than a sub-field of the free-form std_board_info JSON, or justify the embedded choice for the retry path. At minimum specify the exact JSON key and schema for the embedded action.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":397,"context_line":"The NVMe driver\u0027s ``discover()`` method builds ``DriverDevice`` objects"},{"line_number":398,"context_line":"with ``type\u003dNVME`` and reports NVMe capability traits alongside"},{"line_number":399,"context_line":"``OWNER_CYBORG``. The ``std_board_info`` field contains the"},{"line_number":400,"context_line":"``product_id`` and ``pci_address`` of the device\u0027s Physical Function"},{"line_number":401,"context_line":"(PF). Resource provider and deployable names use the format"},{"line_number":402,"context_line":"``\u003chostname\u003e_\u003cpci_address\u003e``."},{"line_number":403,"context_line":""}],"source_content_type":"text/x-rst","patch_set":23,"id":"5a83f307_ba5b20eb","line":400,"updated":"2026-07-01 08:33:42.000000000","message":"Resource-provider and deployable names are derived from \u003chostname\u003e_\u003cpci_address\u003e. PCI addresses change across BIOS/firmware updates or when a device is reseated/moved; hostnames change on rename. The spec does not describe RP/deployable rename or reconciliation on address change.\n\n**Severity**: WARNING | **Confidence**: 0.6\n\n**Impact**: A renamed/moved device creates a new RP while the old one lingers with stale inventory/traits, double-counting capacity in Placement and orphaning device_state rows. Nova uses the same scheme but documents the limitation; this spec does not.\n\n**Suggestion**:\nNote that RP identity follows Nova\u0027s PCI placement translator semantics and document the expected behaviour and operator action when a PCI address changes between discoveries, including cleanup of the stale RP and its device_state.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":445,"context_line":"commands run under privsep with per-device locking as described in the"},{"line_number":446,"context_line":"Security impact section."},{"line_number":447,"context_line":""},{"line_number":448,"context_line":"The agent executes exactly the cleanup action that was locked in during"},{"line_number":449,"context_line":"discovery (see Discovery and Cleanup Resolution section above). There"},{"line_number":450,"context_line":"is no runtime fallback. If the action fails, the device moves to"},{"line_number":451,"context_line":"``error`` state. Operators can re-trigger cleanup via"}],"source_content_type":"text/x-rst","patch_set":23,"id":"54f93504_48a3efa1","line":448,"updated":"2026-07-01 08:33:42.000000000","message":"The no-runtime-fallback stance means any single transient nvme-cli failure (EBUSY, controller-reset race, temporary /dev/nvmeN absence during libvirt rebind) permanently parks the device in \u0027error\u0027 requiring manual intervention, with no bounded automatic retry before escalating.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: The security rationale (never degrade to a weaker erase) is sound for fallback across erase STRENGTHS, but is also applied to retries of the SAME action. This conflates two policies and can cause fleet-wide error accumulation on transient blips.\n\n**Suggestion**:\nSeparate the policies: no fallback to a WEAKER erase stays absolute, but allow a small bounded retry of the locked-in action on transient errors before error state. State this distinction in the Cleanup Policy section.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":491,"context_line":""},{"line_number":492,"context_line":"The device lifecycle is tracked using a new ``device_state`` field"},{"line_number":493,"context_line":"added to the ``Device`` versioned object (``cyborg/objects/device.py``)"},{"line_number":494,"context_line":"and the ``devices`` database table. ``device_state`` is added to all"},{"line_number":495,"context_line":"devices regardless of type; every device follows the same state"},{"line_number":496,"context_line":"transitions with no type-specific checks. The bind guard and"},{"line_number":497,"context_line":"Placement ``reserved`` updates also apply uniformly. This is"}],"source_content_type":"text/x-rst","patch_set":23,"id":"c3b9e540_3dd1cf2c","line":494,"updated":"2026-07-01 08:33:42.000000000","message":"device_state and its bind guard are applied to ALL device types, yet only NVMe is cleaned. For non-NVMe types the conductor just sets device_state\u003d\u0027available\u0027 on unbind, silently adding a new lifecycle contract to GPU/FPGA/QAT/NIC/SSD devices that never had one.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: The new mandatory bind guard (reject unless available) plus reserved updates apply to every type. A device failing online backfill (NULL state) or a driver not wired into the path gets rejected at bind, breaking GPU/FPGA attach in existing deployments.\n\n**Priority**: Before merge\n**Why This Matters**: The template\u0027s Developer impact section requires discussion when a shared path changes. A mandatory bind guard across all device types is broad; its no-op behaviour for existing drivers must be proven, not assumed.\n\n**Recommendation**:\nEither scope device_state and the bind guard to NVMe (or types where supports_cleaning is true), or add a subsection proving the change is behaviour-preserving for every existing driver, including NULL state and interaction with the \u0027status\u0027 enabled/disabled field.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"cccfbd3a0270cec494709953de9b5bcf06843730","unresolved":false,"context_lines":[{"line_number":491,"context_line":""},{"line_number":492,"context_line":"The device lifecycle is tracked using a new ``device_state`` field"},{"line_number":493,"context_line":"added to the ``Device`` versioned object (``cyborg/objects/device.py``)"},{"line_number":494,"context_line":"and the ``devices`` database table. ``device_state`` is added to all"},{"line_number":495,"context_line":"devices regardless of type; every device follows the same state"},{"line_number":496,"context_line":"transitions with no type-specific checks. The bind guard and"},{"line_number":497,"context_line":"Placement ``reserved`` updates also apply uniformly. This is"}],"source_content_type":"text/x-rst","patch_set":23,"id":"4c5c7025_0406becd","line":494,"in_reply_to":"c3b9e540_3dd1cf2c","updated":"2026-07-01 09:15:11.000000000","message":"so this is valid to question but its entirely intentional.\ni think this aspect of the spec is verbose enough already os we can ignore htis","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":540,"context_line":"``POST /v2/devices/{uuid}/clean``, which transitions the device back"},{"line_number":541,"context_line":"to ``pending_cleaning``."},{"line_number":542,"context_line":""},{"line_number":543,"context_line":"On agent restart, ``init_host()`` reconciles device states. A device"},{"line_number":544,"context_line":"in ``allocated`` with no active ARQ indicates a cleanup RPC was missed"},{"line_number":545,"context_line":"while the agent was down; ``init_host()`` moves it to"},{"line_number":546,"context_line":"``pending_cleaning`` and triggers cleanup automatically. A device"}],"source_content_type":"text/x-rst","patch_set":23,"id":"d24c68e6_4098dd8e","line":543,"updated":"2026-07-01 08:33:42.000000000","message":"On restart, init_host() treats any device \u0027allocated\u0027 with no active ARQ as a missed cleanup RPC and cleans it automatically (lines 545-547, 596-599). That assumes every such device was dirtied, but the condition can also arise when an instance was deleted via another unbind branch.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: An unnecessary sanitize/write-zeroes on a device that was allocated but never written wastes device endurance (sanitize counts against media lifetime) and delays return to available by up to the cleanup timeout.\n\n**Suggestion**:\nBe conservative: on allocated+no-ARQ, clean only if the device is known to have run a guest; otherwise transition to available after a cheap sanitize-status check. Document the endurance trade-off of always-cleaning.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":589,"context_line":"Crash recovery is handled by the agent during ``init_host()``"},{"line_number":590,"context_line":"(``cyborg/agent/manager.py``), which runs after the RPC server starts"},{"line_number":591,"context_line":"but before the first ``discover()`` call. The agent queries the"},{"line_number":592,"context_line":"database for devices in ``cleaning`` or ``pending_cleaning`` state on"},{"line_number":593,"context_line":"its host. For each such device, the agent moves it to ``error`` state"},{"line_number":594,"context_line":"so the operator can investigate and re-trigger cleanup via"},{"line_number":595,"context_line":"``POST /v2/devices/{uuid}/clean``. The agent also checks for devices"}],"source_content_type":"text/x-rst","patch_set":23,"id":"cd80942c_71e5fc03","line":592,"updated":"2026-07-01 08:33:42.000000000","message":"Crash recovery moves any device in pending_cleaning/cleaning to error unconditionally (lines 547-549, 593-595). NVMe sanitize is a controller op that survives host restarts; an in-progress sanitize interrupted by an agent restart is declared failed though the firmware job may still complete.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: Operators get spurious error states and must manually re-trigger cleanup for devices that were sanitizing fine. This adds toil and can mask genuinely stuck devices. The spec itself notes sanitize-status polling exists (lines 467-469) but does not use it in recovery.\n\n**Suggestion**:\nIn init_host recovery for a device last seen in \u0027cleaning\u0027, query nvme sanitize-log/status before declaring error; if a sanitize is still running, resume polling to completion/timeout, and only move to error on firmware failure or no job present.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":650,"context_line":"checks whether the device has bound ARQs: devices with bound ARQs"},{"line_number":651,"context_line":"are set to ``allocated``; devices without are set to ``available``."},{"line_number":652,"context_line":""},{"line_number":653,"context_line":"The migration is callable from three entry points:"},{"line_number":654,"context_line":""},{"line_number":655,"context_line":"* ``cyborg-manage db online_data_migrations``"},{"line_number":656,"context_line":"  (``cyborg/cmd/dbsync.py``)"}],"source_content_type":"text/x-rst","patch_set":23,"id":"f45de9fb_2620e069","line":653,"updated":"2026-07-01 08:33:42.000000000","message":"The online data migration backfilling device_state runs from three entry points, two of which are service startup (conductor/agent init_host). Service-startup data migrations are a known fragile OpenStack pattern and can race with the new bind guard.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: If a service starts and serves requests before backfill finishes, devices may still be NULL. The spec does not say how the bind guard treats a NULL device_state, so a freshly-upgraded cluster could reject all binds until migration completes.\n\n**Suggestion**:\nSpecify the bind-guard behaviour for NULL state (e.g. treat NULL as available to preserve pre-migration behaviour). Confirm the cyborg-status pre-upgrade check is mandatory before services start, and that startup migrations are best-effort, not the primary backfill.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"cccfbd3a0270cec494709953de9b5bcf06843730","unresolved":true,"context_lines":[{"line_number":688,"context_line":"to dispatch the ``cleanup_device`` RPC or transition the device"},{"line_number":689,"context_line":"directly to ``available`` on unbind. A driver-independent capability"},{"line_number":690,"context_line":"model is out of scope (see Scope section). The existing ``status``"},{"line_number":691,"context_line":"field remains unchanged."},{"line_number":692,"context_line":""},{"line_number":693,"context_line":""},{"line_number":694,"context_line":"REST API impact"}],"source_content_type":"text/x-rst","patch_set":23,"id":"9c598264_7b5ad19d","line":691,"updated":"2026-07-01 09:15:11.000000000","message":"+1\n\nby using a property rather then an explict filed for now we are free to replace the implemtion in the future without forcing a ovo version bump.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"1a1c3ba2dc2640129f2de76f8065e66408d45902","unresolved":false,"context_lines":[{"line_number":688,"context_line":"to dispatch the ``cleanup_device`` RPC or transition the device"},{"line_number":689,"context_line":"directly to ``available`` on unbind. A driver-independent capability"},{"line_number":690,"context_line":"model is out of scope (see Scope section). The existing ``status``"},{"line_number":691,"context_line":"field remains unchanged."},{"line_number":692,"context_line":""},{"line_number":693,"context_line":""},{"line_number":694,"context_line":"REST API impact"}],"source_content_type":"text/x-rst","patch_set":23,"id":"c617c6da_29be5558","line":691,"in_reply_to":"9c598264_7b5ad19d","updated":"2026-07-01 11:42:18.000000000","message":"Acknowledged","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":694,"context_line":"REST API impact"},{"line_number":695,"context_line":"---------------"},{"line_number":696,"context_line":""},{"line_number":697,"context_line":"A new microversion is required. The exact version number depends on"},{"line_number":698,"context_line":"merge ordering with other in-flight specs; this spec claims the next"},{"line_number":699,"context_line":"available microversion after its dependencies land. The microversion"},{"line_number":700,"context_line":"is driven by the new ``device_state`` field in device responses and the"}],"source_content_type":"text/x-rst","patch_set":23,"id":"1045501b_6d6df5dd","line":697,"updated":"2026-07-01 08:33:42.000000000","message":"The spec claims \u0027the next available microversion\u0027 (lines 697-699) and agent RPC \u00271.1\u0027 (line 781) without pinning them, deferring to merge ordering. It should at least state the BASELINE microversion/RPC version this is built on.\n\n**Severity**: SUGGESTION | **Confidence**: 0.7\n\n**Benefit**: Anchoring the baseline (e.g. \u0027after 2.xx\u0027 / \u0027agent RPC 1.0\u0027) removes ambiguity for the implementer and downstream SDK/client version logic, and makes the N-1 compatibility claim in Upgrade verifiable.\n\n**Recommendation**:\nState the current device API microversion and agent RPC version this builds on, and note the final claimed number is assigned at merge. Mirror the clarity already applied to the Device object version (1.2 -\u003e 1.3).","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":714,"context_line":"  {"},{"line_number":715,"context_line":"    \"uuid\": \"7c8a5f3b-2d4e-4a9c-b1e7-9f8d3c2a1b0e\","},{"line_number":716,"context_line":"    \"type\": \"NVME\","},{"line_number":717,"context_line":"    \"vendor\": \"8086\","},{"line_number":718,"context_line":"    \"model\": \"0001\","},{"line_number":719,"context_line":"    \"std_board_info\": \"{\\\"product_id\\\": \\\"0001\\\","},{"line_number":720,"context_line":"        \\\"pci_address\\\": \\\"0000:01:00.0\\\"}\","}],"source_content_type":"text/x-rst","patch_set":23,"id":"3465a8a5_df372d43","line":717,"updated":"2026-07-01 08:33:42.000000000","message":"API examples render std_board_info as a JSON string wrapping JSON with non-standard indentation (lines 719-720, 736-737), and the device \u0027model\u0027 field carries product_id while \u0027vendor\u0027 carries vendor_id. The semantics of these field names are not stated.\n\n**Severity**: SUGGESTION | **Confidence**: 0.6\n\n**Benefit**: Precise field semantics and valid compact JSON examples prevent implementers and client authors from misreading vendor/model as free-form text versus the PCI vendor_id/product_id they encode, and keep the doc8/RST build clean.\n\n**Recommendation**:\nClarify that \u0027vendor\u0027/\u0027model\u0027 carry the PCI vendor_id/product_id, render the std_board_info example as valid single-line JSON, and add a one-line note mapping the API fields to the device_spec keys.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"cccfbd3a0270cec494709953de9b5bcf06843730","unresolved":false,"context_lines":[{"line_number":714,"context_line":"  {"},{"line_number":715,"context_line":"    \"uuid\": \"7c8a5f3b-2d4e-4a9c-b1e7-9f8d3c2a1b0e\","},{"line_number":716,"context_line":"    \"type\": \"NVME\","},{"line_number":717,"context_line":"    \"vendor\": \"8086\","},{"line_number":718,"context_line":"    \"model\": \"0001\","},{"line_number":719,"context_line":"    \"std_board_info\": \"{\\\"product_id\\\": \\\"0001\\\","},{"line_number":720,"context_line":"        \\\"pci_address\\\": \\\"0000:01:00.0\\\"}\","}],"source_content_type":"text/x-rst","patch_set":23,"id":"a343a6cd_0898639f","line":717,"in_reply_to":"3465a8a5_df372d43","updated":"2026-07-01 09:15:11.000000000","message":"this is something that i think we will adress in the api clean up spec next cycle","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":743,"context_line":"    \"updated_at\": \"2026-05-15T16:30:15Z\""},{"line_number":744,"context_line":"  }"},{"line_number":745,"context_line":""},{"line_number":746,"context_line":"**POST /v2/devices/{uuid}/clean (new endpoint, new microversion):**"},{"line_number":747,"context_line":""},{"line_number":748,"context_line":"This is an admin-only endpoint that triggers device cleanup. It"},{"line_number":749,"context_line":"dispatches the ``cleanup_device`` RPC to the agent, which transitions"}],"source_content_type":"text/x-rst","patch_set":23,"id":"47fccd7d_62a9e1c5","line":746,"updated":"2026-07-01 08:33:42.000000000","message":"The /clean 409 cases (available / allocated / already-cleaning, lines 761-765) and the 400 for unsupported devices omit the case where the device is NVMe but in a transitional agent-side state the conductor cannot observe. The endpoint\u0027s idempotency and concurrency contract is under-specified.\n\n**Severity**: WARNING | **Confidence**: 0.6\n\n**Impact**: Because cleanup is async and state lives on the agent, the API may observe \u0027error\u0027 while the agent has already begun re-cleaning on restart, yielding surprising 409/202 interleavings. CLI/SDK behaviour is undefined without a documented contract.\n\n**Suggestion**:\nDocument the endpoint\u0027s idempotency and concurrency guarantees: behaviour on concurrent POST /clean calls, whether a re-trigger mid-cleanup is a no-op 202 or a 409, and how the conductor-vs-agent state split is reconciled for the API response.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2d2cc7a352b667fada77a037e9d5a05614ac936f","unresolved":false,"context_lines":[{"line_number":755,"context_line":""},{"line_number":756,"context_line":"Error response codes:"},{"line_number":757,"context_line":""},{"line_number":758,"context_line":"* ``400 Bad Request`` — device does not support cleaning"},{"line_number":759,"context_line":"* ``403 Forbidden`` — caller lacks the ``cyborg:device:clean`` policy"},{"line_number":760,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":761,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"}],"source_content_type":"text/x-rst","patch_set":23,"id":"a47a898c_ca23a30e","line":758,"updated":"2026-07-01 08:33:42.000000000","message":"POST /v2/devices/{uuid}/clean overloads the 400 response. The error list gives \u0027400 - device does not support cleaning\u0027 (line 758), yet line 952 also returns 400 for a device whose agent predates the 1.1 RPC. Two distinct conditions share one code with no documented discriminator.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: Operators and SDK/CLI authors cannot distinguish a fundamentally unsupported device from temporary agent-version skew. Both surface identically, so automated remediation and runbooks are ambiguous. API behaviour is permanent once shipped.\n\n**Priority**: Before merge\n**Why This Matters**: The template requires each error code to describe its semantics. Conflating \u0027never cleanable\u0027 with \u0027agent too old\u0027 in one code violates that and degrades operability; codes cannot be renumbered later.\n\n**Recommendation**:\nUse a distinct code for agent-version skew: 409 Conflict or 501 Not Implemented with a clear message is more accurate than 400 for a device that IS cleanable but whose agent cannot yet act. Document both conditions explicitly in the error list.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"1a1c3ba2dc2640129f2de76f8065e66408d45902","unresolved":false,"context_lines":[{"line_number":755,"context_line":""},{"line_number":756,"context_line":"Error response codes:"},{"line_number":757,"context_line":""},{"line_number":758,"context_line":"* ``400 Bad Request`` — device does not support cleaning"},{"line_number":759,"context_line":"* ``403 Forbidden`` — caller lacks the ``cyborg:device:clean`` policy"},{"line_number":760,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":761,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"}],"source_content_type":"text/x-rst","patch_set":23,"id":"2cb15999_2a1bdfc0","line":758,"in_reply_to":"1c0b8a59_c4a57b7b","updated":"2026-07-01 11:42:18.000000000","message":"Done. \n\nAdded 501 Not Implemented for the agent-too-old case.","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"cccfbd3a0270cec494709953de9b5bcf06843730","unresolved":true,"context_lines":[{"line_number":755,"context_line":""},{"line_number":756,"context_line":"Error response codes:"},{"line_number":757,"context_line":""},{"line_number":758,"context_line":"* ``400 Bad Request`` — device does not support cleaning"},{"line_number":759,"context_line":"* ``403 Forbidden`` — caller lacks the ``cyborg:device:clean`` policy"},{"line_number":760,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":761,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"}],"source_content_type":"text/x-rst","patch_set":23,"id":"1c0b8a59_c4a57b7b","line":758,"in_reply_to":"a47a898c_ca23a30e","updated":"2026-07-01 09:15:11.000000000","message":"i think we want to conflat these.\n\nwe do not want to use 409 because thisis not a conflict due to the state of the resouce in the rpc case.\n\n501 Not Implemented \n\nhttps://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Status/501\n\ncalling clean on a non cleanbale deive is a client error\n\ncalling clean on a deivfce that shoudl eb cleanabel but isnt because the agent is too old is operator error so 501 might be mroe correct but this should not happen unless you downgrade the agetn or somethign liek that.\n\nwe do not support downgrades so  my incliation si to just stick with a 400 and we can comunciate if if its an RPC verion issue vs a device supprot issu in the error message.\n\nso good to think about but in this case no change required","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"cccfbd3a0270cec494709953de9b5bcf06843730","unresolved":true,"context_lines":[{"line_number":807,"context_line":"    cctxt.cast(context, \u0027cleanup_device\u0027, device\u003ddevice)"},{"line_number":808,"context_line":""},{"line_number":809,"context_line":"Non-NVMe devices transition directly to ``available`` without an"},{"line_number":810,"context_line":"RPC. If the device is NVMe but the target agent does not support"},{"line_number":811,"context_line":"version ``1.1`` (a 2026.1 agent), the conductor sets"},{"line_number":812,"context_line":"``device_state`` to ``error`` because an NVMe device on an agent"},{"line_number":813,"context_line":"that cannot clean requires operator attention."},{"line_number":814,"context_line":""},{"line_number":815,"context_line":"When the agent does support the RPC, the call returns immediately"},{"line_number":816,"context_line":"so Nova\u0027s instance deletion completes without waiting for cleanup."},{"line_number":817,"context_line":"The agent manages all ``device_state`` transitions during cleanup"}],"source_content_type":"text/x-rst","patch_set":23,"id":"00f02593_f333884d","line":814,"range":{"start_line":810,"start_character":4,"end_line":814,"end_character":1},"updated":"2026-07-01 09:15:11.000000000","message":"so this should never happen but ya if it does that would be the correct approch\n\nour api contract is we will alway clean the device so if we should be abel to and cant we need to set it to error","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"1a1c3ba2dc2640129f2de76f8065e66408d45902","unresolved":false,"context_lines":[{"line_number":807,"context_line":"    cctxt.cast(context, \u0027cleanup_device\u0027, device\u003ddevice)"},{"line_number":808,"context_line":""},{"line_number":809,"context_line":"Non-NVMe devices transition directly to ``available`` without an"},{"line_number":810,"context_line":"RPC. If the device is NVMe but the target agent does not support"},{"line_number":811,"context_line":"version ``1.1`` (a 2026.1 agent), the conductor sets"},{"line_number":812,"context_line":"``device_state`` to ``error`` because an NVMe device on an agent"},{"line_number":813,"context_line":"that cannot clean requires operator attention."},{"line_number":814,"context_line":""},{"line_number":815,"context_line":"When the agent does support the RPC, the call returns immediately"},{"line_number":816,"context_line":"so Nova\u0027s instance deletion completes without waiting for cleanup."},{"line_number":817,"context_line":"The agent manages all ``device_state`` transitions during cleanup"}],"source_content_type":"text/x-rst","patch_set":23,"id":"0722c4fd_a9d8e47a","line":814,"range":{"start_line":810,"start_character":4,"end_line":814,"end_character":1},"in_reply_to":"00f02593_f333884d","updated":"2026-07-01 11:42:18.000000000","message":"Acknowledged","commit_id":"b3ad25d4c749c4e41f988dca422c15338bdd461d"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":1,"context_line":".."},{"line_number":2,"context_line":" This work is licensed under a Creative Commons Attribution 3.0 Unported"},{"line_number":3,"context_line":" License."},{"line_number":4,"context_line":""}],"source_content_type":"text/x-rst","patch_set":24,"id":"d70b145b_ad8c5fe4","line":1,"updated":"2026-07-01 11:58:24.000000000","message":"Commit message uses \u0027blueprint: generic-nvme-driver-with-secure-cleanup\u0027 instead of the canonical OpenStack gerrit convention \u0027Implements: blueprint \u003cname\u003e\u0027. The sibling specs in the same release use \u0027Implements: blueprint\u0027 or \u0027Blueprint:\u0027 formats that gerrit\u0027s launchpad integration recognizes.\n\n**Severity**: HIGH | **Confidence**: 0.9\n\n**Risk**: The non-standard \u0027blueprint:\u0027 footer may not be recognized by Gerrit\u0027s Launchpad integration bot, leaving the blueprint status un-updated and breaking the automated spec-to-blueprint traceability that OpenStack governance relies on.\n\n**Priority**: Before merge\n**Why This Matters**: OpenStack\u0027s CI pipeline uses \u0027Implements: blueprint \u003cname\u003e\u0027 to automatically link commits to Launchpad blueprints. Using an unrecognized keyword breaks this linkage, making the spec\u0027s implementation progress untrackable.\n\n**Recommendation**:\nChange the commit message footer from \u0027blueprint: generic-nvme-driver-with-secure-cleanup\u0027 to \u0027Implements: blueprint generic-nvme-driver-with-secure-cleanup\u0027 to match the convention used by the VFIO variant spec and standard OpenStack practice.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":1,"context_line":".."},{"line_number":2,"context_line":" This work is licensed under a Creative Commons Attribution 3.0 Unported"},{"line_number":3,"context_line":" License."},{"line_number":4,"context_line":""}],"source_content_type":"text/x-rst","patch_set":24,"id":"f57250c0_c9085d55","line":1,"updated":"2026-07-01 11:58:24.000000000","message":"The Assisted-By footer uses \u0027Claude Sonnet 4.6\u0027 which is a non-standard tool name. The style guide examples use lowercase hyphenated identifiers like \u0027claude-code\u0027 or \u0027github-copilot\u0027.\n\n**Severity**: SUGGESTION | **Confidence**: 0.6\n\n**Benefit**: Using a consistent tool identifier format across the project makes AI-assistance tracking and auditing more reliable.\n\n**Recommendation**:\nUse \u0027Assisted-By: claude-code\u0027 or \u0027Assisted-By: claude-sonnet\u0027 to match the style guide convention. The version suffix can be included parenthetically if desired: \u0027Assisted-By: claude-code (Sonnet 4.6)\u0027.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"1a7bef7d36a3bc113e53217ee93f1d747b8bd089","unresolved":false,"context_lines":[{"line_number":1,"context_line":".."},{"line_number":2,"context_line":" This work is licensed under a Creative Commons Attribution 3.0 Unported"},{"line_number":3,"context_line":" License."},{"line_number":4,"context_line":""}],"source_content_type":"text/x-rst","patch_set":24,"id":"5ac464d7_735e136d","line":1,"in_reply_to":"d70b145b_ad8c5fe4","updated":"2026-07-01 12:04:00.000000000","message":"Ack! Done","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":110,"context_line":"NVMe Device Lifecycle Flow"},{"line_number":111,"context_line":"--------------------------"},{"line_number":112,"context_line":""},{"line_number":113,"context_line":"::"},{"line_number":114,"context_line":""},{"line_number":115,"context_line":"    ┌─────────────────────────────────────────────────┐"},{"line_number":116,"context_line":"    │  Operator configures cyborg.conf                │"}],"source_content_type":"text/x-rst","patch_set":24,"id":"a3d1d49e_926e25fd","line":113,"updated":"2026-07-01 11:58:24.000000000","message":"The ASCII art lifecycle diagrams (lines 113-189) are 157 characters wide, exceeding the 79-character line length guideline from the spec template. While inside a literal block (so doc8 does not flag them), they cause horizontal scrolling in Gerrit plain-text review and terminals.\n\n**Severity**: SUGGESTION | **Confidence**: 0.6\n\n**Benefit**: Narrower ASCII diagrams improve review readability in Gerrit and terminal-based tools without changing the rendered HTML output quality.\n\n**Recommendation**:\nConsider narrowing the ASCII lifecycle diagrams to under 80 characters where possible, or breaking the flow into multiple smaller diagrams. The current diagrams are functional but wide.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":297,"context_line":""},{"line_number":298,"context_line":"Cleanup policy is configured per device via the ``clear_action`` and"},{"line_number":299,"context_line":"``clear_strategy`` keys in ``device_spec``. The naming convention is:"},{"line_number":300,"context_line":"``clear_action`` is the verb — *what* to do (sanitize, zero);"},{"line_number":301,"context_line":"``clear_strategy`` is the noun — *how* to do it (crypto, block)."},{"line_number":302,"context_line":""},{"line_number":303,"context_line":"``clear_action`` selects the cleanup operation:"}],"source_content_type":"text/x-rst","patch_set":24,"id":"ac8ce145_483fc4b9","line":300,"updated":"2026-07-01 11:58:24.000000000","message":"The naming convention for clear_strategy is described as \u0027the noun -- how to do it (crypto, block)\u0027 (lines 300-301), but \u0027crypto\u0027 and \u0027block\u0027 function as adjectives/adverbial modifiers in this context, not nouns. This framing could confuse implementers.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: Minor documentation clarity issue that could lead to misinterpretation of the configuration semantics during implementation.\n\n**Suggestion**:\nRephrase to describe clear_strategy as \u0027the modifier selecting the erase mechanism\u0027 or simply remove the verb/noun framing and describe both keys functionally: clear_action selects the operation type, clear_strategy constrains the erase mechanism.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":477,"context_line":"For zero-based cleanup:"},{"line_number":478,"context_line":""},{"line_number":479,"context_line":"#. Resolve the NVMe controller device from the PCI address"},{"line_number":480,"context_line":"#. If more than one namespace exists, delete all namespaces and"},{"line_number":481,"context_line":"   create a single namespace covering the full device (so the"},{"line_number":482,"context_line":"   entire storage is exposed for zeroing)"},{"line_number":483,"context_line":"#. Write zeroes to the full device (``nvme write-zeroes`` if"}],"source_content_type":"text/x-rst","patch_set":24,"id":"0d697d85_7646aa0b","line":480,"updated":"2026-07-01 11:58:24.000000000","message":"The zero-path creates a namespace covering the full device for zeroing (lines 480-482) but does not specify whether it should be deleted after zeroing. Line 488-489 says \u0027no namespace layout guaranteed\u0027 yet the created namespace persists on the controller.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: If the created namespace persists, the next tenant receives a device with a pre-existing namespace rather than a blank controller. This may be fine, but it contradicts the spirit of returning a \u0027clean\u0027 device and could confuse workloads expecting an unconfigured controller.\n\n**Suggestion**:\nExplicitly state whether the created namespace is deleted after zeroing or left in place. If left in place, note that the device is functionally clean (zeroed) but not configurationally clean (has a namespace). Consider deleting the namespace as a final step to return a truly blank controller.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":493,"context_line":""},{"line_number":494,"context_line":"The device lifecycle is tracked using a new ``device_state`` field"},{"line_number":495,"context_line":"added to the ``Device`` versioned object (``cyborg/objects/device.py``)"},{"line_number":496,"context_line":"and the ``devices`` database table. ``device_state`` is added to all"},{"line_number":497,"context_line":"devices regardless of type; every device follows the same state"},{"line_number":498,"context_line":"transitions with no type-specific checks. The bind guard and"},{"line_number":499,"context_line":"Placement ``reserved`` updates also apply uniformly. This is"}],"source_content_type":"text/x-rst","patch_set":24,"id":"dd373fc8_51002d35","line":496,"updated":"2026-07-01 11:58:24.000000000","message":"The device_state field is added to ALL device types (lines 496-498) but only NVMe devices use the cleaning lifecycle. Non-NVMe devices will have device_state values of \u0027available\u0027 or \u0027allocated\u0027 only, which could confuse operators querying device listings.\n\n**Severity**: SUGGESTION | **Confidence**: 0.7\n\n**Benefit**: Documenting the expected device_state values per device type reduces operator confusion and simplifies monitoring/alerting rules.\n\n**Recommendation**:\nAdd a note in the REST API impact section clarifying that non-NVMe devices will only ever show \u0027available\u0027 or \u0027allocated\u0027 device_state values, and that pending_cleaning/cleaning/error are NVMe-specific lifecycle states.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":495,"context_line":"added to the ``Device`` versioned object (``cyborg/objects/device.py``)"},{"line_number":496,"context_line":"and the ``devices`` database table. ``device_state`` is added to all"},{"line_number":497,"context_line":"devices regardless of type; every device follows the same state"},{"line_number":498,"context_line":"transitions with no type-specific checks. The bind guard and"},{"line_number":499,"context_line":"Placement ``reserved`` updates also apply uniformly. This is"},{"line_number":500,"context_line":"separate from the existing ``status`` field which remains"},{"line_number":501,"context_line":"exclusively for the enabled/disabled scheduling control."}],"source_content_type":"text/x-rst","patch_set":24,"id":"2b8497cf_1d080344","line":498,"updated":"2026-07-01 11:58:24.000000000","message":"The bind guard and Placement reserved\u003dtotal updates apply to ALL device types (lines 498-502, 571-572), not just NVMe. Existing GPU, FPGA, QAT, NIC, and SSD devices will have NULL device_state after migration, but the spec does not define how they transition into the new model.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: Existing device types that have never used device_state will have NULL values after migration. If the bind guard rejects NULL (as the invariant requires), all existing device binds could fail until online data migration completes, causing outages for non-NVMe accelerators.\n\n**Priority**: Before merge\n**Why This Matters**: Applying the bind guard universally means schema migration and online data migration must complete before any device can be bound. A conductor restart between migrations would block existing GPU/FPGA workload scheduling.\n\n**Recommendation**:\nSpecify that the bind guard treats NULL as \u0027available\u0027 (passes the guard) for backward compatibility, or document that online data migration must complete before the new conductor/API is restarted. Clarify the exact upgrade ordering: schema sync then online migration then service restart.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":615,"context_line":""},{"line_number":616,"context_line":"Relying on Nova PCI passthrough alone was rejected. Nova has a"},{"line_number":617,"context_line":"``one_time_use`` flag (``nova/compute/pci_placement_translator.py``)"},{"line_number":618,"context_line":"that sets ``reserved\u003dtotal`` when a device is allocated, but Nova"},{"line_number":619,"context_line":"never unreserves the device — it expects an external entity to do so"},{"line_number":620,"context_line":"after cleanup. Nova PCI passthrough has no cleanup mechanism for"},{"line_number":621,"context_line":"stateful devices like NVMe SSDs."}],"source_content_type":"text/x-rst","patch_set":24,"id":"2616700f_028fa1ec","line":618,"updated":"2026-07-01 11:58:24.000000000","message":"The spec references nova/compute/pci_placement_translator.py for the one_time_use flag (line 619) and RP naming scheme (lines 419-422), but does not cite specific Nova spec references or version assumptions. Nova internals can change between releases.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: Relying on undocumented Nova internal behavior for design decisions is fragile. If Nova changes its RP naming or one_time_use semantics, the Cyborg conflict detection logic may break.\n\n**Suggestion**:\nAdd a note that the OWNER_CYBORG trait-based conflict detection is the authoritative mechanism, and the Nova naming convention reference is for context only. Consider adding a cross-project spec liaison note for the Nova PCI team.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":634,"context_line":""},{"line_number":635,"context_line":"An Alembic migration adds a ``device_state`` column to the"},{"line_number":636,"context_line":"``devices`` table as a nullable Enum over the values ``available``,"},{"line_number":637,"context_line":"``allocated``, ``pending_cleaning``, ``cleaning``, and ``error``."},{"line_number":638,"context_line":"All existing rows start as ``NULL``. The column cannot default to"},{"line_number":639,"context_line":"``available`` because devices that are currently bound to instances"},{"line_number":640,"context_line":"would be incorrectly marked as available."}],"source_content_type":"text/x-rst","patch_set":24,"id":"e65448ea_80e2eda6","line":637,"updated":"2026-07-01 11:58:24.000000000","message":"The invariant \u0027reserved\u003dtotal consistent with device_state !\u003d available\u0027 (line 554) is violated between schema migration (nullable, all NULL) and online data migration completion. NULL is not \u0027available\u0027 but reserved may be 0 for devices actually in use.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: During rolling upgrade, the bind guard treats NULL as non-available (safe), but Placement reserved for a NULL device that is actually allocated would be 0, allowing potential double-allocation of a device still in use by a running instance.\n\n**Priority**: Before merge\n**Why This Matters**: A window exists after schema migration but before online data migration completes where Placement reservations and device_state can be inconsistent, potentially allowing a device in use to be allocated to a new tenant.\n\n**Recommendation**:\nAdd an explicit statement that during the upgrade window, the conductor treats devices with NULL device_state as non-allocatable (bind guard rejects NULL), or that the migration sets reserved\u003dtotal for all rows until the online migration confirms their state. Reference the cyborg-status upgrade check as the gate.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":757,"context_line":""},{"line_number":758,"context_line":"Error response codes:"},{"line_number":759,"context_line":""},{"line_number":760,"context_line":"* ``400 Bad Request`` — device does not support cleaning"},{"line_number":761,"context_line":"* ``403 Forbidden`` — caller lacks the ``cyborg:device:clean`` policy"},{"line_number":762,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":763,"context_line":"* ``501 Not Implemented`` — device supports cleaning but the agent"}],"source_content_type":"text/x-rst","patch_set":24,"id":"a300a11c_85a1cc69","line":760,"updated":"2026-07-01 11:58:24.000000000","message":"POST /v2/devices/{uuid}/clean lists 400 for \u0027device does not support cleaning\u0027 (line 760), but non-NVMe devices can never reach error state to need re-cleanup. This error case is either unreachable or implies the endpoint can be called on any device state.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: The 400 error path is ambiguous. If it\u0027s reachable (calling clean on a non-NVMe device), the spec should clarify the valid initial states. If unreachable, documenting it creates confusion for implementers writing API tests.\n\n**Suggestion**:\nClarify whether POST /clean can be called on any device UUID or only NVMe devices in error state. If the endpoint is only meaningful for NVMe devices, consider returning 400 for non-NVME and document that explicitly. Remove the 400 case if it is truly unreachable, or add a test scenario for it.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":803,"context_line":"        device.device_state \u003d \u0027available\u0027"},{"line_number":804,"context_line":"        device.save()"},{"line_number":805,"context_line":"        return"},{"line_number":806,"context_line":"    if not self.client.can_send_version(version):"},{"line_number":807,"context_line":"        # NVMe on old agent -\u003e error"},{"line_number":808,"context_line":"        device.device_state \u003d \u0027error\u0027"},{"line_number":809,"context_line":"        device.save()"}],"source_content_type":"text/x-rst","patch_set":24,"id":"a3d56102_8deb1d0b","line":806,"updated":"2026-07-01 11:58:24.000000000","message":"The spec states the conductor guards the RPC dispatch with can_send_version (line 806) but does not specify what happens if the agent is unreachable (network partition, agent crash) rather than just running an old version. The unbind path itself may silently fail since RPC casts are fire-and-forget.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: If the conductor dispatches the RPC cast and the agent is down, the cast is silently dropped. The device stays in \u0027allocated\u0027 until the next agent restart triggers reconciliation. This is a potentially long window where Placement shows the device as reserved but no cleanup is running.\n\n**Suggestion**:\nDocument that an RPC cast failure (unreachable agent) results in the device remaining in \u0027allocated\u0027 state until agent restart triggers init_host reconciliation. Consider whether the conductor should set device_state to \u0027pending_cleaning\u0027 before casting so reconciliation can detect the missed cleanup sooner.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":923,"context_line":""},{"line_number":924,"context_line":"Upgrade impact"},{"line_number":925,"context_line":"--------------"},{"line_number":926,"context_line":""},{"line_number":927,"context_line":"Enabling the generic NVMe driver is a post-upgrade configuration"},{"line_number":928,"context_line":"activity, not an in-place migration. The required upgrade ordering is:"},{"line_number":929,"context_line":"run ``cyborg-manage db sync`` to apply the schema migration, then"}],"source_content_type":"text/x-rst","patch_set":24,"id":"5f317f2e_f7c621e1","line":926,"updated":"2026-07-01 11:58:24.000000000","message":"The spec does not include an explicit rollback or disable path if the NVMe driver needs to be turned off after deployment. Operators enabling the driver may want to know how to safely revert to the previous state.\n\n**Severity**: SUGGESTION | **Confidence**: 0.7\n\n**Benefit**: Providing a disable/rollback procedure increases operator confidence in adopting the feature.\n\n**Recommendation**:\nAdd a brief note in the Upgrade impact section describing how to disable the NVMe driver (remove from enabled_drivers) and what happens to devices in cleaning/error state when the driver is disabled.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":1025,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":1026,"context_line":""},{"line_number":1027,"context_line":"``nvme-cli`` must be installed on compute nodes and is added to"},{"line_number":1028,"context_line":"``bindep.txt`` as a runtime binary dependency. An os-traits companion"},{"line_number":1029,"context_line":"patch adding ``os_traits/hw/nvme/__init__.py`` with traits ``CES``,"},{"line_number":1030,"context_line":"``BES``, and ``WZS`` must land before or alongside the Cyborg"},{"line_number":1031,"context_line":"implementation. ``futurist`` is added as a new Python dependency."}],"source_content_type":"text/x-rst","patch_set":24,"id":"91febba3_041f4fc6","line":1028,"updated":"2026-07-01 11:58:24.000000000","message":"The os-traits companion patch dependency (lines 986-987, 1029-1030) is listed as needing to land \u0027before or alongside\u0027 the Cyborg implementation, but the spec does not define a fallback if the traits are not yet available in the deployed os-traits version.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: If a Cyborg deployment uses an older os-traits that lacks HW_NVME_CES/BES/WZS, the discover() method would fail when trying to report unknown traits to Placement, or silently drop them.\n\n**Suggestion**:\nSpecify the minimum os-traits version required in Requirements, or document that Cyborg defines the trait strings locally if the os-traits dependency is not yet available. Add os-traits to the Dependencies section with a version constraint.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"2cd74ca6f30c972018027688fe1fba9af7e32e52","unresolved":false,"context_lines":[{"line_number":1028,"context_line":"``bindep.txt`` as a runtime binary dependency. An os-traits companion"},{"line_number":1029,"context_line":"patch adding ``os_traits/hw/nvme/__init__.py`` with traits ``CES``,"},{"line_number":1030,"context_line":"``BES``, and ``WZS`` must land before or alongside the Cyborg"},{"line_number":1031,"context_line":"implementation. ``futurist`` is added as a new Python dependency."},{"line_number":1032,"context_line":""},{"line_number":1033,"context_line":""},{"line_number":1034,"context_line":"Testing"}],"source_content_type":"text/x-rst","patch_set":24,"id":"6f0e34dc_26008619","line":1031,"updated":"2026-07-01 11:58:24.000000000","message":"The spec adds \u0027futurist\u0027 as a new Python dependency (line 1031) but does not specify whether Cyborg already uses it transitively via oslo.service or oslo.concurrency, or whether it needs to be added to requirements.txt explicitly.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: If futurist is already a transitive dependency, listing it as new is misleading. If not, adding it needs careful version pinning against upper-constraints.\n\n**Suggestion**:\nVerify whether futurist is already in Cyborg\u0027s transitive dependency tree (it commonly is via oslo.service). If so, note that it is being promoted to a direct dependency. If not, add it to requirements.txt with appropriate constraints.","commit_id":"689716d94eb86f80903a16a71a891e975dc24cf2"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":173,"context_line":"    └──────────┬──────────────────────┬───────────────┘"},{"line_number":174,"context_line":"               │ SUCCESS              │ FAILURE/TIMEOUT"},{"line_number":175,"context_line":"               ▼                      ▼"},{"line_number":176,"context_line":"    ┌─────────────────────┐  ┌────────────────────────┐"},{"line_number":177,"context_line":"    │  reserved \u003d 0       │  │  reserved \u003d total      │"},{"line_number":178,"context_line":"    │  device_state →     │  │  device_state → error  │"},{"line_number":179,"context_line":"    │    available         │  └──────────┬─────────────┘"}],"source_content_type":"text/x-rst","patch_set":25,"id":"57012622_f882c910","line":176,"updated":"2026-07-01 12:32:48.000000000","message":"Cleanup-success relies on the agent setting Placement reserved\u003d0 (line 177), but the spec never specifies which component owns the unreserve on success vs who set reserved\u003dtotal at bind (the conductor, line 564). Two services mutating the same Placement reserved value is underspecified.\n\n**Severity**: HIGH | **Confidence**: 0.7\n\n**Risk**: If the agent must call update_rp_inventory_reserved() to clear the reservation but lacks the conductor\u0027s placement_client, or a partial failure leaves reserved\u003dtotal after state\u003davailable, the invariant \u0027reserved\u003dtotal iff state !\u003d available\u0027 (line 554) silently breaks and the device is stranded.\n\n**Priority**: Before merge\n**Why This Matters**: The Placement reserved/device_state invariant is the spec\u0027s core safety property preventing reallocation of dirty devices. Ownership of the unreserve write must be unambiguous or the guarantee is unenforceable.\n\n**Recommendation**:\nAdd one sentence stating which service (agent or conductor) calls update_rp_inventory_reserved() to set reserved\u003d0 on cleanup success, and how the agent obtains the placement_client. State the write shares the DB transaction as the device_state\u003davailable transition so the invariant cannot be violated by a crash between the two writes.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":207,"context_line":"  that reflect the operator-selected cleanup guarantee."},{"line_number":208,"context_line":"* **Cleanup policy as an API attribute** per device. Currently"},{"line_number":209,"context_line":"  cleanup policy is operator-only configuration in ``device_spec``."},{"line_number":210,"context_line":"* **Encryption key cleanup for the zero path**. Sanitize CES handles"},{"line_number":211,"context_line":"  key rotation inherently; ensuring no stale keys remain for the zero"},{"line_number":212,"context_line":"  path is deferred to the implementation."},{"line_number":213,"context_line":"* **Minimal privsep context**. Currently all NVMe operations require"}],"source_content_type":"text/x-rst","patch_set":25,"id":"0f72a176_492f6029","line":210,"updated":"2026-07-01 12:32:48.000000000","message":"Security: the zero/write-zeroes path does not erase cryptographic keys, and key cleanup is deferred (out of scope). An encrypted device returns to \u0027available\u0027 with residual keys possibly recoverable, weakening the cross-tenant confidentiality guarantee the spec advertises.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: Residual encryption keys on a zeroed device could let an attacker recover metadata about prior tenants. The cleanup contract (lines 427-433) only guarantees clearing the host-addressable block device; the deferred key-cleanup item is not reconciled with that contract.\n\n**Priority**: Before merge\n**Why This Matters**: The blueprint\u0027s stated motivation (lines 17-21) is preventing sensitive data leaking between tenants. Leaving key material on the zero path creates a gap in that guarantee that the spec should acknowledge in Security impact, not only in the deferred-scope list.\n\n**Recommendation**:\nEither (a) add an explicit note in Security impact that clear_action\u003dzero does NOT clear cryptographic keys and operators needing key destruction must use clear_action\u003dsanitize (CES), or (b) state zero-path devices must not have had encrypted namespaces. Resolve the contract before implementation.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":285,"context_line":"The traits are defined in a new ``os_traits/hw/nvme/__init__.py``"},{"line_number":286,"context_line":"module::"},{"line_number":287,"context_line":""},{"line_number":288,"context_line":"    TRAITS \u003d ["},{"line_number":289,"context_line":"        \u0027CES\u0027,   # Crypto Erase Sanitize Supported (sanicap bit 0)"},{"line_number":290,"context_line":"        \u0027BES\u0027,   # Block Erase Sanitize Supported (sanicap bit 1)"},{"line_number":291,"context_line":"        \u0027WZS\u0027,   # Write Zeroes Supported (ONCS bit 3)"}],"source_content_type":"text/x-rst","patch_set":25,"id":"3e2fe400_afa1476f","line":288,"updated":"2026-07-01 12:32:48.000000000","message":"The os-traits patch adds a TRAITS list of bare \u0027CES\u0027,\u0027BES\u0027,\u0027WZS\u0027 (288-292), but the Placement example (413) uses the full HW_NVME_CES form. The os-traits hw convention (1090) defines full trait constants, so the bare list is ambiguous about registered names.\n\n**Severity**: SUGGESTION | **Confidence**: 0.7\n\n**Benefit**: Aligning the code snippet with the actual os-traits naming convention removes ambiguity for the companion-patch author and for Placement queries.\n\n**Recommendation**:\nShow the TRAITS list with the full names (e.g. HW_NVME_CES) or add a comment that these are suffixes and the registered trait is HW_NVME_\u003cSUFFIX\u003e, matching the existing os_traits/hw/nic module referenced as the model.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":416,"context_line":"already exists for a PCI address but does not have the ``OWNER_CYBORG``"},{"line_number":417,"context_line":"trait, the driver logs an error and skips that device. This prevents"},{"line_number":418,"context_line":"conflicts when the same device is also managed by Nova."},{"line_number":419,"context_line":"Nova\u0027s PCI placement translator (``nova/compute/pci_placement_translator.py``)"},{"line_number":420,"context_line":"uses the same ``\u003chostname\u003e_\u003cpci_address\u003e`` naming scheme,"},{"line_number":421,"context_line":"so a Nova-managed device at the same PCI address would have an RP with"},{"line_number":422,"context_line":"``CUSTOM_PCI_\u003cVENDOR_ID\u003e_\u003cPRODUCT_ID\u003e`` but without ``OWNER_CYBORG``."}],"source_content_type":"text/x-rst","patch_set":25,"id":"da174762_e1e102b9","line":419,"updated":"2026-07-01 12:32:48.000000000","message":"The spec references nova/cyborg source paths as evidence for patterns (pci_placement_translator.py lines 419, 617; image_meta.py line 685; scheduler/rpcapi.py line 799). These are cited as authoritative but the cyborg source tree is not available here to confirm the paths and pattern claims.\n\n**Severity**: SUGGESTION | **Confidence**: 0.6\n\n**Benefit**: Citations that cannot be verified weaken the \u0027follows the pattern established by\u0027 justification; confirming them avoids implementers chasing moved or renamed files.\n\n**Recommendation**:\nDuring implementation, confirm each referenced cross-project path still exists at the cited location and that the claimed pattern (Nova\u0027s one_time_use flag at line 618, volume_clear at line 346) matches current behavior; update the spec if any have drifted.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":503,"context_line":"The state transitions are (``reserved\u003dtotal`` for all states"},{"line_number":504,"context_line":"except ``available`` which has ``reserved\u003d0``)::"},{"line_number":505,"context_line":""},{"line_number":506,"context_line":"    +-----------+       bind       +-----------+"},{"line_number":507,"context_line":"    | available |-----------------\u003e| allocated |"},{"line_number":508,"context_line":"    +-----------+                  +-----------+"},{"line_number":509,"context_line":"          ^                              |"}],"source_content_type":"text/x-rst","patch_set":25,"id":"78cae890_fc221ce6","line":506,"updated":"2026-07-01 12:32:48.000000000","message":"The state-machine diagram (506-522) omits the normal unbind transition from \u0027allocated\u0027. Its only leaving arrow conflates unbind with crash recovery, but the prose (528-535) shows conductor dispatching a cleanup RPC the agent turns into pending_cleaning.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: Readers tracing the diagram alone will not see the primary unbind-\u003epending_cleaning flow and may misunderstand which transitions are normal-operation vs recovery-only.\n\n**Suggestion**:\nAdd a separate \u0027unbind (normal)\u0027 arrow from allocated to pending_cleaning, distinct from the \u0027init_host: missed cleanup\u0027 recovery arrow, so the diagram matches the prose in the Cleanup Policy (lines 435-455) and Device State Machine (lines 528-535) sections.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":581,"context_line":"``future.result(timeout\u003dcleanup_timeout)`` on the futurist thread"},{"line_number":582,"context_line":"pool future. The timeout is configurable::"},{"line_number":583,"context_line":""},{"line_number":584,"context_line":"    [nvme]"},{"line_number":585,"context_line":"    cleanup_timeout \u003d 900      # seconds (default: 15 minutes)"},{"line_number":586,"context_line":""},{"line_number":587,"context_line":"If cleanup does not complete within the timeout, the agent sets"}],"source_content_type":"text/x-rst","patch_set":25,"id":"132be470_8c9fd7b9","line":584,"updated":"2026-07-01 12:32:48.000000000","message":"cleanup_timeout defaults to 900s/15min (585) but Performance Impact (884-892) warns zero/shred on multi-TB drives takes \u0027significantly longer\u0027. Auto resolution (324-326) can pick write-zeroes/shred, so a default-timeout large drive will routinely hit \u0027error\u0027 on first cleanup.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: On large NVMe drives the default auto policy + default timeout will surface as a steady stream of false \u0027error\u0027 states, requiring operator intervention for a configuration mismatch. This undermines the \u0027automated\u0027 promise in the use cases (lines 58-60).\n\n**Suggestion**:\nEither raise the default cleanup_timeout, document that the default is sized for sanitize (CES/BES) and must be increased for zero/shred on large drives, or have the driver log a clear warning at discovery when the auto policy resolves to shred and the configured timeout is likely too low for the detected capacity.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":685,"context_line":"``ImageMeta.obj_make_compatible()`` (``nova/objects/image_meta.py``),"},{"line_number":686,"context_line":"ensuring upgraded conductors can communicate with N-1 agents."},{"line_number":687,"context_line":""},{"line_number":688,"context_line":"A ``supports_cleaning`` property is added, returning"},{"line_number":689,"context_line":"``self.type \u003d\u003d \u0027NVME\u0027``. The conductor uses this to decide whether"},{"line_number":690,"context_line":"to dispatch the ``cleanup_device`` RPC or transition the device"},{"line_number":691,"context_line":"directly to ``available`` on unbind. A driver-independent capability"}],"source_content_type":"text/x-rst","patch_set":25,"id":"548ba6b3_213ca1c9","line":688,"updated":"2026-07-01 12:32:48.000000000","message":"supports_cleaning returns self.type \u003d\u003d \u0027NVME\u0027 (688-690), hardcoding the type. This ties cleanup RPC dispatch to a string literal, not a capability, and the spec flags a driver-independent capability model as out of scope (217-223), making the base contract fragile.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: Any future device type that wants cleanup must edit this property and the conductor dispatch logic, rather than declaring support declaratively. This increases coupling at exactly the extension point the spec wants to keep open (lines 605-611).\n\n**Suggestion**:\nDocument in this section that supports_cleaning is intentionally NVME-only for now and that adding cleanup for a new type requires updating both the property and the conductor guard, so future contributors do not assume the base contract is type-agnostic.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":4690,"name":"melanie witt","display_name":"melwitt","email":"melwittt@gmail.com","username":"melwitt"},"change_message_id":"4f8fb4cb3e872599868ea52bfc612d6328c196ca","unresolved":true,"context_lines":[{"line_number":760,"context_line":"* ``400 Bad Request`` — device does not support cleaning"},{"line_number":761,"context_line":"* ``403 Forbidden`` — caller lacks the ``cyborg:device:clean`` policy"},{"line_number":762,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":763,"context_line":"* ``501 Not Implemented`` — device supports cleaning but the agent"},{"line_number":764,"context_line":"  does not support the cleanup RPC (agent too old)"},{"line_number":765,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"},{"line_number":766,"context_line":"* ``409 Conflict`` — device is in ``allocated`` state (bound to an"}],"source_content_type":"text/x-rst","patch_set":25,"id":"28a65dfd_a87312c0","line":763,"updated":"2026-07-01 18:41:31.000000000","message":"I don\u0027t want to hold up this review but this does not seem normal to me to be using and documenting a HTTP 5XX as an expected error response code? I think this should instead be a 400 or possibly 409.\n\nMore formally I find this documentation that explains not to use HTTP 501 in this manner:\n\nhttps://specs.openstack.org/openstack/api-sig/guidelines/http/response-codes.html#use-of-501-not-implemented","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":760,"context_line":"* ``400 Bad Request`` — device does not support cleaning"},{"line_number":761,"context_line":"* ``403 Forbidden`` — caller lacks the ``cyborg:device:clean`` policy"},{"line_number":762,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":763,"context_line":"* ``501 Not Implemented`` — device supports cleaning but the agent"},{"line_number":764,"context_line":"  does not support the cleanup RPC (agent too old)"},{"line_number":765,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"},{"line_number":766,"context_line":"* ``409 Conflict`` — device is in ``allocated`` state (bound to an"}],"source_content_type":"text/x-rst","patch_set":25,"id":"955ee131_301359a6","line":763,"updated":"2026-07-01 12:32:48.000000000","message":"POST /clean returns 501 when the agent lacks the cleanup RPC (763-764), but the conductor already sets state\u003d\u0027error\u0027 then (807-809). Calling /clean re-dispatches to the same old agent, which still cannot clean it, looping. The 501 recovery story is unclear.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: An operator following the documented recovery flow (POST /clean on an error device) against an N-1 agent gets 501 with no path forward short of upgrading the agent. The spec does not state that the only resolution is an agent upgrade.\n\n**Suggestion**:\nClarify in the REST API section that 501 indicates the device\u0027s agent must be upgraded to a cleanup-capable version and that retrying /clean will not help until then, so operators are not left guessing.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":11604,"name":"sean mooney","email":"smooney@redhat.com","username":"sean-k-mooney"},"change_message_id":"5e4055b6b041ca8454c554b9eef6366b9e2051bd","unresolved":true,"context_lines":[{"line_number":760,"context_line":"* ``400 Bad Request`` — device does not support cleaning"},{"line_number":761,"context_line":"* ``403 Forbidden`` — caller lacks the ``cyborg:device:clean`` policy"},{"line_number":762,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":763,"context_line":"* ``501 Not Implemented`` — device supports cleaning but the agent"},{"line_number":764,"context_line":"  does not support the cleanup RPC (agent too old)"},{"line_number":765,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"},{"line_number":766,"context_line":"* ``409 Conflict`` — device is in ``allocated`` state (bound to an"}],"source_content_type":"text/x-rst","patch_set":25,"id":"e67483ef_b9d045c4","line":763,"in_reply_to":"28a65dfd_a87312c0","updated":"2026-07-01 19:16:29.000000000","message":"oh i tought i also said to not change this\n\ni think 400 is better because something custer treat any 5** error as a provider outage and that can have billing implicaitons\n\nso yes i woudl prefer to keep this as a 400 and is we expose this we can just expose the differnce via the error message but not the code\n\nthe semantics for 400 are that we shoudl not retry the request because it not expected to work unless somethgn else changes\n409 is for trasiant failure where it might work in the future but cant for the currnt state\n\nso either are preferabel over 501\n\n501 is not entrily incorrect but i woudl not use it for this specific case","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":12393,"name":"chandan kumar","display_name":"Chandan Kumar","email":"chkumar@redhat.com","username":"chkumar246"},"change_message_id":"904e21afcb41bc581791fc5627112d9d01e2ad94","unresolved":true,"context_lines":[{"line_number":760,"context_line":"* ``400 Bad Request`` — device does not support cleaning"},{"line_number":761,"context_line":"* ``403 Forbidden`` — caller lacks the ``cyborg:device:clean`` policy"},{"line_number":762,"context_line":"* ``404 Not Found`` — device UUID does not exist"},{"line_number":763,"context_line":"* ``501 Not Implemented`` — device supports cleaning but the agent"},{"line_number":764,"context_line":"  does not support the cleanup RPC (agent too old)"},{"line_number":765,"context_line":"* ``409 Conflict`` — device is in ``available`` state (already clean)"},{"line_number":766,"context_line":"* ``409 Conflict`` — device is in ``allocated`` state (bound to an"}],"source_content_type":"text/x-rst","patch_set":25,"id":"a84b7ef4_857f0409","line":763,"in_reply_to":"e67483ef_b9d045c4","updated":"2026-07-02 04:29:14.000000000","message":"thank you Melanie , Sean for the feedback and HTTP 501 api-sig spec link.\n\nI agree here, 501 is wrong here. \nAs per the API-WG guideline, 501 means the HTTP method itself is not supported or not \"feature not implemented.\n\nChanged it to 400 bad request.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":798,"context_line":"The conductor guards the RPC dispatch with two checks, following"},{"line_number":799,"context_line":"the pattern in Nova\u0027s ``scheduler/rpcapi.py``::"},{"line_number":800,"context_line":""},{"line_number":801,"context_line":"    if not device.supports_cleaning:"},{"line_number":802,"context_line":"        # non-NVMe -\u003e available"},{"line_number":803,"context_line":"        device.device_state \u003d \u0027available\u0027"},{"line_number":804,"context_line":"        device.save()"}],"source_content_type":"text/x-rst","patch_set":25,"id":"e39a6fcb_bc3e70d1","line":801,"updated":"2026-07-01 12:32:48.000000000","message":"Internal inconsistency in the conductor/agent state-ownership contract. The spec asserts \u0027all device state management is handled by the agent\u0027 (lines 441, 534, 821), but the RPC section shows the conductor writing device_state to \u0027available\u0027 (non-NVMe, 802-805) and \u0027error\u0027 (old agent, 807-809).\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: Two services (conductor and agent) both mutating device_state from different hosts is a split-write race. Implementers reading the absolute claim may build a conductor that does not expect concurrent conductor-side writes, or may remove them as redundant.\n\n**Priority**: Before merge\n**Why This Matters**: The ownership boundary is load-bearing for the lock design. If both services write the column, the per-device locking described in Security impact (line 854, @utils.synchronized) must span both services, which a local process lock cannot do.\n\n**Recommendation**:\nReword the absolute statements to \u0027all device state transitions during the cleanup phase are handled by the agent\u0027 and document that the conductor performs terminal non-cleanup transitions (non-NVMe -\u003e available; unsupported-agent -\u003e error). Clarify conductor writes are terminal one-shot transitions outside the cleaning lifecycle.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":808,"context_line":"        device.device_state \u003d \u0027error\u0027"},{"line_number":809,"context_line":"        device.save()"},{"line_number":810,"context_line":"        return"},{"line_number":811,"context_line":"    cctxt.cast(context, \u0027cleanup_device\u0027, device\u003ddevice)"},{"line_number":812,"context_line":""},{"line_number":813,"context_line":"Non-NVMe devices transition directly to ``available`` without an"},{"line_number":814,"context_line":"RPC. If the device is NVMe but the target agent does not support"}],"source_content_type":"text/x-rst","patch_set":25,"id":"e2889a08_e7c93f84","line":811,"updated":"2026-07-01 12:32:48.000000000","message":"Async cleanup uses RPC cast (811) but the spec does not address cast lossiness or conductor retry. If the agent never receives the cast (partition, restart), the device stays \u0027allocated\u0027 until the next init_host reconciliation (598); it stays reserved but cleanup latency is unbounded.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: An operator relying on timely cleanup (e.g. capacity pressure) has no way to detect a lost cast short of polling device_state. There is no notification (explicitly out of scope, lines 857-862) and no periodic reconcile task other than agent restart.\n\n**Suggestion**:\nEither add a periodic conductor-side reconciliation task that re-casts for devices stuck in \u0027allocated\u0027 beyond a threshold, or explicitly document that cleanup is best-effort until the next agent restart and reserved capacity may accumulate. State the expected worst-case cleanup latency.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":881,"context_line":"Performance Impact"},{"line_number":882,"context_line":"------------------"},{"line_number":883,"context_line":""},{"line_number":884,"context_line":"Cleanup runs asynchronously after instance deletion. Devices stay"},{"line_number":885,"context_line":"reserved in Placement for the full cleanup window, which can be up to"},{"line_number":886,"context_line":"15 minutes by default on large drives. Operators should monitor for"},{"line_number":887,"context_line":"accumulation of devices in ``cleaning`` or ``error`` state."}],"source_content_type":"text/x-rst","patch_set":25,"id":"013b98ca_9a4eba5f","line":884,"updated":"2026-07-01 12:32:48.000000000","message":"The spec uses both \u0027cleanup_timeout\u0027 (lines 582-585) and the prose phrase \u0027cleanup window\u0027 (line 886) without cross-referencing. The config example block (lines 584-585) is the single source of truth for the option name and default; referencing it by name in Performance Impact would help operators.\n\n**Severity**: SUGGESTION | **Confidence**: 0.6\n\n**Benefit**: Operators tuning for large drives can locate the relevant config option directly from the performance guidance.\n\n**Recommendation**:\nIn the Performance Impact paragraph, name the option explicitly: \u0027Devices stay reserved for up to [nvme] cleanup_timeout (default 900s) ...\u0027 and link back to the Timeout section.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":936,"context_line":"The existing Inspur NVMe driver and SSD driver are deprecated in this"},{"line_number":937,"context_line":"release in favour of the generic NVMe driver. No new feature"},{"line_number":938,"context_line":"development will be done on the deprecated drivers outside of bug"},{"line_number":939,"context_line":"fixes. The drivers will not be removed until Nova supports resizing"},{"line_number":940,"context_line":"instances with Cyborg-managed devices, ensuring operators have a"},{"line_number":941,"context_line":"migration path from vendor-specific to generic driver. The earliest"},{"line_number":942,"context_line":"target release for removal is 2027.2; the drivers may remain deprecated"}],"source_content_type":"text/x-rst","patch_set":25,"id":"d8b1dfb9_5f0848ba","line":939,"updated":"2026-07-01 12:32:48.000000000","message":"Deprecation depends on an external prerequisite (Nova resize of Cyborg devices, 939-943) with no fallback. Inspur/SSD drivers may \u0027remain deprecated beyond 2027.2\u0027, making the timeline unbounded and dependent on another project.\n\n**Severity**: WARNING | **Confidence**: 0.7\n\n**Impact**: Two drivers with redundant, conflicting functionality (coexistence needs disjoint device sets, lines 367-373) for an indefinite period raises maintenance burden and the risk of the InvalidConfiguration startup failure (line 372) hitting operators.\n\n**Suggestion**:\nAdd an intermediate milestone: emit a deprecation warning in logs and release notes each release the prerequisite is unmet, and define a hard maximum number of releases the deprecated drivers will persist, or a documented escape hatch so deprecation is not gated solely on Nova.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"73c66d1fde7b214a963feb7e1c81021d83607397","unresolved":false,"context_lines":[{"line_number":1093,"context_line":"History"},{"line_number":1094,"context_line":"\u003d\u003d\u003d\u003d\u003d\u003d\u003d"},{"line_number":1095,"context_line":""},{"line_number":1096,"context_line":".. list-table:: Revisions"},{"line_number":1097,"context_line":"   :header-rows: 1"},{"line_number":1098,"context_line":""},{"line_number":1099,"context_line":"   * - Release Name"}],"source_content_type":"text/x-rst","patch_set":25,"id":"85391f21_585e746a","line":1096,"updated":"2026-07-01 12:32:48.000000000","message":"The history table (lines 1096-1102) lists only \u00272026.2 - Introduced\u0027. Given the patch is on patchset 25 and the spec has clearly iterated, recording prior revision notes (major design changes across revisions) per OpenStack spec convention would aid future readers.\n\n**Severity**: SUGGESTION | **Confidence**: 0.6\n\n**Benefit**: Future implementers and reviewers can trace why the design took its current shape (e.g. why no runtime fallback, why clear_action/clear_strategy split).\n\n**Recommendation**:\nAdd rows summarizing key design decisions made during review (e.g. \u0027removed runtime fallback to weaker erase\u0027, \u0027split clear_action/clear_strategy\u0027) even if condensed to one line each.","commit_id":"815d5ed46eb88a8fb0c5f0173491996d4238fdf6"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"515752e28d87d19a7b7c425965f13f1f5bdf4943","unresolved":false,"context_lines":[{"line_number":321,"context_line":"    +--------------+----------------+---------------------------------------+"},{"line_number":322,"context_line":"    | clear_action | clear_strategy | selected cleanup operation             |"},{"line_number":323,"context_line":"    +\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d+"},{"line_number":324,"context_line":"    | auto         | auto           | CES sanitize, else BES sanitize, else |"},{"line_number":325,"context_line":"    |              |                | write zeroes, else shred               |"},{"line_number":326,"context_line":"    +--------------+----------------+---------------------------------------+"},{"line_number":327,"context_line":"    | auto         | crypto         | CES sanitize only                     |"}],"source_content_type":"text/x-rst","patch_set":26,"id":"72f28f67_ba6ffe3f","line":324,"updated":"2026-07-02 04:48:06.000000000","message":"The \u0027shred\u0027 fallback does not guarantee media-level sanitization. The matrix resolves auto/block and zero paths to \u0027write zeroes, else shred\u0027, yet lets shred satisfy the \u0027zero\u0027 action and return the device to \u0027available\u0027.\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: A device cleared only via host-side shred can be returned to available and bound to a new tenant while retaining recoverable tenant data in FTL-remapped pages, undermining the spec\u0027s core anti-leakage security goal.\n\n**Priority**: Before merge\n**Why This Matters**: The blueprint\u0027s motivation is preventing cross-tenant data leakage. Treating shred as a satisfying cleanup that returns a device to available weakens the guarantee for operators selecting clear_action\u003dzero or lacking WZS; they may believe the device is sanitized when only logical blocks were over...\n\n**Recommendation**:\nPick one: (a) classify shred as non-sanitizing best-effort clear that leaves a distinct state needing operator acknowledgement before reuse; (b) drop shred from the chain and require WZS (fail to error if unsupported); or (c) state explicitly in the contract (lines 427-433) that shred does NOT satisfy the minimum guarantee and document the residual-data risk. Reconcile the contract text with the fallback.","commit_id":"b6c43f73c5d042cf2c8a10834e534b8f3bfe826a"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"515752e28d87d19a7b7c425965f13f1f5bdf4943","unresolved":false,"context_lines":[{"line_number":335,"context_line":"    +--------------+----------------+---------------------------------------+"},{"line_number":336,"context_line":"    | sanitize     | block          | BES sanitize only                     |"},{"line_number":337,"context_line":"    +--------------+----------------+---------------------------------------+"},{"line_number":338,"context_line":"    | zero         | auto/block     | write zeroes, else shred              |"},{"line_number":339,"context_line":"    +--------------+----------------+---------------------------------------+"},{"line_number":340,"context_line":"    | zero         | crypto         | invalid configuration                 |"},{"line_number":341,"context_line":"    +--------------+----------------+---------------------------------------+"}],"source_content_type":"text/x-rst","patch_set":26,"id":"ce7938b9_c8ce5db0","line":338,"updated":"2026-07-02 04:48:06.000000000","message":"The policy-matrix table collapses two clear_strategy values into one \u0027auto / block\u0027 row (line 338), inconsistent with the one-cell-per-combo style elsewhere. Also line 343 says \u0027within the zero method\u0027, which could be misread as a code method rather than the zero cleanup action.\n\n**Severity**: SUGGESTION | **Confidence**: 0.8\n\n**Benefit**: A clearer policy matrix reduces implementer ambiguity and makes the auto/block collapse explicit rather than appearing to be a typo.\n\n**Recommendation**:\nEither split the zero/auto and zero/block rows for symmetry, or add a footnote explaining they are collapsed because the resolution is identical. Rephrase \u0027within the zero method\u0027 to \u0027within the zero cleanup action\u0027.","commit_id":"b6c43f73c5d042cf2c8a10834e534b8f3bfe826a"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"515752e28d87d19a7b7c425965f13f1f5bdf4943","unresolved":false,"context_lines":[{"line_number":340,"context_line":"    | zero         | crypto         | invalid configuration                 |"},{"line_number":341,"context_line":"    +--------------+----------------+---------------------------------------+"},{"line_number":342,"context_line":""},{"line_number":343,"context_line":"The resolution chain within the ``zero`` method prefers ``nvme"},{"line_number":344,"context_line":"write-zeroes`` (controller-side, if ``WZS`` is supported) over"},{"line_number":345,"context_line":"host-side zeroing via ``shred``, following Nova\u0027s ``volume_clear``"},{"line_number":346,"context_line":"pattern for LVM volumes."}],"source_content_type":"text/x-rst","patch_set":26,"id":"a6b08fd1_f648f05e","line":343,"updated":"2026-07-02 04:48:06.000000000","message":"The \u0027no runtime fallback\u0027 invariant conflicts with the resolution chain. The locked-in action for auto/block and zero paths is itself a chain (\u0027BES sanitize, else write zeroes, else shred\u0027), yet the spec never says whether the agent may step through it at runtime or must run one operation atomica...\n\n**Severity**: HIGH | **Confidence**: 0.8\n\n**Risk**: Implementers may interpret the chain as a runtime fallback, silently downgrading cleanup strength (sanitize fails -\u003e write-zeroes -\u003e shred) and returning a weakly-cleaned device to available, defeating the fail-safe error-state design.\n\n**Priority**: Before merge\n**Why This Matters**: The no-runtime-fallback policy is the spec\u0027s primary fail-safe: a failed clean must go to error, not be silently downgraded. The current wording is ambiguous enough that an implementation could legitimately step through the chain at runtime, a security regression vs. design intent.\n\n**Recommendation**:\nAdd an explicit sentence: the resolution chain is evaluated exactly once during discovery and a single concrete operation (e.g. CES sanitize, or write-zeroes, or shred) is locked into std_board_info. At cleanup time only that locked-in operation runs; on failure/timeout the device moves to error with no attempt at the next chain entry. Clarify that POST /clean retry re-runs the same locked-in operation, not the chain.","commit_id":"b6c43f73c5d042cf2c8a10834e534b8f3bfe826a"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"515752e28d87d19a7b7c425965f13f1f5bdf4943","unresolved":false,"context_lines":[{"line_number":475,"context_line":"   failure"},{"line_number":476,"context_line":""},{"line_number":477,"context_line":"For zero-based cleanup:"},{"line_number":478,"context_line":""},{"line_number":479,"context_line":"#. Resolve the NVMe controller device from the PCI address"},{"line_number":480,"context_line":"#. If more than one namespace exists, delete all namespaces and"},{"line_number":481,"context_line":"   create a single namespace covering the full device (so the"}],"source_content_type":"text/x-rst","patch_set":26,"id":"cc78668e_7438105b","line":478,"updated":"2026-07-02 04:48:06.000000000","message":"Zero-path namespace consolidation failure handling is underspecified. The order (lines 478-486) deletes all namespaces then creates one spanning namespace before zeroing, but does not define behavior if create-ns fails after delete-ns, leaving zero namespaces.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: A failed create-ns mid-cleanup leaves the device in an inconsistent topology (no namespace) that the retry path is not designed to handle, potentially trapping devices in error state indefinitely or requiring out-of-band operator SSH access the spec aims to eliminate.\n\n**Suggestion**:\nSpecify the failure semantics for the namespace consolidation sub-steps: define whether delete-ns/create-ns are atomic with respect to error state, what topology the retry assumes, and whether a retry re-evaluates the current namespace count rather than assuming the discovery-time topology. Consider documenting that error devices in this state may require manual nvme-cli remediation.","commit_id":"b6c43f73c5d042cf2c8a10834e534b8f3bfe826a"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"515752e28d87d19a7b7c425965f13f1f5bdf4943","unresolved":false,"context_lines":[{"line_number":503,"context_line":"The state transitions are (``reserved\u003dtotal`` for all states"},{"line_number":504,"context_line":"except ``available`` which has ``reserved\u003d0``)::"},{"line_number":505,"context_line":""},{"line_number":506,"context_line":"    +-----------+       bind       +-----------+"},{"line_number":507,"context_line":"    | available |-----------------\u003e| allocated |"},{"line_number":508,"context_line":"    +-----------+                  +-----------+"},{"line_number":509,"context_line":"          ^                              |"}],"source_content_type":"text/x-rst","patch_set":26,"id":"0c35b31a_35c983d6","line":506,"updated":"2026-07-02 04:48:06.000000000","message":"The state-machine ASCII diagram (lines 506-522) omits the initial discovery path: line 524 prose says a new device starts in available, but the diagram has no \u0027new -\u003e available\u0027 edge. The full-width error box also makes the two entry conditions (failure vs crash-recovery) hard to distinguish.\n\n**Severity**: SUGGESTION | **Confidence**: 0.8\n\n**Benefit**: A state diagram that shows all transitions including initial creation gives reviewers and implementers a single source of truth for the lifecycle.\n\n**Recommendation**:\nAdd a \u0027(new)\u0027 source node with an edge to available, and consider splitting the error box or adding labels to clarify that both failure/timeout from cleaning and crash recovery from pending_cleaning/cleaning land in error. Ensure every prose transition in the surrounding text has a corresponding diagram edge.","commit_id":"b6c43f73c5d042cf2c8a10834e534b8f3bfe826a"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"515752e28d87d19a7b7c425965f13f1f5bdf4943","unresolved":false,"context_lines":[{"line_number":557,"context_line":"``init_host()`` so operators can investigate and manually correct the"},{"line_number":558,"context_line":"state."},{"line_number":559,"context_line":""},{"line_number":560,"context_line":"Device Bind"},{"line_number":561,"context_line":"^^^^^^^^^^^"},{"line_number":562,"context_line":""},{"line_number":563,"context_line":"When the Cyborg conductor binds a device to an instance, it sets"}],"source_content_type":"text/x-rst","patch_set":26,"id":"6d8d7db9_99d62d90","line":560,"updated":"2026-07-02 04:48:06.000000000","message":"device_state is added to ALL devices (lines 497-499) and the bind guard applies to all types, not only NVMe (lines 571-573). Until the online migration backfills device_state, existing devices have NULL.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: If the new bind guard does not explicitly handle NULL device_state as equivalent to available, rolling out the schema migration could break scheduling for every existing non-NVMe device until the online data migration runs, creating an upgrade hazard the spec does not warn about.\n\n**Suggestion**:\nAdd an explicit statement that the bind guard treats NULL device_state as available (i.e. the guard rejects only when device_state is not None and device_state !\u003d available), and call out in the Upgrade impact section that the online migration must complete before (or concurrently with) the conductor/api upgrade to avoid bind failures for pre-existing devices.","commit_id":"b6c43f73c5d042cf2c8a10834e534b8f3bfe826a"},{"author":{"_account_id":28006,"name":"teim-ci","display_name":"teim-ci","email":"ci@seanmooney.info","username":"ci-sean-mooney","status":"this is a third-party ci account run by sean-k-mooney on irc\nhosted at zuul.teim.app"},"tag":"autogenerated:zuul:automatic-ci","change_message_id":"515752e28d87d19a7b7c425965f13f1f5bdf4943","unresolved":false,"context_lines":[{"line_number":845,"context_line":"Verification of successful erasure is based on completed"},{"line_number":846,"context_line":"administrative commands and their reported status, not on assuming"},{"line_number":847,"context_line":"detach alone erases media. Cyborg trusts nvme-cli\u0027s output for erasure"},{"line_number":848,"context_line":"confirmation; physical-level verification is the responsibility of"},{"line_number":849,"context_line":"nvme-cli and device firmware."},{"line_number":850,"context_line":""},{"line_number":851,"context_line":"Long-running subprocesses are bounded by the ``cleanup_timeout``"}],"source_content_type":"text/x-rst","patch_set":26,"id":"b26539ae_06d3efb5","line":848,"updated":"2026-07-02 04:48:06.000000000","message":"Async cleanup concurrency bounds are under-specified. The spec uses a futurist pool (lines 445-446) bounded by pool size (lines 852-853) but gives no config option, default, or link to NVMe device count. Sanitize/write-zeroes run up to 900s and hold reserved inventory.\n\n**Severity**: WARNING | **Confidence**: 0.8\n\n**Impact**: Operators cannot size cleanup capacity, and implementers lack guidance on backpressure. A saturated pool could either queue cleanups indefinitely (devices stuck in pending_cleaning consuming reserved inventory) or error devices that could have been cleaned, both degrading fleet availability.\n\n**Suggestion**:\nAdd a configurable [nvme] cleanup_workers (or similar) option with a documented default, and specify the saturation behavior: e.g. when the pool is full, additional cleanup RPCs are queued and processed in order; devices remain in pending_cleaning with reserved\u003dtotal until a worker is free. Mirror Nova\u0027s max_concurrent_* style.","commit_id":"b6c43f73c5d042cf2c8a10834e534b8f3bfe826a"}]}
