)]}'
{"doc/source/afs.rst":[{"author":{"_account_id":4146,"name":"Clark Boylan","email":"cboylan@sapwetik.org","username":"cboylan"},"change_message_id":"c94c576cbdcf1e5db8c13b8b7cd20b22d357854d","unresolved":false,"context_lines":[{"line_number":317,"context_line":"If a fileserver crashes, take the following steps to ensure it\u0027s"},{"line_number":318,"context_line":"usable after recovery:"},{"line_number":319,"context_line":""},{"line_number":320,"context_line":"* Pause mirror updates and volume releas cron jobs"},{"line_number":321,"context_line":""},{"line_number":322,"context_line":"* Reboot the server; fix any filesystem errors and check the salvager"},{"line_number":323,"context_line":"  logs"}],"source_content_type":"text/x-rst","patch_set":1,"id":"5faad753_3503ea75","line":320,"range":{"start_line":320,"start_character":34,"end_line":320,"end_character":40},"updated":"2019-09-10 19:39:19.000000000","message":"Nit: \"release\"","commit_id":"e0fa79e377ae7839458967edf593232954f0c83d"},{"author":{"_account_id":7118,"name":"Ian Wienand","email":"iwienand@redhat.com","username":"iwienand"},"change_message_id":"2a5ac04aba18710c18b3ca53759ef94e0236c95e","unresolved":false,"context_lines":[{"line_number":319,"context_line":""},{"line_number":320,"context_line":"* Pause mirror updates and volume releas cron jobs"},{"line_number":321,"context_line":""},{"line_number":322,"context_line":"* Reboot the server; fix any filesystem errors and check the salvager"},{"line_number":323,"context_line":"  logs"},{"line_number":324,"context_line":""},{"line_number":325,"context_line":"* Check for any stuck volume transactions; remedy as appropriate"}],"source_content_type":"text/x-rst","patch_set":1,"id":"5faad753_f5679250","line":322,"updated":"2019-09-10 19:43:39.000000000","message":"perhaps we should suggest the full salvage as described in [1]\n\n bos salvage -server afs02.dfw.openstack.org -partition a -showlog -orphans attach -forceDAFS\n\n\n\n[1] https://lists.openafs.org/pipermail/openafs-devel/2018-May/020493.html","commit_id":"e0fa79e377ae7839458967edf593232954f0c83d"},{"author":{"_account_id":27803,"name":"Jeffrey Altman","email":"jaltman@auristor.com"},"change_message_id":"ff2f7fe58f13bb2c7006b488754f3eb81a9b2280","unresolved":false,"context_lines":[{"line_number":319,"context_line":""},{"line_number":320,"context_line":"* Pause mirror updates and volume releas cron jobs"},{"line_number":321,"context_line":""},{"line_number":322,"context_line":"* Reboot the server; fix any filesystem errors and check the salvager"},{"line_number":323,"context_line":"  logs"},{"line_number":324,"context_line":""},{"line_number":325,"context_line":"* Check for any stuck volume transactions; remedy as appropriate"}],"source_content_type":"text/x-rst","patch_set":1,"id":"5faad753_d347c04b","line":322,"in_reply_to":"5faad753_f5679250","updated":"2019-09-13 17:48:51.000000000","message":"Unfortunately, that command is unsafe with a dafs fileserver because the vice partition cannot be entirely detached from the fileserver.  \n\nThe safe operation is to \n\n  bos salvage -server \u003cs\u003e -volume \u003cvolid\u003e -orphans attach -forceDAFS\n\nHowever, you must still check for volume ids (not names) that are reported by \"vos listvol -server \u003cs\u003e\" but are unknown to the vl service \"vos examine -volid \u003cv\u003e\".   Those volumes are clones which the salvager will not delete.   Such volume clones must be removed using \"vos zap -server \u003cs\u003e -partition \u003cp\u003e -volid \u003cv\u003e\"","commit_id":"e0fa79e377ae7839458967edf593232954f0c83d"},{"author":{"_account_id":4146,"name":"Clark Boylan","email":"cboylan@sapwetik.org","username":"cboylan"},"change_message_id":"c94c576cbdcf1e5db8c13b8b7cd20b22d357854d","unresolved":false,"context_lines":[{"line_number":322,"context_line":"* Reboot the server; fix any filesystem errors and check the salvager"},{"line_number":323,"context_line":"  logs"},{"line_number":324,"context_line":""},{"line_number":325,"context_line":"* Check for any stuck volume transactions; remedy as appropriate"},{"line_number":326,"context_line":""},{"line_number":327,"context_line":"* Perform a manual release of every volume from a terminal on a server"},{"line_number":328,"context_line":"  using \"-localauth\" in case OpenAFS decides it can\u0027t do an"}],"source_content_type":"text/x-rst","patch_set":1,"id":"5faad753_f5f47289","line":325,"updated":"2019-09-10 19:39:19.000000000","message":"Maybe mention vos unlock as the command to specifically remedy locks?","commit_id":"e0fa79e377ae7839458967edf593232954f0c83d"},{"author":{"_account_id":4146,"name":"Clark Boylan","email":"cboylan@sapwetik.org","username":"cboylan"},"change_message_id":"c53945b4589a9a0de5027dbe49054a8541dd6985","unresolved":false,"context_lines":[{"line_number":322,"context_line":"* Reboot the server; fix any filesystem errors and check the salvager"},{"line_number":323,"context_line":"  logs"},{"line_number":324,"context_line":""},{"line_number":325,"context_line":"* Check for any stuck volume transactions; remedy as appropriate"},{"line_number":326,"context_line":""},{"line_number":327,"context_line":"* Perform a manual release of every volume from a terminal on a server"},{"line_number":328,"context_line":"  using \"-localauth\" in case OpenAFS decides it can\u0027t do an"}],"source_content_type":"text/x-rst","patch_set":1,"id":"5faad753_f3e0bc28","line":325,"in_reply_to":"5faad753_4838fbec","updated":"2019-09-13 17:54:38.000000000","message":"Our fileserver crashed. This prevented `vol examine $volume` from functioning for volumes hosted on that particular fileserver. However the db servers (which I\u0027m guessing are our vol servers) did not crash. This meant that vos listvldb showed locked volumes for volumes that had ongoing vos releases at the time of the crash. These then timed out authentication and became stale.\n\nMy suggestion here is to check vos listvdlb after this incident to look for any stale locks as pausing updates and volume release cron jobs means we have no ongoing releases so nothing should be locked.\n\nAny lock that remains at that point should be removed.","commit_id":"e0fa79e377ae7839458967edf593232954f0c83d"},{"author":{"_account_id":27803,"name":"Jeffrey Altman","email":"jaltman@auristor.com"},"change_message_id":"352be5d0e55dba4b3b2bd82d33d485da4602d44e","unresolved":false,"context_lines":[{"line_number":322,"context_line":"* Reboot the server; fix any filesystem errors and check the salvager"},{"line_number":323,"context_line":"  logs"},{"line_number":324,"context_line":""},{"line_number":325,"context_line":"* Check for any stuck volume transactions; remedy as appropriate"},{"line_number":326,"context_line":""},{"line_number":327,"context_line":"* Perform a manual release of every volume from a terminal on a server"},{"line_number":328,"context_line":"  using \"-localauth\" in case OpenAFS decides it can\u0027t do an"}],"source_content_type":"text/x-rst","patch_set":1,"id":"5faad753_cee39333","line":325,"in_reply_to":"5faad753_f3e0bc28","updated":"2019-09-13 18:47:34.000000000","message":"VL servers are not the same a VOL servers.   The FILE, VOL, and SALVAGE services are co-dependent processes that execute as part of the same \"bnode\" launched by the bosserver.\n\nThe VL servers run on separate hosts in the openstack.org cell.","commit_id":"e0fa79e377ae7839458967edf593232954f0c83d"},{"author":{"_account_id":27803,"name":"Jeffrey Altman","email":"jaltman@auristor.com"},"change_message_id":"ff2f7fe58f13bb2c7006b488754f3eb81a9b2280","unresolved":false,"context_lines":[{"line_number":322,"context_line":"* Reboot the server; fix any filesystem errors and check the salvager"},{"line_number":323,"context_line":"  logs"},{"line_number":324,"context_line":""},{"line_number":325,"context_line":"* Check for any stuck volume transactions; remedy as appropriate"},{"line_number":326,"context_line":""},{"line_number":327,"context_line":"* Perform a manual release of every volume from a terminal on a server"},{"line_number":328,"context_line":"  using \"-localauth\" in case OpenAFS decides it can\u0027t do an"}],"source_content_type":"text/x-rst","patch_set":1,"id":"5faad753_4838fbec","line":325,"in_reply_to":"5faad753_f5f47289","updated":"2019-09-13 17:48:51.000000000","message":"Volume transactions will not survive a volserver restart; they certainly will not survive a machine crash.\n\nVolume forwards that involve a volserver terminating unexpectedly might require manual termination of the transaction(s) on the other volserver involved in the volume forward.\n\n  vos status\n  vos endtrans\n\nWhat kind of crash are we documenting recovery of?","commit_id":"e0fa79e377ae7839458967edf593232954f0c83d"},{"author":{"_account_id":7118,"name":"Ian Wienand","email":"iwienand@redhat.com","username":"iwienand"},"change_message_id":"2a5ac04aba18710c18b3ca53759ef94e0236c95e","unresolved":false,"context_lines":[{"line_number":326,"context_line":""},{"line_number":327,"context_line":"* Perform a manual release of every volume from a terminal on a server"},{"line_number":328,"context_line":"  using \"-localauth\" in case OpenAFS decides it can\u0027t do an"},{"line_number":329,"context_line":"  incremental update."},{"line_number":330,"context_line":""},{"line_number":331,"context_line":"* Re-enable cron jobs"},{"line_number":332,"context_line":""}],"source_content_type":"text/x-rst","patch_set":1,"id":"5faad753_d527f611","line":329,"updated":"2019-09-10 19:43:39.000000000","message":"Looking at the logs @ http://paste.openstack.org/show/774655/ i\u0027m not 100% sure it was a straight case of couldn\u0027t do incremental ... it doesn\u0027t have the same \"Deleting extant RO_DONTUSE site on afs02.dfw.openstack.org... done\" that usually proceeds that.  we did look at the timing of the reboots and this failure in #openstack-infra relating to this at the time","commit_id":"e0fa79e377ae7839458967edf593232954f0c83d"},{"author":{"_account_id":1,"name":"James E. Blair","email":"jim@acmegating.com","username":"corvus"},"change_message_id":"37b52c778fb31538336bfc7a0bc28aed8b4c76e2","unresolved":false,"context_lines":[{"line_number":322,"context_line":"* Reboot the server; fix any filesystem errors and check the salvager"},{"line_number":323,"context_line":"  logs. To clean up orphaned files you can run::"},{"line_number":324,"context_line":""},{"line_number":325,"context_line":"    bos salvage -server $FILESERVER -partition a -showlog -orphans attach -forceDAFS"},{"line_number":326,"context_line":""},{"line_number":327,"context_line":"* Check for any stuck volume transactions; remedy as appropriate."},{"line_number":328,"context_line":"  In particular ``vos listvldb`` will show locked volumes for which there"}],"source_content_type":"text/x-rst","patch_set":2,"id":"5faad753_8815939e","line":325,"updated":"2019-09-13 17:24:04.000000000","message":"Can this run with the fileserver process running?","commit_id":"4b62a24db8da1ddbb13ca5018cb53c4c2099d977"}]}
