)]}'
{"doc/source/troubleshooting_guide.rst":[{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[],"source_content_type":"","patch_set":1,"id":"bf51134e_012f1b8e","updated":"2020-06-22 20:10:10.000000000","message":"nit: iDRAC is proprietary to Dell. We should replace iDRAC with BMC for generality.","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[{"line_number":248,"context_line":"portion of the nodes in the environment, usually due to hardware issues or other"},{"line_number":249,"context_line":"basic problems with PXE or networking in a new environment."},{"line_number":250,"context_line":""},{"line_number":251,"context_line":"After executing the site deployment, logon to the MaaS dashboard to monitor node"},{"line_number":252,"context_line":"deployment progress. MaaS has several deployment states to be aware of, where"},{"line_number":253,"context_line":"the node can become stuck for different reasons."},{"line_number":254,"context_line":""}],"source_content_type":"text/x-rst","patch_set":1,"id":"bf51134e_af7e1bac","line":251,"range":{"start_line":251,"start_character":37,"end_line":251,"end_character":42},"updated":"2020-06-22 20:10:10.000000000","message":"login\n\nHow do I access the MAAS dashboard?","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[{"line_number":278,"context_line":"This phase is an ideal opportunity to validate that all nodes are discovered."},{"line_number":279,"context_line":"Look over the list of nodes in MaaS to identify any missing hardware. The"},{"line_number":280,"context_line":"expectation is that you should see a number of nodes registered in MaaS that"},{"line_number":281,"context_line":"equal the number of ``BaremetalNode`` defined in your site definition. Failure"},{"line_number":282,"context_line":"to discover nodes is usually due to a PXE boot problem or misconfiguration:"},{"line_number":283,"context_line":""},{"line_number":284,"context_line":"    1. Basic network fabric communication failure or misconfiguration. The"}],"source_content_type":"text/x-rst","patch_set":1,"id":"bf51134e_af573b26","line":281,"range":{"start_line":281,"start_character":20,"end_line":281,"end_character":37},"updated":"2020-06-22 20:10:10.000000000","message":"This should be plural.\n\nWe could change this to \n\n  ``BaremetalNode`` documents","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[{"line_number":278,"context_line":"This phase is an ideal opportunity to validate that all nodes are discovered."},{"line_number":279,"context_line":"Look over the list of nodes in MaaS to identify any missing hardware. The"},{"line_number":280,"context_line":"expectation is that you should see a number of nodes registered in MaaS that"},{"line_number":281,"context_line":"equal the number of ``BaremetalNode`` defined in your site definition. Failure"},{"line_number":282,"context_line":"to discover nodes is usually due to a PXE boot problem or misconfiguration:"},{"line_number":283,"context_line":""},{"line_number":284,"context_line":"    1. Basic network fabric communication failure or misconfiguration. The"}],"source_content_type":"text/x-rst","patch_set":1,"id":"bf51134e_8f54f722","line":281,"range":{"start_line":281,"start_character":54,"end_line":281,"end_character":69},"updated":"2020-06-22 20:10:10.000000000","message":"site documents","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[{"line_number":326,"context_line":"        rate depending on your luck with timing, which is another reason to fix"},{"line_number":327,"context_line":"        this issue even if you are able to PXE boot nodes after enough retries."},{"line_number":328,"context_line":""},{"line_number":329,"context_line":"    3. LACP fallback not configured. In environments where PXE traffic rides on"},{"line_number":330,"context_line":"       the same interfaces that will participate in an LACP bond, the ``LACP"},{"line_number":331,"context_line":"       fallback`` or equivalent setting must be enabled."},{"line_number":332,"context_line":""}],"source_content_type":"text/x-rst","patch_set":1,"id":"bf51134e_2f396bf1","line":329,"range":{"start_line":329,"start_character":7,"end_line":329,"end_character":36},"updated":"2020-06-22 20:10:10.000000000","message":"\"is not configured\"","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[{"line_number":330,"context_line":"       the same interfaces that will participate in an LACP bond, the ``LACP"},{"line_number":331,"context_line":"       fallback`` or equivalent setting must be enabled."},{"line_number":332,"context_line":""},{"line_number":333,"context_line":"    4. Server PXE booting off the wrong network interface. Check iDRAC/iLO"},{"line_number":334,"context_line":"       console to see which NIC is used for PXE during boot."},{"line_number":335,"context_line":""},{"line_number":336,"context_line":"    5. Server not configured for PXE boot or was configured for PXE \"next boot\""}],"source_content_type":"text/x-rst","patch_set":1,"id":"bf51134e_0f3e27d9","line":333,"range":{"start_line":333,"start_character":17,"end_line":333,"end_character":26},"updated":"2020-06-22 20:10:10.000000000","message":"is booting","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[{"line_number":333,"context_line":"    4. Server PXE booting off the wrong network interface. Check iDRAC/iLO"},{"line_number":334,"context_line":"       console to see which NIC is used for PXE during boot."},{"line_number":335,"context_line":""},{"line_number":336,"context_line":"    5. Server not configured for PXE boot or was configured for PXE \"next boot\""},{"line_number":337,"context_line":"       but not persistently for subsequent boots. Check the iDRAC/iLO console to"},{"line_number":338,"context_line":"       see what boot device the server is attempting."},{"line_number":339,"context_line":""}],"source_content_type":"text/x-rst","patch_set":1,"id":"bf51134e_4f23ffbd","line":336,"range":{"start_line":336,"start_character":14,"end_line":336,"end_character":28},"updated":"2020-06-22 20:10:10.000000000","message":"is not configured","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[{"line_number":344,"context_line":"only be initiated when the node is off. If MaaS will not allow the node to be"},{"line_number":345,"context_line":"commissioned because it is powered on, this suggests an IPMI problem (e.g., the"},{"line_number":346,"context_line":"node power-off that should have happened after step 1 above may not have"},{"line_number":347,"context_line":"occurred). Refer to the Troubleshooting IPMI Issues section in this cookbook to"},{"line_number":348,"context_line":"get more information."},{"line_number":349,"context_line":""},{"line_number":350,"context_line":"When working properly, MaaS will power-on the node via IPMI and PXE boot it"}],"source_content_type":"text/x-rst","patch_set":1,"id":"bf51134e_8f29979c","line":347,"range":{"start_line":347,"start_character":63,"end_line":347,"end_character":76},"updated":"2020-06-22 20:10:10.000000000","message":"what cookbook?","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[{"line_number":398,"context_line":""},{"line_number":399,"context_line":"If there are commission failures with no node logs available, the basic network"},{"line_number":400,"context_line":"may not be functioning in the discovery OS used for commissioning. In this case,"},{"line_number":401,"context_line":"ou should login to the iDRAC/iLO belonging to the node and watch the console as"},{"line_number":402,"context_line":"the commission attempt is made. Most of the commission output should be posted"},{"line_number":403,"context_line":"to the console. At this time, Drydock does not support configuring a custom MaaS"},{"line_number":404,"context_line":"pre-seed which would allow setting a password, so currently the only information"}],"source_content_type":"text/x-rst","patch_set":1,"id":"bf51134e_813a0bc9","line":401,"range":{"start_line":401,"start_character":0,"end_line":401,"end_character":2},"updated":"2020-06-22 20:10:10.000000000","message":"You lost me when you switched to Francais :)","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[{"line_number":411,"context_line":"being discovered, commissioned, and deployed by MaaS at any one time add to the"},{"line_number":412,"context_line":"total MaaS load."},{"line_number":413,"context_line":""},{"line_number":414,"context_line":"In some cases, commission timeouts may still occur due to slow network"},{"line_number":415,"context_line":"(taking too long to fetchpackages) or some other part of the commissioning"},{"line_number":416,"context_line":"process that is exceeding the default timeout (15 minutes). Unfortunately,"},{"line_number":417,"context_line":"MaaS does not presently allow configuration of this timeout, so it cannot be"},{"line_number":418,"context_line":"extended at this time."},{"line_number":419,"context_line":""},{"line_number":420,"context_line":"One very good commissioning sanity check to perform is to look at the entire"},{"line_number":421,"context_line":"list of nodes, and compare the reported resources for CPU, memory, disk, and"},{"line_number":422,"context_line":"network. In particular, look for nodes with resource values that don\u0027t match the"},{"line_number":423,"context_line":"others, as this is a likely indicator of hardware problems that should be"},{"line_number":424,"context_line":"investigated now instead of when they cause unexpected problems later. Note the"},{"line_number":425,"context_line":"following:"},{"line_number":426,"context_line":""},{"line_number":427,"context_line":"    - If you have nodes with non-matching CPU counts, then you likely have a"},{"line_number":428,"context_line":"      heterogeneous mix of different types of hardware with different types of"},{"line_number":429,"context_line":"      CPUs. In this case you should perform a careful audit of the hardware to"},{"line_number":430,"context_line":"      ensure that you are booting the right nodes, as its expected all nodes to"},{"line_number":431,"context_line":"      have matching specifications in Airship. If you are certain this is a"},{"line_number":432,"context_line":"      valid exception, and you do intend to leverage multiple hardware profiles,"},{"line_number":433,"context_line":"      you should ensure at this stage that the hardware profiles and allocated"},{"line_number":434,"context_line":"      node labels in your site manifests match the resource reports from MaaS."},{"line_number":435,"context_line":""},{"line_number":436,"context_line":"    - Ensure that the ``isolcpus`` and ``vcpu_pin_set`` in your site manifests"},{"line_number":437,"context_line":"      \"agree\" with the number of CPUs reported for data plane nodes in your"},{"line_number":438,"context_line":"      environment. You should have a greater number of reported CPUs than the"},{"line_number":439,"context_line":"      number of ``isolcpus`` listed for your data plane host profile(s). If you"},{"line_number":440,"context_line":"      are using more than one data plane host profile, ensure that all profiles"},{"line_number":441,"context_line":"      agree with MaaS resource reporting."},{"line_number":442,"context_line":""},{"line_number":443,"context_line":"    - If you have a small number of nodes reporting less RAM than the others,"},{"line_number":444,"context_line":"      it is likely you have some bad RAM modules that need replacement. You can"},{"line_number":445,"context_line":"      verify this by logging into the iDRAC/iLO of affected nodes and viewing"},{"line_number":446,"context_line":"      the RAM health information. Report the nodes with bad RAM to data center"},{"line_number":447,"context_line":"      personnel so they can be replaced. Nodes with less RAM than expected by"},{"line_number":448,"context_line":"      the hardware profile will cause failure of hugepage allocation, which"},{"line_number":449,"context_line":"      expects a specific number of 1G pages to be available. If you have a"},{"line_number":450,"context_line":"      larger number of nodes with disagreeing RAM values, you may have"},{"line_number":451,"context_line":"      non-homogeneous hardware; see CPU count disagreements above."},{"line_number":452,"context_line":""},{"line_number":453,"context_line":"    - Ensure that the number of ``hugepages`` in your site manifests \"agree\""},{"line_number":454,"context_line":"      with the amount of RAM reported for data plane nodes in your environment."},{"line_number":455,"context_line":"      You should have a greater amount of RAM reported (in GB) than the number"},{"line_number":456,"context_line":"      of hugepages. If you are using more than one data-plane host profile,"},{"line_number":457,"context_line":"      ensure that all profiles agree with MaaS resource reporting."},{"line_number":458,"context_line":""},{"line_number":459,"context_line":"    - If you have a small number of nodes with non-matching disk counts or disk"},{"line_number":460,"context_line":"      capacity, it is likely that hardware RAID may be configured differently"},{"line_number":461,"context_line":"      (or not configured) or that there are disk failures on these nodes. Login"},{"line_number":462,"context_line":"      to the iDRAC/iLO of these nodes to check their disk health and 114 have"},{"line_number":463,"context_line":"      data center personnel replace any failed disks. (Ceph in particular will"},{"line_number":464,"context_line":"      have issues if the expected number of disks are not present.) Also check"},{"line_number":465,"context_line":"      the RAID configuration, and make any changes required to reconfigure RAID"},{"line_number":466,"context_line":"      settings consistently for nodes of each kind. Note: Control plane nodes"},{"line_number":467,"context_line":"      are expected to have different disk configuration from the data plane"},{"line_number":468,"context_line":"      nodes, so you should see two different disk layouts. Ensure that your"},{"line_number":469,"context_line":"      site manifest agrees with the nodes selected for control plane versus the"},{"line_number":470,"context_line":"      ones selected for data plane based on the disk configurations (control"},{"line_number":471,"context_line":"      plane nodes will have HDDs in JBOD for Ceph, whereas data plane nodes"},{"line_number":472,"context_line":"      will have the HDDs in one hardware RAID array)."},{"line_number":473,"context_line":""},{"line_number":474,"context_line":"    - Ensure that the disk configurations for hardware profiles in site"},{"line_number":475,"context_line":"      manifests match the available disk resources reported by MaaS."},{"line_number":476,"context_line":""},{"line_number":477,"context_line":"    - If you have a small number of nodes with non-matching NIC counts, it is"},{"line_number":478,"context_line":"      likely that these nodes may have NIC failures, or (in case of blades) that"},{"line_number":479,"context_line":"      the blade chassis needs to be reseated. Login to iDRAC/iLO for these nodes"},{"line_number":480,"context_line":"      to examine NIC health and report any failed or link-down NIC to data"},{"line_number":481,"context_line":"      center personnel. Missing NICs should not cause a provisioning failure as"},{"line_number":482,"context_line":"      long as it is an unused NIC or part of a bond that has at least one"},{"line_number":483,"context_line":"      operational link."},{"line_number":484,"context_line":""},{"line_number":485,"context_line":"Also note that nodes that report \"0\" for a resource likely have not completed"},{"line_number":486,"context_line":"commissioning, have failed commissioning, were released/rediscovered, or may"},{"line_number":487,"context_line":"have encountered some other commissioning-related problem. Updates to resource"},{"line_number":488,"context_line":"totals are made incrementally (e.g., first CPU, then memory, etc.) with some"},{"line_number":489,"context_line":"delay between each update for each node. In these cases you may need to wait for"},{"line_number":490,"context_line":"commissioning to complete or may want to try re-commissioning the node again to"},{"line_number":491,"context_line":"see if any erroneously reported resource information corrects itself."},{"line_number":492,"context_line":""},{"line_number":493,"context_line":"By default, again MaaS will power off the node after successful commissioning."},{"line_number":494,"context_line":""}],"source_content_type":"text/x-rst","patch_set":1,"id":"bf51134e_01543bfe","line":491,"range":{"start_line":414,"start_character":0,"end_line":491,"end_character":69},"updated":"2020-06-22 20:10:10.000000000","message":"I think a lot of this is repeated. Please verify","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":28618,"name":"Drew Walters","email":"drewwalters@microsoft.com","username":"drewwalters96"},"change_message_id":"130bb16415b852f4a1f4cbbbe5ba6c39d3ede2d8","unresolved":false,"context_lines":[{"line_number":500,"context_line":"final target OS, and do one final reboot that boots to local disk."},{"line_number":501,"context_line":""},{"line_number":502,"context_line":"Deployments can fail for some of the same reasons as commissioning (see loading"},{"line_number":503,"context_line":"problems in previous section; also apt mirror availability). MaaS provides a"},{"line_number":504,"context_line":"node deployment log which contains more details of specific deployment failures."}],"source_content_type":"text/x-rst","patch_set":1,"id":"bf51134e_c1452345","line":504,"range":{"start_line":503,"start_character":60,"end_line":504,"end_character":80},"updated":"2020-06-22 20:10:10.000000000","message":"Where can I retrieve this log?","commit_id":"8fad64729acb80e655b496f7bdb63944192b76ef"},{"author":{"_account_id":22348,"name":"Zuul","username":"zuul","tags":["SERVICE_USER"]},"tag":"autogenerated:zuul:check","change_message_id":"61fd1722d3904a972c294bf34fe57b2dca40acc2","unresolved":false,"context_lines":[{"line_number":318,"context_line":""},{"line_number":319,"context_line":"    # Exec into the Monitor pod"},{"line_number":320,"context_line":"    sudo kubectl exec -it -n ceph ${CEPH_MON} -- ceph -s"},{"line_number":321,"context_line":"---------------------------------"},{"line_number":322,"context_line":"Node Provisioning Troubleshooting"},{"line_number":323,"context_line":"---------------------------------"},{"line_number":324,"context_line":""}],"source_content_type":"text/x-rst","patch_set":3,"id":"bf51134e_b31bfa70","line":321,"updated":"2020-06-24 17:41:47.000000000","message":"docs: Literal block ends without a blank line; unexpected unindent.","commit_id":"6e178adaa77a47e87e8d9fd126239543b0fbc145"},{"author":{"_account_id":22348,"name":"Zuul","username":"zuul","tags":["SERVICE_USER"]},"tag":"autogenerated:zuul:check","change_message_id":"d3748a8e1731b002733abe3fae398191eb2ab421","unresolved":false,"context_lines":[{"line_number":318,"context_line":""},{"line_number":319,"context_line":"    # Exec into the Monitor pod"},{"line_number":320,"context_line":"    sudo kubectl exec -it -n ceph ${CEPH_MON} -- ceph -s"},{"line_number":321,"context_line":"---------------------------------"},{"line_number":322,"context_line":"Node Provisioning Troubleshooting"},{"line_number":323,"context_line":"---------------------------------"},{"line_number":324,"context_line":""}],"source_content_type":"text/x-rst","patch_set":4,"id":"bf51134e_b36c9a9c","line":321,"updated":"2020-06-24 18:00:06.000000000","message":"docs: Literal block ends without a blank line; unexpected unindent.","commit_id":"f8c251e0b749e99053022a0a931bee971a4c18ee"},{"author":{"_account_id":22348,"name":"Zuul","username":"zuul","tags":["SERVICE_USER"]},"tag":"autogenerated:zuul:check","change_message_id":"88ecfdbb1a9b6b65ce30f86ee9b0252e00ae4f52","unresolved":false,"context_lines":[{"line_number":318,"context_line":""},{"line_number":319,"context_line":"    # Exec into the Monitor pod"},{"line_number":320,"context_line":"    sudo kubectl exec -it -n ceph ${CEPH_MON} -- ceph -s"},{"line_number":321,"context_line":"---------------------------------"},{"line_number":322,"context_line":"Node Provisioning Troubleshooting"},{"line_number":323,"context_line":"---------------------------------"},{"line_number":324,"context_line":""}],"source_content_type":"text/x-rst","patch_set":6,"id":"bf51134e_33d08ac9","line":321,"updated":"2020-06-24 18:32:06.000000000","message":"docs: Literal block ends without a blank line; unexpected unindent.","commit_id":"1a8a526d45783e2f03880f6258cc91c65f404394"}]}
