#!/usr/bin/env bash # # JbTecWiz Support Centre -- generated fix script # # Fault : A Kubernetes node goes NotReady # Fix : Deal with an unresponsive container runtime # Source: https://jbtecwiz.com/support/lnx-ctr-node-notready # # Run as : Root shell on the node # Expect : 40 minutes # Risk : high # Reversible : yes # # WHEN THIS IS THE RIGHT FIX # PLEG is not healthy, or container operations time out. # # HOW TO UNDO IT # kubectl uncordon returns the node to service; nothing persistent was # changed. # # Walks the fix one step at a time and asks before each. Steps with no # command are yours to do -- it prints those and waits. DRYRUN=1 prints # without executing; UNATTENDED=1 does not ask. # # -------------------------------------------------------------------- # NO WARRANTY - USE AT YOUR OWN RISK # # This script is provided by JbTecWiz as-is and with no warranty of any # kind, express or implied. You run it entirely at your own risk. # # JbTecWiz accepts no liability for any loss or damage arising from its # use, including but not limited to data loss, downtime, or configuration # changes that turn out to be wrong for your system. # # You are responsible for reading this script before running it, for # satisfying yourself that it suits the machine in front of you, and for # having a working backup first. Some steps cannot be undone. # -------------------------------------------------------------------- set -uo pipefail DRYRUN="${DRYRUN:-0}" UNATTENDED="${UNATTENDED:-0}" failed=0 if [ "$(id -u)" -ne 0 ]; then echo " This fix is documented as needing root. Re-run with sudo." >&2 exit 3 fi rule() { printf "\n%s\n" "$(printf '-%.0s' $(seq 1 70))"; if [ $# -gt 0 ]; then echo "$1"; fi; } prose() { echo "$1" | fold -s -w 74 | sed "s/^/ /"; } # Returns 0 when the caller should run the command, 1 when it should not. # A manual step always returns 1 -- there is nothing for the caller to run. step() { # step [command lines...] local n="$1" dotext="$2" why="$3" mode="$4"; shift 4 rule " Step $n of 5" prose "$dotext" if [ -n "$why" ]; then echo; prose "$why"; fi if [ "$mode" = "manual" ]; then echo; echo " -> Do this yourself, then press Enter to carry on." if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r _; fi return 1 fi echo; printf " %s\n" "$@"; echo if [ "$DRYRUN" = "1" ]; then echo " (dry run -- not executed)"; return 1; fi if [ "$UNATTENDED" = "0" ]; then read -r -p " Run this step? [Y]es / [S]kip / [Q]uit " a case "$a" in [Qq]*) echo " Stopped at your request."; exit 0 ;; [Ss]*) echo " Skipped."; return 1 ;; esac fi return 0 } rule echo " A Kubernetes node goes NotReady" echo " Deal with an unresponsive container runtime" echo echo " Risk: high Reversible 40 minutes" echo prose 'No warranty. Use at your own risk - JbTecWiz accepts no liability. Read it before you run it, and have a backup.' rule echo if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r -p " Ready? [y/N] " go case "$go" in [Yy]*) ;; *) echo " Nothing was changed."; exit 0;; esac fi if step 1 'PLEG is the kubelet'\''s pod lifecycle event generator. It reports unhealthy when the runtime takes too long to list containers, which usually means the runtime is overloaded or wedged.' '' cmd 'sudo crictl info | head -20' 'sudo crictl ps 2>&1 | head'; then sudo crictl info | head -20 sudo crictl ps 2>&1 | head if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 1 failed. The rest of the fix may depend on it." >&2 fi fi if step 2 'Check how many containers the node is holding -- an accumulation of exited containers slows every list operation.' 'PLEG lists every container on every cycle. A node holding thousands of dead containers cannot complete that within the timeout, and the node flaps between Ready and NotReady.' cmd 'sudo crictl ps -a | wc -l' 'sudo crictl pods | wc -l'; then sudo crictl ps -a | wc -l sudo crictl pods | wc -l if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 2 failed. The rest of the fix may depend on it." >&2 fi fi if step 3 'Remove exited containers.' '' cmd 'sudo crictl rm --all --force 2>/dev/null || true'; then sudo crictl rm --all --force 2>/dev/null || true if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 3 failed. The rest of the fix may depend on it." >&2 fi fi if step 4 'Check I/O wait, which is the other common cause.' '' cmd 'iostat -x 2 5' 'vmstat 2 5'; then iostat -x 2 5 vmstat 2 5 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 4 failed. The rest of the fix may depend on it." >&2 fi fi if step 5 'Cordon and drain before restarting the runtime, so workloads move rather than being killed.' '' cmd 'kubectl cordon node01' 'kubectl drain node01 --ignore-daemonsets --delete-emptydir-data' 'sudo systemctl restart containerd kubelet' 'kubectl uncordon node01'; then kubectl cordon node01 kubectl drain node01 --ignore-daemonsets --delete-emptydir-data sudo systemctl restart containerd kubelet kubectl uncordon node01 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 5 failed. The rest of the fix may depend on it." >&2 fi fi rule " Confirm it worked" prose 'The node stays Ready and PLEG warnings stop appearing.' if [ "$DRYRUN" = "0" ]; then sudo journalctl -u kubelet --since '10 min ago' | grep -ci pleg fi rule if [ "$failed" -gt 0 ]; then echo " Finished with $failed failed step(s)." echo " Read the full write-up at https://jbtecwiz.com/support/lnx-ctr-node-notready" else echo " Finished." fi echo prose 'To undo: kubectl uncordon returns the node to service; nothing persistent was changed.' rule