#!/usr/bin/env bash # # JbTecWiz Support Centre -- generated fix script # # Fault : A PersistentVolumeClaim stays Pending and the pod never starts # Fix : Understand the normal case before treating it as a fault # Source: https://jbtecwiz.com/support/lnx-ctr-pvc-pending # # Run as : Root shell # Expect : 15 minutes # Risk : low # Reversible : yes # # WHEN THIS IS THE RIGHT FIX # "waiting for first consumer to be created before binding". # # HOW TO UNDO IT # Nothing was changed. # # Walks the fix one step at a time and asks before each. Steps with no # command are yours to do -- it prints those and waits. DRYRUN=1 prints # without executing; UNATTENDED=1 does not ask. # # -------------------------------------------------------------------- # NO WARRANTY - USE AT YOUR OWN RISK # # This script is provided by JbTecWiz as-is and with no warranty of any # kind, express or implied. You run it entirely at your own risk. # # JbTecWiz accepts no liability for any loss or damage arising from its # use, including but not limited to data loss, downtime, or configuration # changes that turn out to be wrong for your system. # # You are responsible for reading this script before running it, for # satisfying yourself that it suits the machine in front of you, and for # having a working backup first. Some steps cannot be undone. # -------------------------------------------------------------------- set -uo pipefail DRYRUN="${DRYRUN:-0}" UNATTENDED="${UNATTENDED:-0}" failed=0 if [ "$(id -u)" -ne 0 ]; then echo " This fix is documented as needing root. Re-run with sudo." >&2 exit 3 fi rule() { printf "\n%s\n" "$(printf '-%.0s' $(seq 1 70))"; if [ $# -gt 0 ]; then echo "$1"; fi; } prose() { echo "$1" | fold -s -w 74 | sed "s/^/ /"; } # Returns 0 when the caller should run the command, 1 when it should not. # A manual step always returns 1 -- there is nothing for the caller to run. step() { # step [command lines...] local n="$1" dotext="$2" why="$3" mode="$4"; shift 4 rule " Step $n of 4" prose "$dotext" if [ -n "$why" ]; then echo; prose "$why"; fi if [ "$mode" = "manual" ]; then echo; echo " -> Do this yourself, then press Enter to carry on." if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r _; fi return 1 fi echo; printf " %s\n" "$@"; echo if [ "$DRYRUN" = "1" ]; then echo " (dry run -- not executed)"; return 1; fi if [ "$UNATTENDED" = "0" ]; then read -r -p " Run this step? [Y]es / [S]kip / [Q]uit " a case "$a" in [Qq]*) echo " Stopped at your request."; exit 0 ;; [Ss]*) echo " Skipped."; return 1 ;; esac fi return 0 } rule echo " A PersistentVolumeClaim stays Pending and the pod never starts" echo " Understand the normal case before treating it as a fault" echo echo " Risk: low Reversible 15 minutes" echo prose 'No warranty. Use at your own risk - JbTecWiz accepts no liability. Read it before you run it, and have a backup.' rule echo if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r -p " Ready? [y/N] " go case "$go" in [Yy]*) ;; *) echo " Nothing was changed."; exit 0;; esac fi if step 1 'This message is usually not an error. A storage class with WaitForFirstConsumer deliberately delays binding until a pod is scheduled, so the volume is created in the right zone.' 'Chasing this as a storage fault is a common waste of an afternoon. If no pod is using the claim yet, Pending is the correct state and it will bind the moment one is scheduled.' cmd 'kubectl get storageclass -o custom-columns=NAME:.metadata.name,MODE:.volumeBindingMode,PROVISIONER:.provisioner'; then kubectl get storageclass -o custom-columns=NAME:.metadata.name,MODE:.volumeBindingMode,PROVISIONER:.provisioner if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 1 failed. The rest of the fix may depend on it." >&2 fi fi if step 2 'Check whether a pod actually references it.' '' cmd 'kubectl get pods -o json | jq -r '\''.items[] | select(.spec.volumes[]?.persistentVolumeClaim.claimName=="myclaim") | .metadata.name'\'''; then kubectl get pods -o json | jq -r '.items[] | select(.spec.volumes[]?.persistentVolumeClaim.claimName=="myclaim") | .metadata.name' if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 2 failed. The rest of the fix may depend on it." >&2 fi fi if step 3 'If a pod does reference it and it is still Pending, look at why the pod cannot be scheduled -- that is the real blocker.' '' cmd 'kubectl describe pod mypod | sed -n '\''/Events:/,$p'\'''; then kubectl describe pod mypod | sed -n '/Events:/,$p' if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 3 failed. The rest of the fix may depend on it." >&2 fi fi if step 4 'Check node affinity and zone constraints, which commonly make a volume unschedulable.' '' cmd 'kubectl get nodes -L topology.kubernetes.io/zone'; then kubectl get nodes -L topology.kubernetes.io/zone if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 4 failed. The rest of the fix may depend on it." >&2 fi fi rule " Confirm it worked" prose 'The claim binds once a pod is scheduled.' if [ "$DRYRUN" = "0" ]; then kubectl get pvc fi rule if [ "$failed" -gt 0 ]; then echo " Finished with $failed failed step(s)." echo " Read the full write-up at https://jbtecwiz.com/support/lnx-ctr-pvc-pending" else echo " Finished." fi echo prose 'To undo: Nothing was changed.' rule