#!/usr/bin/env bash # # JbTecWiz Support Centre -- generated fix script # # Fault : ImagePullBackOff, ErrImagePull and registry rate limits # Fix : Stop pulling anonymously # Source: https://jbtecwiz.com/support/lnx-ctr-imagepull # # Run as : Root shell # Expect : 30 minutes # Risk : low # Reversible : yes # # WHEN THIS IS THE RIGHT FIX # A rate limit message. # # HOW TO UNDO IT # Remove the secret and the service account patch. # # Walks the fix one step at a time and asks before each. Steps with no # command are yours to do -- it prints those and waits. DRYRUN=1 prints # without executing; UNATTENDED=1 does not ask. # # -------------------------------------------------------------------- # NO WARRANTY - USE AT YOUR OWN RISK # # This script is provided by JbTecWiz as-is and with no warranty of any # kind, express or implied. You run it entirely at your own risk. # # JbTecWiz accepts no liability for any loss or damage arising from its # use, including but not limited to data loss, downtime, or configuration # changes that turn out to be wrong for your system. # # You are responsible for reading this script before running it, for # satisfying yourself that it suits the machine in front of you, and for # having a working backup first. Some steps cannot be undone. # -------------------------------------------------------------------- set -uo pipefail DRYRUN="${DRYRUN:-0}" UNATTENDED="${UNATTENDED:-0}" failed=0 if [ "$(id -u)" -ne 0 ]; then echo " This fix is documented as needing root. Re-run with sudo." >&2 exit 3 fi rule() { printf "\n%s\n" "$(printf '-%.0s' $(seq 1 70))"; if [ $# -gt 0 ]; then echo "$1"; fi; } prose() { echo "$1" | fold -s -w 74 | sed "s/^/ /"; } # Returns 0 when the caller should run the command, 1 when it should not. # A manual step always returns 1 -- there is nothing for the caller to run. step() { # step [command lines...] local n="$1" dotext="$2" why="$3" mode="$4"; shift 4 rule " Step $n of 6" prose "$dotext" if [ -n "$why" ]; then echo; prose "$why"; fi if [ "$mode" = "manual" ]; then echo; echo " -> Do this yourself, then press Enter to carry on." if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r _; fi return 1 fi echo; printf " %s\n" "$@"; echo if [ "$DRYRUN" = "1" ]; then echo " (dry run -- not executed)"; return 1; fi if [ "$UNATTENDED" = "0" ]; then read -r -p " Run this step? [Y]es / [S]kip / [Q]uit " a case "$a" in [Qq]*) echo " Stopped at your request."; exit 0 ;; [Ss]*) echo " Skipped."; return 1 ;; esac fi return 0 } rule echo " ImagePullBackOff, ErrImagePull and registry rate limits" echo " Stop pulling anonymously" echo echo " Risk: low Reversible 30 minutes" echo prose 'No warranty. Use at your own risk - JbTecWiz accepts no liability. Read it before you run it, and have a backup.' rule echo if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r -p " Ready? [y/N] " go case "$go" in [Yy]*) ;; *) echo " Nothing was changed."; exit 0;; esac fi if step 1 'Confirm the message.' '' cmd 'kubectl describe pod mypod | tail -20'; then kubectl describe pod mypod | tail -20 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 1 failed. The rest of the fix may depend on it." >&2 fi fi step 2 'Anonymous pulls from Docker Hub are limited per IP address, so an entire office or cluster behind one NAT address shares one small quota.' 'This is why the problem appears suddenly on a cluster that has worked for months -- the limit is per source address, not per machine, and one busy pipeline consumes it for everyone.' manual || true if step 3 'Authenticate, which raises the limit substantially even on a free account.' '' cmd 'kubectl create secret docker-registry dockerhub \' ' --docker-server=https://index.docker.io/v1/ \' ' --docker-username=USER --docker-password=TOKEN'; then kubectl create secret docker-registry dockerhub \ --docker-server=https://index.docker.io/v1/ \ --docker-username=USER --docker-password=TOKEN if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 3 failed. The rest of the fix may depend on it." >&2 fi fi if step 4 'Attach it to the service account so every pod uses it without per-pod configuration.' '' cmd 'kubectl patch serviceaccount default -p '\''{"imagePullSecrets":[{"name":"dockerhub"}]}'\'''; then kubectl patch serviceaccount default -p '{"imagePullSecrets":[{"name":"dockerhub"}]}' if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 4 failed. The rest of the fix may depend on it." >&2 fi fi if step 5 'Better, run a pull-through cache or mirror so images are fetched once for the whole cluster.' '' cmd 'kubectl get nodes -o wide'; then kubectl get nodes -o wide if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 5 failed. The rest of the fix may depend on it." >&2 fi fi if step 6 'Set imagePullPolicy to IfNotPresent for images with fixed tags, so a restart does not pull again.' '' cmd 'kubectl get deploy myapp -o jsonpath='\''{.spec.template.spec.containers[*].imagePullPolicy}'\'''; then kubectl get deploy myapp -o jsonpath='{.spec.template.spec.containers[*].imagePullPolicy}' if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 6 failed. The rest of the fix may depend on it." >&2 fi fi rule " Confirm it worked" prose 'Pods start and the pull events show no rate limiting.' if [ "$DRYRUN" = "0" ]; then kubectl get events --sort-by=.lastTimestamp | tail -20 fi rule if [ "$failed" -gt 0 ]; then echo " Finished with $failed failed step(s)." echo " Read the full write-up at https://jbtecwiz.com/support/lnx-ctr-imagepull" else echo " Finished." fi echo prose 'To undo: Remove the secret and the service account patch.' rule