#!/usr/bin/env bash # # JbTecWiz Support Centre -- generated fix script # # Fault : Load average is high but the CPU looks idle # Fix : Separate CPU load from I/O wait # Source: https://jbtecwiz.com/support/lnx-high-load # # Run as : Shell # Expect : 30 minutes # Risk : low # Reversible : yes # # WHEN THIS IS THE RIGHT FIX # Load is high and CPU is not. The load average alone cannot tell you # which, so start by splitting them. # # HOW TO UNDO IT # None -- diagnostic. # # Walks the fix one step at a time and asks before each. Steps with no # command are yours to do -- it prints those and waits. DRYRUN=1 prints # without executing; UNATTENDED=1 does not ask. # # -------------------------------------------------------------------- # NO WARRANTY - USE AT YOUR OWN RISK # # This script is provided by JbTecWiz as-is and with no warranty of any # kind, express or implied. You run it entirely at your own risk. # # JbTecWiz accepts no liability for any loss or damage arising from its # use, including but not limited to data loss, downtime, or configuration # changes that turn out to be wrong for your system. # # You are responsible for reading this script before running it, for # satisfying yourself that it suits the machine in front of you, and for # having a working backup first. Some steps cannot be undone. # -------------------------------------------------------------------- set -uo pipefail DRYRUN="${DRYRUN:-0}" UNATTENDED="${UNATTENDED:-0}" failed=0 rule() { printf "\n%s\n" "$(printf '-%.0s' $(seq 1 70))"; if [ $# -gt 0 ]; then echo "$1"; fi; } prose() { echo "$1" | fold -s -w 74 | sed "s/^/ /"; } # Returns 0 when the caller should run the command, 1 when it should not. # A manual step always returns 1 -- there is nothing for the caller to run. step() { # step [command lines...] local n="$1" dotext="$2" why="$3" mode="$4"; shift 4 rule " Step $n of 6" prose "$dotext" if [ -n "$why" ]; then echo; prose "$why"; fi if [ "$mode" = "manual" ]; then echo; echo " -> Do this yourself, then press Enter to carry on." if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r _; fi return 1 fi echo; printf " %s\n" "$@"; echo if [ "$DRYRUN" = "1" ]; then echo " (dry run -- not executed)"; return 1; fi if [ "$UNATTENDED" = "0" ]; then read -r -p " Run this step? [Y]es / [S]kip / [Q]uit " a case "$a" in [Qq]*) echo " Stopped at your request."; exit 0 ;; [Ss]*) echo " Skipped."; return 1 ;; esac fi return 0 } rule echo " Load average is high but the CPU looks idle" echo " Separate CPU load from I/O wait" echo echo " Risk: low Reversible 30 minutes" echo prose 'No warranty. Use at your own risk - JbTecWiz accepts no liability. Read it before you run it, and have a backup.' rule echo if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r -p " Ready? [y/N] " go case "$go" in [Yy]*) ;; *) echo " Nothing was changed."; exit 0;; esac fi if step 1 'Look at the CPU breakdown. The %wa column is I/O wait.' 'In vmstat, the '\''b'\'' column counts processes blocked on I/O. If b is high and '\''r'\'' is low, no amount of CPU would help.' cmd 'top -bn2 -d1 | grep -E '\''^%Cpu'\'' | tail -1' 'vmstat 1 5'; then top -bn2 -d1 | grep -E '^%Cpu' | tail -1 vmstat 1 5 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 1 failed. The rest of the fix may depend on it." >&2 fi fi if step 2 'List the processes actually in uninterruptible sleep.' 'The wchan column names the kernel function they are stuck in -- nfs_wait, io_schedule and similar point straight at the subsystem.' cmd 'ps -eo state,pid,user,wchan:30,cmd | awk '\''$1 ~ /D/'\'''; then ps -eo state,pid,user,wchan:30,cmd | awk '$1 ~ /D/' if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 2 failed. The rest of the fix may depend on it." >&2 fi fi if step 3 'Find which device is saturated.' '%util near 100 with a high await means that device is the bottleneck. A high await with low %util usually means the storage behind it, not the disk itself.' cmd 'iostat -xz 1 5'; then iostat -xz 1 5 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 3 failed. The rest of the fix may depend on it." >&2 fi fi if step 4 'Attribute the I/O to a process.' '' cmd 'sudo iotop -oPa -n 5'; then sudo iotop -oPa -n 5 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 4 failed. The rest of the fix may depend on it." >&2 fi fi if step 5 'Check for hung NFS mounts, a very common cause of unkillable D-state processes.' '' cmd 'mount -t nfs,nfs4' 'sudo dmesg -T | grep -i '\''nfs.*not responding'\'' | tail'; then mount -t nfs,nfs4 sudo dmesg -T | grep -i 'nfs.*not responding' | tail if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 5 failed. The rest of the fix may depend on it." >&2 fi fi if step 6 'Look for kernel hung-task warnings, which name the stuck process and its stack.' '' cmd 'sudo dmesg -T | grep -iA10 '\''hung_task\|blocked for more than'\'' | tail -40'; then sudo dmesg -T | grep -iA10 'hung_task\|blocked for more than' | tail -40 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 6 failed. The rest of the fix may depend on it." >&2 fi fi rule " Confirm it worked" prose 'The load figure tracks the CPU run queue again once the I/O source is dealt with.' if [ "$DRYRUN" = "0" ]; then uptime; vmstat 1 3 fi rule if [ "$failed" -gt 0 ]; then echo " Finished with $failed failed step(s)." echo " Read the full write-up at https://jbtecwiz.com/support/lnx-high-load" else echo " Finished." fi echo prose 'To undo: None -- diagnostic.' rule