#!/usr/bin/env bash # # JbTecWiz Support Centre -- generated fix script # # Fault : "server reached pm.max_children setting" -- the site stalls under load # Fix : Find what is holding the workers # Source: https://jbtecwiz.com/support/lnx-web-fpm-children # # Run as : Root shell # Expect : 40 minutes # Risk : low # Reversible : yes # # WHEN THIS IS THE RIGHT FIX # Before adding workers. More workers running the same slow query just # moves the queue into the database. # # HOW TO UNDO IT # Nothing was changed on the server itself. # # Walks the fix one step at a time and asks before each. Steps with no # command are yours to do -- it prints those and waits. DRYRUN=1 prints # without executing; UNATTENDED=1 does not ask. # # -------------------------------------------------------------------- # NO WARRANTY - USE AT YOUR OWN RISK # # This script is provided by JbTecWiz as-is and with no warranty of any # kind, express or implied. You run it entirely at your own risk. # # JbTecWiz accepts no liability for any loss or damage arising from its # use, including but not limited to data loss, downtime, or configuration # changes that turn out to be wrong for your system. # # You are responsible for reading this script before running it, for # satisfying yourself that it suits the machine in front of you, and for # having a working backup first. Some steps cannot be undone. # -------------------------------------------------------------------- set -uo pipefail DRYRUN="${DRYRUN:-0}" UNATTENDED="${UNATTENDED:-0}" failed=0 if [ "$(id -u)" -ne 0 ]; then echo " This fix is documented as needing root. Re-run with sudo." >&2 exit 3 fi rule() { printf "\n%s\n" "$(printf '-%.0s' $(seq 1 70))"; if [ $# -gt 0 ]; then echo "$1"; fi; } prose() { echo "$1" | fold -s -w 74 | sed "s/^/ /"; } # Returns 0 when the caller should run the command, 1 when it should not. # A manual step always returns 1 -- there is nothing for the caller to run. step() { # step [command lines...] local n="$1" dotext="$2" why="$3" mode="$4"; shift 4 rule " Step $n of 5" prose "$dotext" if [ -n "$why" ]; then echo; prose "$why"; fi if [ "$mode" = "manual" ]; then echo; echo " -> Do this yourself, then press Enter to carry on." if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r _; fi return 1 fi echo; printf " %s\n" "$@"; echo if [ "$DRYRUN" = "1" ]; then echo " (dry run -- not executed)"; return 1; fi if [ "$UNATTENDED" = "0" ]; then read -r -p " Run this step? [Y]es / [S]kip / [Q]uit " a case "$a" in [Qq]*) echo " Stopped at your request."; exit 0 ;; [Ss]*) echo " Skipped."; return 1 ;; esac fi return 0 } rule echo " "server reached pm.max_children setting" -- the site stalls under load" echo " Find what is holding the workers" echo echo " Risk: low Reversible 40 minutes" echo prose 'No warranty. Use at your own risk - JbTecWiz accepts no liability. Read it before you run it, and have a backup.' rule echo if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r -p " Ready? [y/N] " go case "$go" in [Yy]*) ;; *) echo " Nothing was changed."; exit 0;; esac fi if step 1 'Turn on the status page and watch it under load.' '' cmd 'grep -n '\''pm.status_path'\'' /etc/php/8.2/fpm/pool.d/www.conf' 'curl -s http://localhost/fpm-status?full | head -40'; then grep -n 'pm.status_path' /etc/php/8.2/fpm/pool.d/www.conf curl -s http://localhost/fpm-status?full | head -40 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 1 failed. The rest of the fix may depend on it." >&2 fi fi if step 2 'Enable the slow log if it is not on, then read what it catches.' 'The slow log records a full stack trace of what each stuck request was doing at the moment it passed the threshold. It usually names one query or one HTTP call, and that is the actual fault.' cmd 'sudo tail -50 /var/log/php-fpm-slow.log'; then sudo tail -50 /var/log/php-fpm-slow.log if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 2 failed. The rest of the fix may depend on it." >&2 fi fi if step 3 'Check the database for queries that are running long.' '' cmd 'mysql -e "SELECT id,user,host,db,command,time,state,LEFT(info,80) FROM information_schema.processlist WHERE time > 2 ORDER BY time DESC\G"'; then mysql -e "SELECT id,user,host,db,command,time,state,LEFT(info,80) FROM information_schema.processlist WHERE time > 2 ORDER BY time DESC\G" if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 3 failed. The rest of the fix may depend on it." >&2 fi fi step 4 'Add a timeout to any outbound HTTP call in the application. A request to a third party with no timeout holds a worker until the kernel gives up, which can be minutes.' '' manual || true step 5 'Fix the slow query -- usually a missing index -- before touching the pool configuration.' '' manual || true rule " Confirm it worked" prose 'The slow log is quiet and active workers stay well below the maximum at peak.' rule if [ "$failed" -gt 0 ]; then echo " Finished with $failed failed step(s)." echo " Read the full write-up at https://jbtecwiz.com/support/lnx-web-fpm-children" else echo " Finished." fi echo prose 'To undo: Nothing was changed on the server itself.' rule