#!/usr/bin/env bash # # JbTecWiz Support Centre -- generated fix script # # Fault : mdadm array degraded, or missing after a reboot # Fix : Replace the failed member # Source: https://jbtecwiz.com/support/lnx-sto-mdadm # # Run as : Root shell # Expect : 1-8 hours depending on size # Risk : high # Reversible : NO -- read the undo note below # # WHEN THIS IS THE RIGHT FIX # The array is degraded but running. Do this promptly -- a second # failure during a rebuild loses everything on a RAID5. # # HOW TO UNDO IT # None -- the replaced disk's data is rebuilt from parity. # # Walks the fix one step at a time and asks before each. Steps with no # command are yours to do -- it prints those and waits. DRYRUN=1 prints # without executing; UNATTENDED=1 does not ask. # # -------------------------------------------------------------------- # NO WARRANTY - USE AT YOUR OWN RISK # # This script is provided by JbTecWiz as-is and with no warranty of any # kind, express or implied. You run it entirely at your own risk. # # JbTecWiz accepts no liability for any loss or damage arising from its # use, including but not limited to data loss, downtime, or configuration # changes that turn out to be wrong for your system. # # You are responsible for reading this script before running it, for # satisfying yourself that it suits the machine in front of you, and for # having a working backup first. Some steps cannot be undone. # -------------------------------------------------------------------- set -uo pipefail DRYRUN="${DRYRUN:-0}" UNATTENDED="${UNATTENDED:-0}" failed=0 # - THIS SCRIPT WILL NOT RUN UNTIL YOU EDIT IT - cat >&2 <<'EOF' This script needs editing before it can run. Replace each of these with a real value: example.com (a placeholder domain) Then delete this block near the top of the script. EOF exit 2 if [ "$(id -u)" -ne 0 ]; then echo " This fix is documented as needing root. Re-run with sudo." >&2 exit 3 fi rule() { printf "\n%s\n" "$(printf '-%.0s' $(seq 1 70))"; if [ $# -gt 0 ]; then echo "$1"; fi; } prose() { echo "$1" | fold -s -w 74 | sed "s/^/ /"; } # Returns 0 when the caller should run the command, 1 when it should not. # A manual step always returns 1 -- there is nothing for the caller to run. step() { # step [command lines...] local n="$1" dotext="$2" why="$3" mode="$4"; shift 4 rule " Step $n of 6" prose "$dotext" if [ -n "$why" ]; then echo; prose "$why"; fi if [ "$mode" = "manual" ]; then echo; echo " -> Do this yourself, then press Enter to carry on." if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r _; fi return 1 fi echo; printf " %s\n" "$@"; echo if [ "$DRYRUN" = "1" ]; then echo " (dry run -- not executed)"; return 1; fi if [ "$UNATTENDED" = "0" ]; then read -r -p " Run this step? [Y]es / [S]kip / [Q]uit " a case "$a" in [Qq]*) echo " Stopped at your request."; exit 0 ;; [Ss]*) echo " Skipped."; return 1 ;; esac fi return 0 } rule echo " mdadm array degraded, or missing after a reboot" echo " Replace the failed member" echo echo " Risk: high NOT REVERSIBLE 1-8 hours depending on size" echo prose 'No warranty. Use at your own risk - JbTecWiz accepts no liability. Read it before you run it, and have a backup.' rule echo if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r -p " Ready? [y/N] " go case "$go" in [Yy]*) ;; *) echo " Nothing was changed."; exit 0;; esac fi if step 1 'Read the state and identify the failed device.' '' cmd 'cat /proc/mdstat' 'sudo mdadm --detail /dev/md0'; then cat /proc/mdstat sudo mdadm --detail /dev/md0 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 1 failed. The rest of the fix may depend on it." >&2 fi fi if step 2 'Check the health of every remaining member before rebuilding. A rebuild reads every sector of every disk and is exactly when a second marginal drive fails.' 'This is the step that separates a routine replacement from losing the array. If a second disk is showing pending sectors, back up before rebuilding, not after.' cmd 'for d in /dev/sd{a,b,c,d}; do echo "== $d"; sudo smartctl -H $d; done'; then for d in /dev/sd{a,b,c,d}; do echo "== $d"; sudo smartctl -H $d; done if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 2 failed. The rest of the fix may depend on it." >&2 fi fi if step 3 'Remove the failed member.' '' cmd 'sudo mdadm --manage /dev/md0 --fail /dev/sdc1 --remove /dev/sdc1'; then sudo mdadm --manage /dev/md0 --fail /dev/sdc1 --remove /dev/sdc1 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 3 failed. The rest of the fix may depend on it." >&2 fi fi if step 4 'Partition the replacement to match, then add it.' '' cmd 'sudo sfdisk -d /dev/sda | sudo sfdisk /dev/sdc' 'sudo mdadm --manage /dev/md0 --add /dev/sdc1'; then sudo sfdisk -d /dev/sda | sudo sfdisk /dev/sdc sudo mdadm --manage /dev/md0 --add /dev/sdc1 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 4 failed. The rest of the fix may depend on it." >&2 fi fi if step 5 'Watch the rebuild. Do not reboot during it.' '' cmd 'watch -n5 cat /proc/mdstat'; then watch -n5 cat /proc/mdstat if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 5 failed. The rest of the fix may depend on it." >&2 fi fi if step 6 'Set up email alerting so the next failure is noticed on the day it happens.' '' cmd 'sudo sed -i '\''s/^MAILADDR.*/MAILADDR admin@example.com/'\'' /etc/mdadm/mdadm.conf' 'sudo mdadm --monitor --scan --test --oneshot'; then sudo sed -i 's/^MAILADDR.*/MAILADDR admin@example.com/' /etc/mdadm/mdadm.conf sudo mdadm --monitor --scan --test --oneshot if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 6 failed. The rest of the fix may depend on it." >&2 fi fi rule " Confirm it worked" prose 'mdstat shows all members [UU] with no rebuild in progress.' if [ "$DRYRUN" = "0" ]; then cat /proc/mdstat; sudo mdadm --detail /dev/md0 | grep -E 'State|Active|Working|Failed' fi rule if [ "$failed" -gt 0 ]; then echo " Finished with $failed failed step(s)." echo " Read the full write-up at https://jbtecwiz.com/support/lnx-sto-mdadm" else echo " Finished." fi echo prose 'To undo: None -- the replaced disk'\''s data is rebuilt from parity.' rule