#!/usr/bin/env bash # # JbTecWiz Support Centre -- generated fix script # # Fault : XFS: "Structure needs cleaning" or a filesystem that will not mount # Fix : Replay the log, then repair # Source: https://jbtecwiz.com/support/lnx-sto-xfs-repair # # Run as : Root shell, filesystem unmounted # Expect : 30-90 minutes # Risk : high # Reversible : NO -- read the undo note below # # WHEN THIS IS THE RIGHT FIX # After the checks above. Order matters here more than anywhere. # # HOW TO UNDO IT # Restore the image taken beforehand. Without it there is no way back. # # Walks the fix one step at a time and asks before each. Steps with no # command are yours to do -- it prints those and waits. DRYRUN=1 prints # without executing; UNATTENDED=1 does not ask. # # -------------------------------------------------------------------- # NO WARRANTY - USE AT YOUR OWN RISK # # This script is provided by JbTecWiz as-is and with no warranty of any # kind, express or implied. You run it entirely at your own risk. # # JbTecWiz accepts no liability for any loss or damage arising from its # use, including but not limited to data loss, downtime, or configuration # changes that turn out to be wrong for your system. # # You are responsible for reading this script before running it, for # satisfying yourself that it suits the machine in front of you, and for # having a working backup first. Some steps cannot be undone. # -------------------------------------------------------------------- set -uo pipefail DRYRUN="${DRYRUN:-0}" UNATTENDED="${UNATTENDED:-0}" failed=0 if [ "$(id -u)" -ne 0 ]; then echo " This fix is documented as needing root. Re-run with sudo." >&2 exit 3 fi rule() { printf "\n%s\n" "$(printf '-%.0s' $(seq 1 70))"; if [ $# -gt 0 ]; then echo "$1"; fi; } prose() { echo "$1" | fold -s -w 74 | sed "s/^/ /"; } # Returns 0 when the caller should run the command, 1 when it should not. # A manual step always returns 1 -- there is nothing for the caller to run. step() { # step [command lines...] local n="$1" dotext="$2" why="$3" mode="$4"; shift 4 rule " Step $n of 6" prose "$dotext" if [ -n "$why" ]; then echo; prose "$why"; fi if [ "$mode" = "manual" ]; then echo; echo " -> Do this yourself, then press Enter to carry on." if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r _; fi return 1 fi echo; printf " %s\n" "$@"; echo if [ "$DRYRUN" = "1" ]; then echo " (dry run -- not executed)"; return 1; fi if [ "$UNATTENDED" = "0" ]; then read -r -p " Run this step? [Y]es / [S]kip / [Q]uit " a case "$a" in [Qq]*) echo " Stopped at your request."; exit 0 ;; [Ss]*) echo " Skipped."; return 1 ;; esac fi return 0 } rule echo " XFS: "Structure needs cleaning" or a filesystem that will not mount" echo " Replay the log, then repair" echo echo " Risk: high NOT REVERSIBLE 30-90 minutes" echo prose 'No warranty. Use at your own risk - JbTecWiz accepts no liability. Read it before you run it, and have a backup.' rule echo if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r -p " Ready? [y/N] " go case "$go" in [Yy]*) ;; *) echo " Nothing was changed."; exit 0;; esac fi if step 1 'Try mounting first. A dirty log replays on mount, and that alone resolves many of these.' 'xfs_repair with -L destroys the log, and the log holds the most recent metadata changes. Mounting to replay it cleanly first is the difference between losing nothing and losing the last few minutes of writes.' cmd 'sudo mount /dev/sdb1 /data'; then sudo mount /dev/sdb1 /data if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 1 failed. The rest of the fix may depend on it." >&2 fi fi if step 2 'If it mounts, unmount cleanly and then run a check.' '' cmd 'sudo umount /data' 'sudo xfs_repair -n /dev/sdb1'; then sudo umount /data sudo xfs_repair -n /dev/sdb1 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 2 failed. The rest of the fix may depend on it." >&2 fi fi step 3 'The -n flag reports without changing anything. Read the output before running it for real.' '' manual || true if step 4 'Run the repair.' '' cmd 'sudo xfs_repair /dev/sdb1'; then sudo xfs_repair /dev/sdb1 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 4 failed. The rest of the fix may depend on it." >&2 fi fi if step 5 'Only if it refuses because the log is dirty and cannot be replayed, use -L. This discards the log and loses the transactions in it.' '-L is the option every search result offers first. It is a last resort, it is not reversible, and it should follow a failed mount rather than replace it.' cmd 'sudo xfs_repair -L /dev/sdb1'; then sudo xfs_repair -L /dev/sdb1 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 5 failed. The rest of the fix may depend on it." >&2 fi fi if step 6 'Mount and check what landed in lost+found.' '' cmd 'sudo mount /dev/sdb1 /data' 'ls -la /data/lost+found | head -20'; then sudo mount /dev/sdb1 /data ls -la /data/lost+found | head -20 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 6 failed. The rest of the fix may depend on it." >&2 fi fi rule " Confirm it worked" prose 'The filesystem mounts, xfs_repair -n is clean, and the application'\''s data is present and correct.' if [ "$DRYRUN" = "0" ]; then sudo umount /data && sudo xfs_repair -n /dev/sdb1 && sudo mount /dev/sdb1 /data fi rule if [ "$failed" -gt 0 ]; then echo " Finished with $failed failed step(s)." echo " Read the full write-up at https://jbtecwiz.com/support/lnx-sto-xfs-repair" else echo " Finished." fi echo prose 'To undo: Restore the image taken beforehand. Without it there is no way back.' rule