#!/usr/bin/env bash # # JbTecWiz Support Centre -- generated fix script # # Fault : "Read-only file system" -- the filesystem remounted itself read-only # Fix : Establish whether the disk is failing before touching the filesystem # Source: https://jbtecwiz.com/support/lnx-readonly-fs # # Run as : Shell as root # Expect : 45 minutes # Risk : low # Reversible : yes # # WHEN THIS IS THE RIGHT FIX # Any read-only remount. Do this first -- running a repair on a dying # disk can finish it off. # # HOW TO UNDO IT # None -- this is diagnostic. # # Walks the fix one step at a time and asks before each. Steps with no # command are yours to do -- it prints those and waits. DRYRUN=1 prints # without executing; UNATTENDED=1 does not ask. # # -------------------------------------------------------------------- # NO WARRANTY - USE AT YOUR OWN RISK # # This script is provided by JbTecWiz as-is and with no warranty of any # kind, express or implied. You run it entirely at your own risk. # # JbTecWiz accepts no liability for any loss or damage arising from its # use, including but not limited to data loss, downtime, or configuration # changes that turn out to be wrong for your system. # # You are responsible for reading this script before running it, for # satisfying yourself that it suits the machine in front of you, and for # having a working backup first. Some steps cannot be undone. # -------------------------------------------------------------------- set -uo pipefail DRYRUN="${DRYRUN:-0}" UNATTENDED="${UNATTENDED:-0}" failed=0 if [ "$(id -u)" -ne 0 ]; then echo " This fix is documented as needing root. Re-run with sudo." >&2 exit 3 fi rule() { printf "\n%s\n" "$(printf '-%.0s' $(seq 1 70))"; if [ $# -gt 0 ]; then echo "$1"; fi; } prose() { echo "$1" | fold -s -w 74 | sed "s/^/ /"; } # Returns 0 when the caller should run the command, 1 when it should not. # A manual step always returns 1 -- there is nothing for the caller to run. step() { # step [command lines...] local n="$1" dotext="$2" why="$3" mode="$4"; shift 4 rule " Step $n of 5" prose "$dotext" if [ -n "$why" ]; then echo; prose "$why"; fi if [ "$mode" = "manual" ]; then echo; echo " -> Do this yourself, then press Enter to carry on." if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r _; fi return 1 fi echo; printf " %s\n" "$@"; echo if [ "$DRYRUN" = "1" ]; then echo " (dry run -- not executed)"; return 1; fi if [ "$UNATTENDED" = "0" ]; then read -r -p " Run this step? [Y]es / [S]kip / [Q]uit " a case "$a" in [Qq]*) echo " Stopped at your request."; exit 0 ;; [Ss]*) echo " Skipped."; return 1 ;; esac fi return 0 } rule echo " "Read-only file system" -- the filesystem remounted itself read-only" echo " Establish whether the disk is failing before touching the filesystem" echo echo " Risk: low Reversible 45 minutes" echo prose 'No warranty. Use at your own risk - JbTecWiz accepts no liability. Read it before you run it, and have a backup.' rule echo if [ "$UNATTENDED" = "0" ] && [ "$DRYRUN" = "0" ]; then read -r -p " Ready? [y/N] " go case "$go" in [Yy]*) ;; *) echo " Nothing was changed."; exit 0;; esac fi if step 1 'Read the kernel messages around the event.' '' cmd 'sudo dmesg -T | grep -iE '\''error|i/o|remount|ext4-fs|xfs|reset|medium'\'' | tail -40'; then sudo dmesg -T | grep -iE 'error|i/o|remount|ext4-fs|xfs|reset|medium' | tail -40 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 1 failed. The rest of the fix may depend on it." >&2 fi fi if step 2 'Check the drive'\''s own health.' 'Reallocated_Sector_Ct and Current_Pending_Sector climbing are the disk telling you it is dying. Repairing a filesystem on top of that buys hours, not days.' cmd 'sudo smartctl -a /dev/sda | grep -iE '\''result|reallocated|pending|uncorrect|wear|health'\'''; then sudo smartctl -a /dev/sda | grep -iE 'result|reallocated|pending|uncorrect|wear|health' if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 2 failed. The rest of the fix may depend on it." >&2 fi fi if step 3 'If it is a VM, check the hypervisor and storage path -- a datastore that briefly disconnected produces exactly this.' '' cmd 'sudo dmesg -T | grep -iE '\''sd [0-9]|scsi|iscsi|virtio|path'\'' | tail -30'; then sudo dmesg -T | grep -iE 'sd [0-9]|scsi|iscsi|virtio|path' | tail -30 if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 3 failed. The rest of the fix may depend on it." >&2 fi fi if step 4 'If the disk is failing, take an image now, before repairing anything.' 'ddrescue copies the good blocks first and comes back for the bad ones, which is the opposite of what dd does and the reason it is the right tool here.' cmd 'sudo ddrescue -f -n /dev/sda /dev/sdb /root/rescue.map'; then sudo ddrescue -f -n /dev/sda /dev/sdb /root/rescue.map if [ $? -ne 0 ]; then failed=$((failed+1)) echo " Step 4 failed. The rest of the fix may depend on it." >&2 fi fi step 5 'Replace the hardware, then restore.' '' manual || true rule " Confirm it worked" prose 'SMART reports no growing defect counts and dmesg is clean under load.' rule if [ "$failed" -gt 0 ]; then echo " Finished with $failed failed step(s)." echo " Read the full write-up at https://jbtecwiz.com/support/lnx-readonly-fs" else echo " Finished." fi echo prose 'To undo: None -- this is diagnostic.' rule