Files
esh-pfi-infrastructure/scripts/nh3-dev-development-backup.sh
T
vh de11e00dfb fix(dev-backup): chmod before prune so retention actually deletes; fail the unit on a bad prune
rsync -a copies a read-only source dir (0555) as read-only, so the hourly
prune's rm -rf could not unlink inside it. From 2026-07-18 every pruned
snapshot was left as a 22-entry husk while the run still logged OK. The
1,740 husks on nh3-nas were removed (0 errors; 48 full snapshots kept).

The prune now runs chmod -R u+w before rm -rf, logs its error count and
the number of snapshots kept, and exits 3 on a failed prune (exit 1 on a
failed rsync) so the systemd unit shows failed instead of passing.
2026-10-01 05:59:07 -07:00

60 lines
3.0 KiB
Bash
Executable File

#!/usr/bin/env bash
# Hourly off-box snapshot of ~/development -> nh3-nas via rsync --link-dest
# hardlink snapshots. Penance for the 2026-07-12 soong-lab clobber: uncommitted
# dev work now has an hourly, versioned, off-box safety net. Secrets + heavy
# reconstructable dirs are excluded. Snapshots are timestamped dirs on the NAS;
# unchanged files hardlink to the previous snapshot (space-efficient). Retention:
# newest 48 hourly snapshots.
set -uo pipefail
SRC="$HOME/development/"
DEST_HOST="nh3-nas"
DEST_BASE="/volume1/Backup/nh3-dev-development"
STAMP="$(date +%Y-%m-%d_%H%M)"
LOG="$HOME/.config/dev-backup/dev-backup.log"
exec >>"$LOG" 2>&1
echo "=== $(date -Is) snapshot $STAMP start ==="
# previous snapshot for hardlink dedup
PREV="$(ssh -o ConnectTimeout=15 -o BatchMode=yes "$DEST_HOST" "ls -1d $DEST_BASE/20* 2>/dev/null | sort | tail -1" || true)"
LINKDEST=()
[ -n "$PREV" ] && LINKDEST=(--link-dest="$PREV")
echo "link-dest: ${PREV:-<none, first full snapshot>}"
ssh -o BatchMode=yes "$DEST_HOST" "mkdir -p '$DEST_BASE/$STAMP'"
rsync -a --delete --numeric-ids \
--exclude='node_modules/' --exclude='.venv/' --exclude='venv/' --exclude='__pycache__/' \
--exclude='.pytest_cache/' --exclude='.mypy_cache/' --exclude='.ruff_cache/' --exclude='.cache/' \
--exclude='dist/' --exclude='build/' --exclude='.next/' --exclude='target/' --exclude='*.pyc' \
--exclude='.env' --exclude='.env.*' --exclude='*.pem' --exclude='*.key' --exclude='id_*' \
--exclude='*.sqlite' --exclude='*.sqlite3' --exclude='*.db-wal' --exclude='*.db-shm' \
"${LINKDEST[@]}" \
"$SRC" "$DEST_HOST:$DEST_BASE/$STAMP/"
RC=$?
echo "rsync rc=$RC"
# rc 0 = ok; rc 24 = some files vanished mid-transfer (benign for a live tree)
if [ "$RC" -eq 0 ] || [ "$RC" -eq 24 ]; then
ssh -o BatchMode=yes "$DEST_HOST" "ln -sfn '$DEST_BASE/$STAMP' '$DEST_BASE/latest'"
# retention: keep newest 48 hourly snapshots.
# chmod BEFORE rm: rsync -a copies a read-only source dir (mode 0555) as read-only, and rm cannot
# unlink inside it. Without the chmod this prune failed every run from 2026-07-18 to 2026-10-01 and
# left 1,740 husk dirs behind (removed 2026-10-01), while the run still logged OK. Errors are COUNTED,
# not dumped (one failing run logged ~400 MB), and a failed prune now fails the unit.
PRUNE_LIST="ls -1d $DEST_BASE/20* 2>/dev/null | sort | head -n -48"
PRUNE_ERR="$(ssh -o BatchMode=yes "$DEST_HOST" "$PRUNE_LIST | xargs -r chmod -R u+w 2>&1; $PRUNE_LIST | xargs -r rm -rf 2>&1" | wc -l)"
KEPT="$(ssh -o BatchMode=yes "$DEST_HOST" "ls -1d $DEST_BASE/20* 2>/dev/null | wc -l")"
echo "retention prune: ${PRUNE_ERR:-?} error lines, ${KEPT:-?} snapshots on the NAS (expected 0 and <=48)"
if [ "${PRUNE_ERR:-x}" = 0 ] && [ "${KEPT:-999}" -le 48 ] 2>/dev/null; then
echo "=== $(date -Is) snapshot $STAMP OK (rc=$RC) ==="
else
echo "=== $(date -Is) snapshot $STAMP taken (rc=$RC) but RETENTION FAILED ==="
exit 3
fi
else
echo "=== $(date -Is) snapshot $STAMP FAILED rc=$RC — keeping partial for inspection ==="
exit 1
fi