Files
esh-pfi-infrastructure/scripts/nh3-dev-development-backup.sh
T
vh 902e16630f feat(dev-backup): add daily and weekly retention (48 hourly + 30 daily + 12 weekly)
Prime's ruling 2026-10-02. retention.py picks the snapshots to delete:
the newest 48, plus the newest of each of the last 30 days and of each of
the last 12 ISO weeks, counting only days and weeks that have snapshots.
Names that are not exactly YYYY-MM-DD_HHMM are never selected, and the
NAS side refuses any path outside that pattern. The unit fails unless
the number kept equals the number expected. Live run: deleted 1, 0
errors, 48 kept as expected.
2026-10-02 07:42:45 -07:00

71 lines
3.8 KiB
Bash
Executable File

#!/usr/bin/env bash
# Hourly off-box snapshot of ~/development -> nh3-nas via rsync --link-dest
# hardlink snapshots. Penance for the 2026-07-12 soong-lab clobber: uncommitted
# dev work now has an hourly, versioned, off-box safety net. Secrets + heavy
# reconstructable dirs are excluded. Snapshots are timestamped dirs on the NAS;
# unchanged files hardlink to the previous snapshot (space-efficient). Retention:
# 48 hourly + 30 daily + 12 weekly (retention.py; dailies and weeklies added 2026-10-02).
set -uo pipefail
SRC="$HOME/development/"
DEST_HOST="nh3-nas"
DEST_BASE="/volume1/Backup/nh3-dev-development"
STAMP="$(date +%Y-%m-%d_%H%M)"
LOG="$HOME/.config/dev-backup/dev-backup.log"
exec >>"$LOG" 2>&1
echo "=== $(date -Is) snapshot $STAMP start ==="
# previous snapshot for hardlink dedup
PREV="$(ssh -o ConnectTimeout=15 -o BatchMode=yes "$DEST_HOST" "ls -1d $DEST_BASE/20* 2>/dev/null | sort | tail -1" || true)"
LINKDEST=()
[ -n "$PREV" ] && LINKDEST=(--link-dest="$PREV")
echo "link-dest: ${PREV:-<none, first full snapshot>}"
ssh -o BatchMode=yes "$DEST_HOST" "mkdir -p '$DEST_BASE/$STAMP'"
rsync -a --delete --numeric-ids \
--exclude='node_modules/' --exclude='.venv/' --exclude='venv/' --exclude='__pycache__/' \
--exclude='.pytest_cache/' --exclude='.mypy_cache/' --exclude='.ruff_cache/' --exclude='.cache/' \
--exclude='dist/' --exclude='build/' --exclude='.next/' --exclude='target/' --exclude='*.pyc' \
--exclude='.env' --exclude='.env.*' --exclude='*.pem' --exclude='*.key' --exclude='id_*' \
--exclude='*.sqlite' --exclude='*.sqlite3' --exclude='*.db-wal' --exclude='*.db-shm' \
"${LINKDEST[@]}" \
"$SRC" "$DEST_HOST:$DEST_BASE/$STAMP/"
RC=$?
echo "rsync rc=$RC"
# rc 0 = ok; rc 24 = some files vanished mid-transfer (benign for a live tree)
if [ "$RC" -eq 0 ] || [ "$RC" -eq 24 ]; then
ssh -o BatchMode=yes "$DEST_HOST" "ln -sfn '$DEST_BASE/$STAMP' '$DEST_BASE/latest'"
# retention (Prime, 2026-10-02): newest 48 hourly + newest of each of the last 30 days + newest of
# each of the last 12 ISO weeks. retention.py (next to this script) picks what to delete; anything
# whose name is not exactly YYYY-MM-DD_HHMM is never selected, and the remote side refuses any path
# outside $DEST_BASE/20??-??-??_???? as a second guard.
# chmod BEFORE rm: rsync -a copies a read-only source dir (mode 0555) as read-only, and rm cannot
# unlink inside it. Without the chmod this prune failed every run from 2026-07-18 to 2026-10-01 and
# left 1,740 husk dirs behind (removed 2026-10-01), while the run still logged OK. Errors are COUNTED,
# not dumped (one failing run logged ~400 MB), and a failed prune fails the unit.
ALL="$(ssh -o BatchMode=yes "$DEST_HOST" "ls -1d $DEST_BASE/20* 2>/dev/null")"
DELETE="$(printf '%s\n' "$ALL" | python3 "$(dirname "$0")/retention.py")"
EXPECT="$(printf '%s\n' "$ALL" | python3 "$(dirname "$0")/retention.py" --keep-count)"
NDEL="$(printf '%s' "$DELETE" | grep -c . || true)"
PRUNE_ERR="$(printf '%s\n' "$DELETE" | grep . | ssh -o BatchMode=yes "$DEST_HOST" "while read -r d; do
case \"\$d\" in
$DEST_BASE/20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]_[0-9][0-9][0-9][0-9]) chmod -R u+w \"\$d\" 2>&1 && rm -rf \"\$d\" 2>&1 || echo \"FAILED \$d\" ;;
*) echo \"REFUSED \$d\" ;;
esac
done" | wc -l)"
KEPT="$(ssh -o BatchMode=yes "$DEST_HOST" "ls -1d $DEST_BASE/20* 2>/dev/null | wc -l")"
echo "retention prune: deleted ${NDEL:-?}, ${PRUNE_ERR:-?} error lines, ${KEPT:-?} snapshots on the NAS (expected 0 errors and ${EXPECT:-?} kept)"
if [ "${PRUNE_ERR:-x}" = 0 ] && [ -n "${EXPECT:-}" ] && [ "${KEPT:-x}" = "$EXPECT" ]; then
echo "=== $(date -Is) snapshot $STAMP OK (rc=$RC) ==="
else
echo "=== $(date -Is) snapshot $STAMP taken (rc=$RC) but RETENTION FAILED ==="
exit 3
fi
else
echo "=== $(date -Is) snapshot $STAMP FAILED rc=$RC — keeping partial for inspection ==="
exit 1
fi