feat(dev-backup): add daily and weekly retention (48 hourly + 30 daily + 12 weekly)
Prime's ruling 2026-10-02. retention.py picks the snapshots to delete: the newest 48, plus the newest of each of the last 30 days and of each of the last 12 ISO weeks, counting only days and weeks that have snapshots. Names that are not exactly YYYY-MM-DD_HHMM are never selected, and the NAS side refuses any path outside that pattern. The unit fails unless the number kept equals the number expected. Live run: deleted 1, 0 errors, 48 kept as expected.
This commit is contained in:
@@ -4,7 +4,7 @@
|
||||
# dev work now has an hourly, versioned, off-box safety net. Secrets + heavy
|
||||
# reconstructable dirs are excluded. Snapshots are timestamped dirs on the NAS;
|
||||
# unchanged files hardlink to the previous snapshot (space-efficient). Retention:
|
||||
# newest 48 hourly snapshots.
|
||||
# 48 hourly + 30 daily + 12 weekly (retention.py; dailies and weeklies added 2026-10-02).
|
||||
set -uo pipefail
|
||||
|
||||
SRC="$HOME/development/"
|
||||
@@ -38,16 +38,27 @@ echo "rsync rc=$RC"
|
||||
# rc 0 = ok; rc 24 = some files vanished mid-transfer (benign for a live tree)
|
||||
if [ "$RC" -eq 0 ] || [ "$RC" -eq 24 ]; then
|
||||
ssh -o BatchMode=yes "$DEST_HOST" "ln -sfn '$DEST_BASE/$STAMP' '$DEST_BASE/latest'"
|
||||
# retention: keep newest 48 hourly snapshots.
|
||||
# retention (Prime, 2026-10-02): newest 48 hourly + newest of each of the last 30 days + newest of
|
||||
# each of the last 12 ISO weeks. retention.py (next to this script) picks what to delete; anything
|
||||
# whose name is not exactly YYYY-MM-DD_HHMM is never selected, and the remote side refuses any path
|
||||
# outside $DEST_BASE/20??-??-??_???? as a second guard.
|
||||
# chmod BEFORE rm: rsync -a copies a read-only source dir (mode 0555) as read-only, and rm cannot
|
||||
# unlink inside it. Without the chmod this prune failed every run from 2026-07-18 to 2026-10-01 and
|
||||
# left 1,740 husk dirs behind (removed 2026-10-01), while the run still logged OK. Errors are COUNTED,
|
||||
# not dumped (one failing run logged ~400 MB), and a failed prune now fails the unit.
|
||||
PRUNE_LIST="ls -1d $DEST_BASE/20* 2>/dev/null | sort | head -n -48"
|
||||
PRUNE_ERR="$(ssh -o BatchMode=yes "$DEST_HOST" "$PRUNE_LIST | xargs -r chmod -R u+w 2>&1; $PRUNE_LIST | xargs -r rm -rf 2>&1" | wc -l)"
|
||||
# not dumped (one failing run logged ~400 MB), and a failed prune fails the unit.
|
||||
ALL="$(ssh -o BatchMode=yes "$DEST_HOST" "ls -1d $DEST_BASE/20* 2>/dev/null")"
|
||||
DELETE="$(printf '%s\n' "$ALL" | python3 "$(dirname "$0")/retention.py")"
|
||||
EXPECT="$(printf '%s\n' "$ALL" | python3 "$(dirname "$0")/retention.py" --keep-count)"
|
||||
NDEL="$(printf '%s' "$DELETE" | grep -c . || true)"
|
||||
PRUNE_ERR="$(printf '%s\n' "$DELETE" | grep . | ssh -o BatchMode=yes "$DEST_HOST" "while read -r d; do
|
||||
case \"\$d\" in
|
||||
$DEST_BASE/20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]_[0-9][0-9][0-9][0-9]) chmod -R u+w \"\$d\" 2>&1 && rm -rf \"\$d\" 2>&1 || echo \"FAILED \$d\" ;;
|
||||
*) echo \"REFUSED \$d\" ;;
|
||||
esac
|
||||
done" | wc -l)"
|
||||
KEPT="$(ssh -o BatchMode=yes "$DEST_HOST" "ls -1d $DEST_BASE/20* 2>/dev/null | wc -l")"
|
||||
echo "retention prune: ${PRUNE_ERR:-?} error lines, ${KEPT:-?} snapshots on the NAS (expected 0 and <=48)"
|
||||
if [ "${PRUNE_ERR:-x}" = 0 ] && [ "${KEPT:-999}" -le 48 ] 2>/dev/null; then
|
||||
echo "retention prune: deleted ${NDEL:-?}, ${PRUNE_ERR:-?} error lines, ${KEPT:-?} snapshots on the NAS (expected 0 errors and ${EXPECT:-?} kept)"
|
||||
if [ "${PRUNE_ERR:-x}" = 0 ] && [ -n "${EXPECT:-}" ] && [ "${KEPT:-x}" = "$EXPECT" ]; then
|
||||
echo "=== $(date -Is) snapshot $STAMP OK (rc=$RC) ==="
|
||||
else
|
||||
echo "=== $(date -Is) snapshot $STAMP taken (rc=$RC) but RETENTION FAILED ==="
|
||||
|
||||
Reference in New Issue
Block a user