# esh-pve-nas cutover, step 1 of 5 — quiesce esh-docker-vm's hard NFS mounts. # # Run: scripts/elway infra-ops@10.0.50.45 --playbook playbooks/esh-cutover-1-quiesce-docker-vm.yaml # # Why this is first and why it is not optional: /mnt/books and /mnt/backup are # `hard` NFS from CT 103 on esh-pve-nas. A hard mount does not fail when the # server goes away — it blocks forever in D-state, and the only known remedy is # rebooting THIS host. /mnt/books was deliberately left hard because calibre's # SQLite risks corruption under `soft`, so the mount option is not the fix; the # quiesce is. # # Measured 2026-08-18: exactly one container binds these paths # (calibre-web-automated -> /mnt/books/calibre/{ingest,calibre_library}) and # /mnt/backup has no container consumers at all. The blast radius is one service, # not the seventeen containers on this host. # # Reversed by playbooks/esh-cutover-5-restore.yaml. steps: # No --format here: elway substitutes {{ ... }}, so Go template braces in a # shell command are a booby trap. --filter + -q avoids them entirely. - name: Stop the only container holding the NFS mounts shell: sudo -n docker stop calibre-web-automated when: "test -n \"$(sudo -n docker ps -q --filter name=^calibre-web-automated$)\"" - name: Confirm nothing else has files open under the mounts shell: | busy=$(sudo -n lsof +D /mnt/books +D /mnt/backup 2>/dev/null | tail -n +2 | wc -l) if [ "$busy" -ne 0 ]; then echo "STILL BUSY — refusing to unmount:" sudo -n lsof +D /mnt/books +D /mnt/backup 2>/dev/null | head -20 exit 1 fi echo "no open files under either mount" changed_when: "false" - name: Unmount /mnt/books shell: sudo -n umount /mnt/books when: "mountpoint -q /mnt/books" - name: Unmount /mnt/backup shell: sudo -n umount /mnt/backup when: "mountpoint -q /mnt/backup" verify: - name: Neither NFS mount remains shell: "! findmnt -t nfs,nfs4 -o TARGET | grep -qE '/mnt/(books|backup)'" changed_when: "false" - name: The other sixteen containers are still up shell: | n=$(sudo -n docker ps -q | wc -l) echo "$n containers still running" test "$n" -ge 10 changed_when: "false"