Files
esh-pfi-infrastructure/scripts/upgrade-irv-ml1-cuda.sh
T
vh c9c4262bdb scripts/upgrade-irv-ml1-cuda: drop pipe-to-head (SIGPIPE under pipefail)
Phase 1 silently aborted right after the 'cuda-drivers candidate' step
— before snapshotting, stopping containers, installing the new driver,
or rebooting. Symptom: phase2 showed driver still at 570.124.06 and
containers with multi-hour uptimes (proof they were never stopped).

Cause: `apt-cache policy cuda-drivers | head -10`. head reads its 10
lines and closes the pipe early; that delivers SIGPIPE to apt-cache,
which under `set -o pipefail` propagates as a non-zero exit, and
`set -e` immediately terminates the remote shell. The same hazard
existed in phase2's `docker ps ... | head -15`.

Fix is one-line per call site: drop the head. apt-cache policy output
for cuda-drivers is short (~10 lines), and docker ps's container list
on irv-ml1 is small enough to print fully.
2026-04-25 16:51:28 -07:00

227 lines
8.4 KiB
Bash
Executable File

#!/usr/bin/env bash
# upgrade-irv-ml1-cuda.sh — bump irv-ml1's NVIDIA driver to support
# CUDA 12.9+ (currently capped at 12.8 per driver 570.124.06).
#
# Why two phases: the upgrade involves a reboot, which kills the SSH
# session mid-flight. Splitting at the reboot lets phase 2 wait for
# the host to come back and verify the new driver took.
#
# Usage:
# scripts/upgrade-irv-ml1-cuda.sh phase1
# Stop GPU containers, add NVIDIA CUDA APT repo, install
# cuda-drivers (currently 580 series), reboot.
# (interactive — sudo password prompted once)
#
# ... wait ~2 min for irv-ml1 to come back ...
#
# scripts/upgrade-irv-ml1-cuda.sh phase2
# Verify new driver + CUDA version, restart containers,
# and print the next step (flip Kokoro to GPU variant).
#
# Implementation note: the remote work runs from a script scp'd to
# /tmp on irv-ml1, then invoked via `ssh -t host bash /tmp/script`.
# Piping a heredoc to `ssh -t host bash -s` doesn't work — `-t` needs
# a real TTY but heredoc puts stdin in pipe mode (orientation.md
# gotcha; tripped over once already).
#
# Idempotent: rerunning phase1 after the repo is already added is a
# no-op until the apt step. phase2 can be re-run safely; container
# restarts via `docker compose up -d` are idempotent.
set -euo pipefail
HOST=irv-ml1
REMOTE_SCRIPT=/tmp/upgrade-cuda-remote.sh
say() { printf '\033[1;36m▸ %s\033[0m\n' "$*"; }
ok() { printf '\033[1;32m✓ %s\033[0m\n' "$*"; }
warn() { printf '\033[1;33m⚠ %s\033[0m\n' "$*"; }
err() { printf '\033[1;31m✗ %s\033[0m\n' "$*" >&2; }
# Build the remote-side script as a local tempfile, scp it, exec it
# via `ssh -t`. The remote script takes its own phase as $1.
write_remote_script() {
local local_tmp
local_tmp=$(mktemp)
cat > "$local_tmp" <<'REMOTE'
#!/usr/bin/env bash
# Remote half of upgrade-irv-ml1-cuda.sh — runs on irv-ml1.
set -euo pipefail
CONTAINERS=(comfyui cosyvoice qwen3-tts index-tts parakeet kokoro)
remote_phase1() {
echo "=== current driver + CUDA ==="
nvidia-smi --query-gpu=driver_version --format=csv,noheader
nvidia-smi | grep "CUDA Version"
echo
echo "=== prime sudo (one password prompt for everything below) ==="
sudo -v
echo
echo "=== ensuring NVIDIA CUDA APT repo for Debian 12 is configured ==="
# Clean up any orphan file from a prior failed run (script used to
# add /etc/apt/sources.list.d/nvidia-cuda.list with its own keyring,
# which collided with the upstream-provided cuda-debian12-x86_64.list
# using cuda-archive-keyring.gpg; APT refused both with
# 'Conflicting values set for option Signed-By').
if [ -f /etc/apt/sources.list.d/nvidia-cuda.list ]; then
echo " removing orphan /etc/apt/sources.list.d/nvidia-cuda.list"
sudo rm -f /etc/apt/sources.list.d/nvidia-cuda.list /etc/apt/keyrings/nvidia-cuda.gpg
fi
# Detect an existing cuda repo entry pointing at developer.download.nvidia.com.
# If found, trust it (it's how driver 570 got installed in the first place).
if grep -RlsE 'developer\.download\.nvidia\.com/compute/cuda/repos/debian12' /etc/apt/sources.list.d/ \
>/dev/null; then
echo " cuda repo already configured — skipping add"
else
echo " no cuda repo found — adding upstream-style entry"
sudo install -d -m 0755 /usr/share/keyrings
sudo apt-get install -y curl gpg
curl -fsSL https://developer.download.nvidia.com/compute/cuda/repos/debian12/x86_64/3bf863cc.pub \
| sudo gpg --dearmor --yes -o /usr/share/keyrings/cuda-archive-keyring.gpg
echo "deb [signed-by=/usr/share/keyrings/cuda-archive-keyring.gpg] https://developer.download.nvidia.com/compute/cuda/repos/debian12/x86_64/ /" \
| sudo tee /etc/apt/sources.list.d/cuda-debian12-x86_64.list >/dev/null
fi
sudo apt-get update -qq
echo
echo "=== cuda-drivers candidate ==="
# Don't `| head` here — apt-cache policy emits enough that head
# closes the pipe early and SIGPIPEs apt-cache, which under
# set -o pipefail aborts the whole script before we get to the
# actual install + reboot. Output is short anyway.
apt-cache policy cuda-drivers
echo
echo "=== snapshotting current nvidia packages (rollback reference) ==="
dpkg -l | awk '/^ii\s+(nvidia-|cuda-)/ {print $2 "=" $3}' \
| sudo tee /tmp/nvidia-pre-upgrade.txt >/dev/null
echo "snapshot at /tmp/nvidia-pre-upgrade.txt"
echo
echo "=== stopping GPU containers ==="
for stack in "${CONTAINERS[@]}"; do
if [ -d "/opt/docker/compose/$stack" ]; then
echo " stopping $stack"
(cd "/opt/docker/compose/$stack" && docker compose stop) || echo " (no-op)"
fi
done
echo
echo "=== installing cuda-drivers (DKMS rebuild ~3-5 min) ==="
sudo DEBIAN_FRONTEND=noninteractive apt-get install -y cuda-drivers
echo
echo "=== rebooting in 5 s ==="
sleep 5
sudo systemctl reboot
}
remote_phase2() {
echo "=== new driver + CUDA ==="
nvidia-smi --query-gpu=driver_version --format=csv,noheader
nvidia-smi | grep "CUDA Version"
echo
echo "=== restarting GPU containers ==="
for stack in comfyui cosyvoice qwen3-tts index-tts parakeet; do
if [ -d "/opt/docker/compose/$stack" ]; then
echo " starting $stack"
(cd "/opt/docker/compose/$stack" && docker compose up -d) || echo " (failed — check manually)"
fi
done
sleep 8
echo
echo "=== container status ==="
# `| head -15` would SIGPIPE docker ps under pipefail; just print all.
docker ps --format "table {{.Names}}\t{{.Status}}"
}
remote_rollback() {
echo "=== rollback: reinstalling pre-upgrade nvidia packages ==="
[ -f /tmp/nvidia-pre-upgrade.txt ] || { echo "no snapshot found"; exit 1; }
sudo apt-get install -y --allow-downgrades $(tr '\n' ' ' < /tmp/nvidia-pre-upgrade.txt)
echo "=== rebooting in 5 s ==="
sleep 5
sudo systemctl reboot
}
case "${1:-}" in
phase1) remote_phase1 ;;
phase2) remote_phase2 ;;
rollback) remote_rollback ;;
*) echo "remote: usage: $0 {phase1|phase2|rollback}" >&2; exit 2 ;;
esac
REMOTE
scp -q "$local_tmp" "$HOST:$REMOTE_SCRIPT"
rm -f "$local_tmp"
ssh "$HOST" "chmod +x $REMOTE_SCRIPT"
}
phase1() {
say "Phase 1: pre-flight + install + reboot"
say "Staging remote script on $HOST"
write_remote_script
say "Running phase1 (sudo will prompt once via ssh -t)"
# No piped stdin — ssh -t's TTY is real and sudo can prompt.
ssh -t "$HOST" "$REMOTE_SCRIPT phase1"
ok "Phase 1 dispatched. Host rebooting."
say "Wait ~2 min, then run: $0 phase2"
}
phase2() {
say "Phase 2: wait for host, verify, restart containers"
say "waiting for $HOST to come back online…"
for i in $(seq 1 60); do
if ssh -o ConnectTimeout=3 -o BatchMode=yes "$HOST" true 2>/dev/null; then
ok "$HOST reachable"
break
fi
printf '.'
sleep 3
done
echo
# Re-stage in case the previous /tmp script was wiped on reboot.
write_remote_script
ssh "$HOST" "$REMOTE_SCRIPT phase2"
ok "Phase 2 complete."
cat <<'NEXT'
────────────────────────────────────────────────────────────────────
NEXT — flip Kokoro from cpu → gpu variant now that CUDA 12.9 is live:
ssh irv-ml1 '
cd /opt/docker/compose/kokoro
sed -i "s/^KOKORO_VARIANT=.*/KOKORO_VARIANT=gpu/" .env
sed -i "s/^KOKORO_USE_GPU=.*/KOKORO_USE_GPU=true/" .env
sed -i "s/^KOKORO_GPU_DEVICES=.*/KOKORO_GPU_DEVICES=0/" .env
'
scripts/elway irv-ml1 --playbook playbooks/deploy-kokoro.yaml
After that, the two outstanding deploys:
scripts/elway irv-ml1 --playbook playbooks/deploy-vibevoice.yaml
scripts/elway irv-ml1 --playbook playbooks/deploy-chatterbox.yaml
────────────────────────────────────────────────────────────────────
NEXT
}
rollback() {
say "Rollback: reinstalling pre-upgrade nvidia packages on $HOST"
write_remote_script
ssh -t "$HOST" "$REMOTE_SCRIPT rollback"
ok "Rollback dispatched. Host rebooting."
}
case "${1:-}" in
phase1) phase1 ;;
phase2) phase2 ;;
rollback) rollback ;;
*) echo "usage: $0 {phase1|phase2|rollback}" >&2; exit 2 ;;
esac