From 92501a29c1c82aeba079b4c136b24fb5481c170e Mon Sep 17 00:00:00 2001 From: Vuong Hoang Date: Wed, 30 Sep 2026 13:30:29 -0700 Subject: [PATCH] fix(scriberr-rebuild): memory stage works on a shared GPU 3 Scriberr moved to fv-ml1 GPU 3 on 2026-09-30, so the card is no longer idle when the rebuild runs. The memory stage now requires >= 20 GB free instead of an idle card (GPUs 0-2 still fail that) and attributes the peak only to the host PIDs of its own container, captured with docker top alongside the 0.2 s nvidia-smi samples. Verified with five processes on the card: peak 5,496 MiB, identical to the exclusive-card figure. --- scripts/scriberr-rebuild | 35 ++++++++++++++++++------------- stacks/scriberr/patches/README.md | 2 +- 2 files changed, 22 insertions(+), 15 deletions(-) diff --git a/scripts/scriberr-rebuild b/scripts/scriberr-rebuild index fe96287..a776d02 100755 --- a/scripts/scriberr-rebuild +++ b/scripts/scriberr-rebuild @@ -23,16 +23,19 @@ # unit the slicer's pure-function tests, under the live env's numpy/librosa # seam patched script, production invocation, short fixture at # --chunk-len 10, JSON validated against the Go struct -# memory same on a long recording at --chunk-len 120 on an IDLE GPU, -# nvidia-smi sampled every 0.2 s; per-process peak <= the budget +# memory same on a long recording at --chunk-len 120 on a GPU with room +# (>= 20 GB free); nvidia-smi sampled every 0.2 s, and only this +# run's own container PIDs count, so a Scriberr job on the same +# card does not pollute the peak. Per-process peak <= the budget # # Usage: # scripts/scriberr-rebuild [--sha SHA40] [--suffix NAME] [--gpu N] # [--budget MIB] [--memory-audio PATH] [--reuse-image] # # Defaults: --sha PINNED_SHA below, --suffix slicer1, --gpu 3, --budget 5496, -# --memory-audio the public 30-min SCOTUS fixture. The GPU must be idle -# (< 100 MiB used), which in practice means GPU 3; GPUs 0-2 run live seats. +# --memory-audio the public 30-min SCOTUS fixture. The GPU must have >= 20 GB +# free, which in practice means GPU 3 (Scriberr's own card since 2026-09-30, +# idle at 0 MiB, ~5.5 GB during a job); GPUs 0-2 are full of vLLM seats. set -euo pipefail PINNED_SHA=a353078fd96b8aca4002681813524b7397c90df1 # upstream HEAD 2026-09-20 @@ -224,13 +227,13 @@ gpu_run() { # $1 audio path on host, $2 --chunk-len, $3 --min-chunks --output /tmp/out.json --chunk-len $2 >/tmp/run.log 2>&1 || { tail -5 /tmp/run.log; exit 1; }; \ python3 /tools/seam-check.py /tmp/out.json --min-chunks $3'" } -gpu_idle() { - local used - used=$("${SSH[@]}" "nvidia-smi -i $GPU --query-gpu=memory.used --format=csv,noheader,nounits") \ +gpu_idle() { # "room", not "idle": Scriberr itself may be running a job on this card + local free + free=$("${SSH[@]}" "nvidia-smi -i $GPU --query-gpu=memory.free --format=csv,noheader,nounits") \ || fail "$1" "could not read GPU $GPU memory on fv-ml1" - used=${used//[!0-9]/} - [ -n "$used" ] && [ "$used" -lt 100 ] \ - || fail "$1" "GPU $GPU is not idle (${used:-?} MiB used); refusing to share a live card" + free=${free//[!0-9]/} + [ -n "$free" ] && [ "$free" -ge 20000 ] \ + || fail "$1" "GPU $GPU has only ${free:-?} MiB free, need 20000; refusing to crowd a live card" } # ── seam ─────────────────────────────────────────────────────────────────── @@ -241,17 +244,21 @@ else fail seam "$out"; fi # ── memory ───────────────────────────────────────────────────────────────── "${SSH[@]}" "test -s $MEM_AUDIO" || fail memory "memory audio $MEM_AUDIO not found on fv-ml1" gpu_idle memory +# Two host-side loops: every compute process on the card (0.2 s), and the host +# PIDs inside this run's container (0.5 s). Only rows whose PID was ours count. "${SSH[@]}" "nohup nvidia-smi -i $GPU --query-compute-apps=pid,used_memory \ - --format=csv,noheader,nounits -lms 200 $SAMPLES 2>/dev/null & echo \$! >$SAMPLES.pid" \ - || fail memory "could not start the nvidia-smi sampler" + --format=csv,noheader,nounits -lms 200 $SAMPLES 2>/dev/null & echo \$! >$SAMPLES.pid; \ + : >$SAMPLES.ours; nohup bash -c 'while true; do docker top $CNAME -eo pid 2>/dev/null | tail -n +2 >>$SAMPLES.ours; sleep 0.5; done' \ + /dev/null 2>&1 & echo \$! >>$SAMPLES.pid" \ + || fail memory "could not start the samplers" record run "$CNAME" "transient $TAG on GPU $GPU, --chunk-len 120, env ro; removed on exit" rc=0; out=$(gpu_run "$MEM_AUDIO" 120 2 2>&1) || rc=$? # `||`, not set +e: keeps the ERR trap quiet "${SSH[@]}" "kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid" || true read -r pids peak < <("${SSH[@]}" \ - "awk -F', *' 'NF==2 {if (!(\$1 in p)) {p[\$1]=1; n++}; if (\$2+0>m) m=\$2+0} END {print n+0, m+0}' $SAMPLES") \ + "awk -F', *' 'FNR==NR {ours[\$1+0]=1; next} NF==2 && ((\$1+0) in ours) {if (!(\$1 in p)) {p[\$1]=1; n++}; if (\$2+0>m) m=\$2+0} END {print n+0, m+0}' $SAMPLES.ours $SAMPLES") \ || fail memory "no GPU samples could be read back from $SAMPLES" [ "$rc" = 0 ] || fail memory "memory run failed: $out" -[ "$pids" = 1 ] || fail memory "saw $pids processes on GPU $GPU during the run; the peak is not attributable" +[ "$pids" = 1 ] || fail memory "saw $pids of this run's own processes on GPU $GPU (expected 1); the peak is not attributable" if [ "$peak" -le "$BUDGET" ]; then pass memory "peak $peak MiB <= budget $BUDGET MiB (0.2 s samples, GPU $GPU); $(tail -1 <<<"$out")" else diff --git a/stacks/scriberr/patches/README.md b/stacks/scriberr/patches/README.md index 67acf7e..ae832aa 100644 --- a/stacks/scriberr/patches/README.md +++ b/stacks/scriberr/patches/README.md @@ -115,7 +115,7 @@ It clones that sha into a new `/opt/docker/src/scriberr--`, `git apply --check`s each patch (a conflict stops the run and names the patch), builds `scriberr:local-blackwell--` without touching older tags, then runs the embed, unit, seam and memory stages and prints a PASS/FAIL table. -Memory runs on GPU 3 and refuses a GPU that is not idle. A conflict means the +Memory runs on GPU 3, which Scriberr itself now occupies (2026-09-30); the stage needs ≥ 20 GB free and counts only its own container's PIDs, so a Scriberr job on the card neither blocks nor pollutes it. A conflict means the patch needs rebasing: in a checkout of the new sha, `git am -3` the old patch, resolve, run the unit tests, then `git format-patch -1 --stdout > 0001-parakeet-pause-aware-slicer.patch` and re-run the rebuild.