fix(scriberr-rebuild): memory stage works on a shared GPU 3

Scriberr moved to fv-ml1 GPU 3 on 2026-09-30, so the card is no longer idle when the rebuild runs. The memory stage now requires >= 20 GB free instead of an idle card (GPUs 0-2 still fail that) and attributes the peak only to the host PIDs of its own container, captured with docker top alongside the 0.2 s nvidia-smi samples. Verified with five processes on the card: peak 5,496 MiB, identical to the exclusive-card figure.
This commit is contained in:
vh
2026-09-30 13:30:29 -07:00
parent e3dbb08d84
commit 92501a29c1
2 changed files with 22 additions and 15 deletions
+21 -14
View File
@@ -23,16 +23,19 @@
# unit the slicer's pure-function tests, under the live env's numpy/librosa # unit the slicer's pure-function tests, under the live env's numpy/librosa
# seam patched script, production invocation, short fixture at # seam patched script, production invocation, short fixture at
# --chunk-len 10, JSON validated against the Go struct # --chunk-len 10, JSON validated against the Go struct
# memory same on a long recording at --chunk-len 120 on an IDLE GPU, # memory same on a long recording at --chunk-len 120 on a GPU with room
# nvidia-smi sampled every 0.2 s; per-process peak <= the budget # (>= 20 GB free); nvidia-smi sampled every 0.2 s, and only this
# run's own container PIDs count, so a Scriberr job on the same
# card does not pollute the peak. Per-process peak <= the budget
# #
# Usage: # Usage:
# scripts/scriberr-rebuild [--sha SHA40] [--suffix NAME] [--gpu N] # scripts/scriberr-rebuild [--sha SHA40] [--suffix NAME] [--gpu N]
# [--budget MIB] [--memory-audio PATH] [--reuse-image] # [--budget MIB] [--memory-audio PATH] [--reuse-image]
# #
# Defaults: --sha PINNED_SHA below, --suffix slicer1, --gpu 3, --budget 5496, # Defaults: --sha PINNED_SHA below, --suffix slicer1, --gpu 3, --budget 5496,
# --memory-audio the public 30-min SCOTUS fixture. The GPU must be idle # --memory-audio the public 30-min SCOTUS fixture. The GPU must have >= 20 GB
# (< 100 MiB used), which in practice means GPU 3; GPUs 0-2 run live seats. # free, which in practice means GPU 3 (Scriberr's own card since 2026-09-30,
# idle at 0 MiB, ~5.5 GB during a job); GPUs 0-2 are full of vLLM seats.
set -euo pipefail set -euo pipefail
PINNED_SHA=a353078fd96b8aca4002681813524b7397c90df1 # upstream HEAD 2026-09-20 PINNED_SHA=a353078fd96b8aca4002681813524b7397c90df1 # upstream HEAD 2026-09-20
@@ -224,13 +227,13 @@ gpu_run() { # $1 audio path on host, $2 --chunk-len, $3 --min-chunks
--output /tmp/out.json --chunk-len $2 >/tmp/run.log 2>&1 || { tail -5 /tmp/run.log; exit 1; }; \ --output /tmp/out.json --chunk-len $2 >/tmp/run.log 2>&1 || { tail -5 /tmp/run.log; exit 1; }; \
python3 /tools/seam-check.py /tmp/out.json --min-chunks $3'" python3 /tools/seam-check.py /tmp/out.json --min-chunks $3'"
} }
gpu_idle() { gpu_idle() { # "room", not "idle": Scriberr itself may be running a job on this card
local used local free
used=$("${SSH[@]}" "nvidia-smi -i $GPU --query-gpu=memory.used --format=csv,noheader,nounits") \ free=$("${SSH[@]}" "nvidia-smi -i $GPU --query-gpu=memory.free --format=csv,noheader,nounits") \
|| fail "$1" "could not read GPU $GPU memory on fv-ml1" || fail "$1" "could not read GPU $GPU memory on fv-ml1"
used=${used//[!0-9]/} free=${free//[!0-9]/}
[ -n "$used" ] && [ "$used" -lt 100 ] \ [ -n "$free" ] && [ "$free" -ge 20000 ] \
|| fail "$1" "GPU $GPU is not idle (${used:-?} MiB used); refusing to share a live card" || fail "$1" "GPU $GPU has only ${free:-?} MiB free, need 20000; refusing to crowd a live card"
} }
# ── seam ─────────────────────────────────────────────────────────────────── # ── seam ───────────────────────────────────────────────────────────────────
@@ -241,17 +244,21 @@ else fail seam "$out"; fi
# ── memory ───────────────────────────────────────────────────────────────── # ── memory ─────────────────────────────────────────────────────────────────
"${SSH[@]}" "test -s $MEM_AUDIO" || fail memory "memory audio $MEM_AUDIO not found on fv-ml1" "${SSH[@]}" "test -s $MEM_AUDIO" || fail memory "memory audio $MEM_AUDIO not found on fv-ml1"
gpu_idle memory gpu_idle memory
# Two host-side loops: every compute process on the card (0.2 s), and the host
# PIDs inside this run's container (0.5 s). Only rows whose PID was ours count.
"${SSH[@]}" "nohup nvidia-smi -i $GPU --query-compute-apps=pid,used_memory \ "${SSH[@]}" "nohup nvidia-smi -i $GPU --query-compute-apps=pid,used_memory \
--format=csv,noheader,nounits -lms 200 </dev/null >$SAMPLES 2>/dev/null & echo \$! >$SAMPLES.pid" \ --format=csv,noheader,nounits -lms 200 </dev/null >$SAMPLES 2>/dev/null & echo \$! >$SAMPLES.pid; \
|| fail memory "could not start the nvidia-smi sampler" : >$SAMPLES.ours; nohup bash -c 'while true; do docker top $CNAME -eo pid 2>/dev/null | tail -n +2 >>$SAMPLES.ours; sleep 0.5; done' \
</dev/null >/dev/null 2>&1 & echo \$! >>$SAMPLES.pid" \
|| fail memory "could not start the samplers"
record run "$CNAME" "transient $TAG on GPU $GPU, --chunk-len 120, env ro; removed on exit" record run "$CNAME" "transient $TAG on GPU $GPU, --chunk-len 120, env ro; removed on exit"
rc=0; out=$(gpu_run "$MEM_AUDIO" 120 2 2>&1) || rc=$? # `||`, not set +e: keeps the ERR trap quiet rc=0; out=$(gpu_run "$MEM_AUDIO" 120 2 2>&1) || rc=$? # `||`, not set +e: keeps the ERR trap quiet
"${SSH[@]}" "kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid" || true "${SSH[@]}" "kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid" || true
read -r pids peak < <("${SSH[@]}" \ read -r pids peak < <("${SSH[@]}" \
"awk -F', *' 'NF==2 {if (!(\$1 in p)) {p[\$1]=1; n++}; if (\$2+0>m) m=\$2+0} END {print n+0, m+0}' $SAMPLES") \ "awk -F', *' 'FNR==NR {ours[\$1+0]=1; next} NF==2 && ((\$1+0) in ours) {if (!(\$1 in p)) {p[\$1]=1; n++}; if (\$2+0>m) m=\$2+0} END {print n+0, m+0}' $SAMPLES.ours $SAMPLES") \
|| fail memory "no GPU samples could be read back from $SAMPLES" || fail memory "no GPU samples could be read back from $SAMPLES"
[ "$rc" = 0 ] || fail memory "memory run failed: $out" [ "$rc" = 0 ] || fail memory "memory run failed: $out"
[ "$pids" = 1 ] || fail memory "saw $pids processes on GPU $GPU during the run; the peak is not attributable" [ "$pids" = 1 ] || fail memory "saw $pids of this run's own processes on GPU $GPU (expected 1); the peak is not attributable"
if [ "$peak" -le "$BUDGET" ]; then if [ "$peak" -le "$BUDGET" ]; then
pass memory "peak $peak MiB <= budget $BUDGET MiB (0.2 s samples, GPU $GPU); $(tail -1 <<<"$out")" pass memory "peak $peak MiB <= budget $BUDGET MiB (0.2 s samples, GPU $GPU); $(tail -1 <<<"$out")"
else else
+1 -1
View File
@@ -115,7 +115,7 @@ It clones that sha into a new `/opt/docker/src/scriberr-<sha7>-<suffix>`,
`git apply --check`s each patch (a conflict stops the run and names the patch), `git apply --check`s each patch (a conflict stops the run and names the patch),
builds `scriberr:local-blackwell-<sha7>-<suffix>` without touching older tags, builds `scriberr:local-blackwell-<sha7>-<suffix>` without touching older tags,
then runs the embed, unit, seam and memory stages and prints a PASS/FAIL table. then runs the embed, unit, seam and memory stages and prints a PASS/FAIL table.
Memory runs on GPU 3 and refuses a GPU that is not idle. A conflict means the Memory runs on GPU 3, which Scriberr itself now occupies (2026-09-30); the stage needs ≥ 20 GB free and counts only its own container's PIDs, so a Scriberr job on the card neither blocks nor pollutes it. A conflict means the
patch needs rebasing: in a checkout of the new sha, `git am -3` the old patch, patch needs rebasing: in a checkout of the new sha, `git am -3` the old patch,
resolve, run the unit tests, then `git format-patch -1 --stdout > resolve, run the unit tests, then `git format-patch -1 --stdout >
0001-parakeet-pause-aware-slicer.patch` and re-run the rebuild. 0001-parakeet-pause-aware-slicer.patch` and re-run the rebuild.