fix(scriberr-rebuild): memory stage works on a shared GPU 3
Scriberr moved to fv-ml1 GPU 3 on 2026-09-30, so the card is no longer idle when the rebuild runs. The memory stage now requires >= 20 GB free instead of an idle card (GPUs 0-2 still fail that) and attributes the peak only to the host PIDs of its own container, captured with docker top alongside the 0.2 s nvidia-smi samples. Verified with five processes on the card: peak 5,496 MiB, identical to the exclusive-card figure.
This commit is contained in:
+21
-14
@@ -23,16 +23,19 @@
|
|||||||
# unit the slicer's pure-function tests, under the live env's numpy/librosa
|
# unit the slicer's pure-function tests, under the live env's numpy/librosa
|
||||||
# seam patched script, production invocation, short fixture at
|
# seam patched script, production invocation, short fixture at
|
||||||
# --chunk-len 10, JSON validated against the Go struct
|
# --chunk-len 10, JSON validated against the Go struct
|
||||||
# memory same on a long recording at --chunk-len 120 on an IDLE GPU,
|
# memory same on a long recording at --chunk-len 120 on a GPU with room
|
||||||
# nvidia-smi sampled every 0.2 s; per-process peak <= the budget
|
# (>= 20 GB free); nvidia-smi sampled every 0.2 s, and only this
|
||||||
|
# run's own container PIDs count, so a Scriberr job on the same
|
||||||
|
# card does not pollute the peak. Per-process peak <= the budget
|
||||||
#
|
#
|
||||||
# Usage:
|
# Usage:
|
||||||
# scripts/scriberr-rebuild [--sha SHA40] [--suffix NAME] [--gpu N]
|
# scripts/scriberr-rebuild [--sha SHA40] [--suffix NAME] [--gpu N]
|
||||||
# [--budget MIB] [--memory-audio PATH] [--reuse-image]
|
# [--budget MIB] [--memory-audio PATH] [--reuse-image]
|
||||||
#
|
#
|
||||||
# Defaults: --sha PINNED_SHA below, --suffix slicer1, --gpu 3, --budget 5496,
|
# Defaults: --sha PINNED_SHA below, --suffix slicer1, --gpu 3, --budget 5496,
|
||||||
# --memory-audio the public 30-min SCOTUS fixture. The GPU must be idle
|
# --memory-audio the public 30-min SCOTUS fixture. The GPU must have >= 20 GB
|
||||||
# (< 100 MiB used), which in practice means GPU 3; GPUs 0-2 run live seats.
|
# free, which in practice means GPU 3 (Scriberr's own card since 2026-09-30,
|
||||||
|
# idle at 0 MiB, ~5.5 GB during a job); GPUs 0-2 are full of vLLM seats.
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
PINNED_SHA=a353078fd96b8aca4002681813524b7397c90df1 # upstream HEAD 2026-09-20
|
PINNED_SHA=a353078fd96b8aca4002681813524b7397c90df1 # upstream HEAD 2026-09-20
|
||||||
@@ -224,13 +227,13 @@ gpu_run() { # $1 audio path on host, $2 --chunk-len, $3 --min-chunks
|
|||||||
--output /tmp/out.json --chunk-len $2 >/tmp/run.log 2>&1 || { tail -5 /tmp/run.log; exit 1; }; \
|
--output /tmp/out.json --chunk-len $2 >/tmp/run.log 2>&1 || { tail -5 /tmp/run.log; exit 1; }; \
|
||||||
python3 /tools/seam-check.py /tmp/out.json --min-chunks $3'"
|
python3 /tools/seam-check.py /tmp/out.json --min-chunks $3'"
|
||||||
}
|
}
|
||||||
gpu_idle() {
|
gpu_idle() { # "room", not "idle": Scriberr itself may be running a job on this card
|
||||||
local used
|
local free
|
||||||
used=$("${SSH[@]}" "nvidia-smi -i $GPU --query-gpu=memory.used --format=csv,noheader,nounits") \
|
free=$("${SSH[@]}" "nvidia-smi -i $GPU --query-gpu=memory.free --format=csv,noheader,nounits") \
|
||||||
|| fail "$1" "could not read GPU $GPU memory on fv-ml1"
|
|| fail "$1" "could not read GPU $GPU memory on fv-ml1"
|
||||||
used=${used//[!0-9]/}
|
free=${free//[!0-9]/}
|
||||||
[ -n "$used" ] && [ "$used" -lt 100 ] \
|
[ -n "$free" ] && [ "$free" -ge 20000 ] \
|
||||||
|| fail "$1" "GPU $GPU is not idle (${used:-?} MiB used); refusing to share a live card"
|
|| fail "$1" "GPU $GPU has only ${free:-?} MiB free, need 20000; refusing to crowd a live card"
|
||||||
}
|
}
|
||||||
|
|
||||||
# ── seam ───────────────────────────────────────────────────────────────────
|
# ── seam ───────────────────────────────────────────────────────────────────
|
||||||
@@ -241,17 +244,21 @@ else fail seam "$out"; fi
|
|||||||
# ── memory ─────────────────────────────────────────────────────────────────
|
# ── memory ─────────────────────────────────────────────────────────────────
|
||||||
"${SSH[@]}" "test -s $MEM_AUDIO" || fail memory "memory audio $MEM_AUDIO not found on fv-ml1"
|
"${SSH[@]}" "test -s $MEM_AUDIO" || fail memory "memory audio $MEM_AUDIO not found on fv-ml1"
|
||||||
gpu_idle memory
|
gpu_idle memory
|
||||||
|
# Two host-side loops: every compute process on the card (0.2 s), and the host
|
||||||
|
# PIDs inside this run's container (0.5 s). Only rows whose PID was ours count.
|
||||||
"${SSH[@]}" "nohup nvidia-smi -i $GPU --query-compute-apps=pid,used_memory \
|
"${SSH[@]}" "nohup nvidia-smi -i $GPU --query-compute-apps=pid,used_memory \
|
||||||
--format=csv,noheader,nounits -lms 200 </dev/null >$SAMPLES 2>/dev/null & echo \$! >$SAMPLES.pid" \
|
--format=csv,noheader,nounits -lms 200 </dev/null >$SAMPLES 2>/dev/null & echo \$! >$SAMPLES.pid; \
|
||||||
|| fail memory "could not start the nvidia-smi sampler"
|
: >$SAMPLES.ours; nohup bash -c 'while true; do docker top $CNAME -eo pid 2>/dev/null | tail -n +2 >>$SAMPLES.ours; sleep 0.5; done' \
|
||||||
|
</dev/null >/dev/null 2>&1 & echo \$! >>$SAMPLES.pid" \
|
||||||
|
|| fail memory "could not start the samplers"
|
||||||
record run "$CNAME" "transient $TAG on GPU $GPU, --chunk-len 120, env ro; removed on exit"
|
record run "$CNAME" "transient $TAG on GPU $GPU, --chunk-len 120, env ro; removed on exit"
|
||||||
rc=0; out=$(gpu_run "$MEM_AUDIO" 120 2 2>&1) || rc=$? # `||`, not set +e: keeps the ERR trap quiet
|
rc=0; out=$(gpu_run "$MEM_AUDIO" 120 2 2>&1) || rc=$? # `||`, not set +e: keeps the ERR trap quiet
|
||||||
"${SSH[@]}" "kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid" || true
|
"${SSH[@]}" "kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid" || true
|
||||||
read -r pids peak < <("${SSH[@]}" \
|
read -r pids peak < <("${SSH[@]}" \
|
||||||
"awk -F', *' 'NF==2 {if (!(\$1 in p)) {p[\$1]=1; n++}; if (\$2+0>m) m=\$2+0} END {print n+0, m+0}' $SAMPLES") \
|
"awk -F', *' 'FNR==NR {ours[\$1+0]=1; next} NF==2 && ((\$1+0) in ours) {if (!(\$1 in p)) {p[\$1]=1; n++}; if (\$2+0>m) m=\$2+0} END {print n+0, m+0}' $SAMPLES.ours $SAMPLES") \
|
||||||
|| fail memory "no GPU samples could be read back from $SAMPLES"
|
|| fail memory "no GPU samples could be read back from $SAMPLES"
|
||||||
[ "$rc" = 0 ] || fail memory "memory run failed: $out"
|
[ "$rc" = 0 ] || fail memory "memory run failed: $out"
|
||||||
[ "$pids" = 1 ] || fail memory "saw $pids processes on GPU $GPU during the run; the peak is not attributable"
|
[ "$pids" = 1 ] || fail memory "saw $pids of this run's own processes on GPU $GPU (expected 1); the peak is not attributable"
|
||||||
if [ "$peak" -le "$BUDGET" ]; then
|
if [ "$peak" -le "$BUDGET" ]; then
|
||||||
pass memory "peak $peak MiB <= budget $BUDGET MiB (0.2 s samples, GPU $GPU); $(tail -1 <<<"$out")"
|
pass memory "peak $peak MiB <= budget $BUDGET MiB (0.2 s samples, GPU $GPU); $(tail -1 <<<"$out")"
|
||||||
else
|
else
|
||||||
|
|||||||
@@ -115,7 +115,7 @@ It clones that sha into a new `/opt/docker/src/scriberr-<sha7>-<suffix>`,
|
|||||||
`git apply --check`s each patch (a conflict stops the run and names the patch),
|
`git apply --check`s each patch (a conflict stops the run and names the patch),
|
||||||
builds `scriberr:local-blackwell-<sha7>-<suffix>` without touching older tags,
|
builds `scriberr:local-blackwell-<sha7>-<suffix>` without touching older tags,
|
||||||
then runs the embed, unit, seam and memory stages and prints a PASS/FAIL table.
|
then runs the embed, unit, seam and memory stages and prints a PASS/FAIL table.
|
||||||
Memory runs on GPU 3 and refuses a GPU that is not idle. A conflict means the
|
Memory runs on GPU 3, which Scriberr itself now occupies (2026-09-30); the stage needs ≥ 20 GB free and counts only its own container's PIDs, so a Scriberr job on the card neither blocks nor pollutes it. A conflict means the
|
||||||
patch needs rebasing: in a checkout of the new sha, `git am -3` the old patch,
|
patch needs rebasing: in a checkout of the new sha, `git am -3` the old patch,
|
||||||
resolve, run the unit tests, then `git format-patch -1 --stdout >
|
resolve, run the unit tests, then `git format-patch -1 --stdout >
|
||||||
0001-parakeet-pause-aware-slicer.patch` and re-run the rebuild.
|
0001-parakeet-pause-aware-slicer.patch` and re-run the rebuild.
|
||||||
|
|||||||
Reference in New Issue
Block a user