Files
esh-pfi-infrastructure/scripts/scriberr-rebuild
T
vh ee3db68db1 feat(scriberr): overlap-and-stitch Parakeet slicer patch, rebuild script, bench
Carry patches/0001 on our Scriberr build (upstream a353078): adjacent
buffered chunks overlap by 4 s inside --chunk-len and hand over at a word
both chunks transcribed alike, instead of cutting at fixed marks with no
overlap. Pause-aware cutting is included as an opt-in (--pause-search);
it measured neutral once the stitch was right. The Go<->Python CLI and
JSON seam is unchanged.

Bench (4 recordings, 118 min, 3 cut placements each, against a no-cut
whole-file reference; metrics only, private audio stays on fv-ml1):
cuts with an error within +-3 s fall from 52% (93/179) to 22% (41/184)
against a 19% background; floor +-0.08. Positive control: upstream's
cutter +0.33 over background. A-vs-A byte-identical in-process and
across CLI processes. Peak GPU memory unchanged at 5,496 MiB (n=3).
Also found: Parakeet skips runs of >=10 words mid-chunk with any
slicer, upstream's included; not addressed here.

scripts/scriberr-rebuild clones a pinned upstream sha into a new
/opt/docker/src dir, git-apply-checks the patches, builds a distinct
tag, and checks embed, unit tests, the JSON seam (scriberr-seam-check.py)
and the memory budget on idle GPU 3. Deploy stays manual. The upstream
PR is prepared under patches/upstream-pr/ and not opened.
2026-09-30 12:10:32 -07:00

264 lines
14 KiB
Bash
Executable File

#!/usr/bin/env bash
# scriberr-rebuild — rebuild Scriberr's Blackwell image at a PINNED upstream sha
# with our patches applied, then prove the result before anyone deploys it.
#
# Runs on nh3-dev and drives fv-ml1 over ssh. Deploy is a SEPARATE manual step
# (stacks/scriberr/patches/README.md § Deploy); this script never touches the
# live container, its .env, or GPU 1.
#
# Why this exists: upstream publishes no sm_120 image, so we build from source,
# and we carry a patch to the Parakeet slicer (stacks/scriberr/patches/). Upstream
# moves slowly, so an upgrade should be one command plus a verdict.
#
# Stages (each prints PASS/FAIL; the first FAIL stops the run):
# clone clean shallow clone of upstream at the pinned sha, in a NEW dir
# /opt/docker/src/scriberr-<sha7>-<suffix> (reused only if it already
# holds that sha with every patch applied)
# patch `git apply --check` then `git apply`, patch by patch; a conflict
# stops the run loudly and names the patch
# build docker build -f Dockerfile.cuda.12.9 -t scriberr:local-blackwell-<sha7>-<suffix>
# — a DISTINCT tag, so the running image is never overwritten
# embed the patched script's exact bytes are inside the new Go binary
# (Scriberr rewrites the env's copy from this embed on every start)
# unit the slicer's pure-function tests, under the live env's numpy/librosa
# seam patched script, production invocation, short fixture at
# --chunk-len 10, JSON validated against the Go struct
# memory same on a long recording at --chunk-len 120 on an IDLE GPU,
# nvidia-smi sampled every 0.2 s; per-process peak <= the budget
#
# Usage:
# scripts/scriberr-rebuild [--sha SHA40] [--suffix NAME] [--gpu N]
# [--budget MIB] [--memory-audio PATH] [--reuse-image]
#
# Defaults: --sha PINNED_SHA below, --suffix slicer1, --gpu 3, --budget 5496,
# --memory-audio the public 30-min SCOTUS fixture. The GPU must be idle
# (< 100 MiB used), which in practice means GPU 3; GPUs 0-2 run live seats.
set -euo pipefail
PINNED_SHA=a353078fd96b8aca4002681813524b7397c90df1 # upstream HEAD 2026-09-20
UPSTREAM=https://github.com/rishikanthc/Scriberr.git
HOST=${SCRIBERR_REBUILD_HOST:-infra-ops@10.251.50.54}
ENV_DIR=/tank/scriberr/whisperx-env # live env, always mounted READ-ONLY
TOOLS=/opt/docker/src/scriberr-rebuild # fixtures + seam checker on fv-ml1
SCRIPT_REL=internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py
TEST_REL=internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py
SEAM_AUDIO_REL=tests/data/AMI-Corpus-IB4002.Mix-Headset-clip.wav
SHA=$PINNED_SHA SUFFIX=slicer1 GPU=3 BUDGET=5496 REUSE_IMAGE=0
MEM_AUDIO=$TOOLS/fixtures/scotus-22-451-first30m.wav
while [ $# -gt 0 ]; do
case $1 in
--sha) SHA=$2; shift 2 ;;
--suffix) SUFFIX=$2; shift 2 ;;
--gpu) GPU=$2; shift 2 ;;
--budget) BUDGET=$2; shift 2 ;;
--memory-audio) MEM_AUDIO=$2; shift 2 ;;
--reuse-image) REUSE_IMAGE=1; shift ;;
-h|--help) sed -n '2,/^set -euo/p' "$0" | sed '$d; s/^# \{0,1\}//'; exit 0 ;;
*) echo "unknown argument: $1 (see --help)" >&2; exit 2 ;;
esac
done
[[ $SHA =~ ^[0-9a-f]{40}$ ]] || { echo "--sha must be a full 40-char sha (GitHub fetches by full sha)" >&2; exit 2; }
[[ $SUFFIX =~ ^[a-z0-9][a-z0-9.-]*$ ]] || { echo "--suffix must be [a-z0-9.-]" >&2; exit 2; }
[[ $GPU =~ ^[0-9]+$ && $BUDGET =~ ^[0-9]+$ ]] || { echo "--gpu and --budget must be integers" >&2; exit 2; }
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
PATCH_DIR=$REPO_ROOT/stacks/scriberr/patches
mapfile -t PATCHES < <(find "$PATCH_DIR" -maxdepth 1 -name '*.patch' | sort)
[ ${#PATCHES[@]} -gt 0 ] || { echo "no patches in $PATCH_DIR" >&2; exit 2; }
# The only paths a reused build dir may differ from upstream in.
mapfile -t PATCHED_PATHS < <(sed -n 's#^+++ b/##p' "${PATCHES[@]}" | sort -u)
SHA7=${SHA:0:7}
TAG=scriberr:local-blackwell-$SHA7-$SUFFIX
BUILD_DIR=/opt/docker/src/scriberr-$SHA7-$SUFFIX
PATCH_SUM=$(cat "${PATCHES[@]}" | sha256sum | cut -c1-16)
CNAME=scriberr-rebuild-$SHA7-$SUFFIX # every GPU container, so cleanup can find it
SAMPLES=/tmp/$CNAME.mem.csv
BUILD_LOG=/tmp/$CNAME.build.log
MIN_FREE_GB=20 # under Docker's root dir (zroot); an image adds ~0.1-6 GB
# One multiplexed ssh connection serves the ~15 remote calls of a run.
CM=(-o BatchMode=yes -o ControlMaster=auto -o "ControlPath=$HOME/.ssh/cm-%C" -o ControlPersist=120)
SSH=(ssh "${CM[@]}" "$HOST")
SCP=(scp -q "${CM[@]}")
RESULTS=()
pass() { RESULTS+=("PASS $1 $2"); echo "== PASS $1: $2"; }
fail() {
RESULTS+=("FAIL $1 $2"); echo "== FAIL $1: $2" >&2
summary; exit 1
}
summary() {
echo; echo "scriberr-rebuild upstream=$SHA7 patches=$PATCH_SUM tag=$TAG"
printf ' %s\n' "${RESULTS[@]}"
}
record() { # best effort: the audit trail must not become a failure point
"$REPO_ROOT/scripts/ops-log" record --host fv-ml1 --action "$1" --target "$2" \
--detail "$3" >/dev/null 2>&1 || echo "(ops-log record failed; continuing)" >&2
}
remote() { "${SSH[@]}" bash -s -- "$@"; }
# Whatever happens (a FAIL, Ctrl-C, a dropped connection), never leave the
# nvidia-smi sampler or a transient GPU container behind on fv-ml1.
cleanup() {
"${SSH[@]}" "[ -s $SAMPLES.pid ] && kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid; \
docker rm -f $CNAME >/dev/null 2>&1; true" 2>/dev/null || true
}
trap cleanup EXIT
trap 'exit 130' INT TERM
# An unguarded remote call that fails must still say so and print the table.
set -E
trap 'echo "== ABORT: unexpected failure at line $LINENO (see output above)" >&2; summary' ERR
# Every container run: the new image, the live env READ-ONLY, the build tree
# read-only, and nothing else writable but the container's own /tmp.
DOCKER_RUN="docker run --rm --user 1002:1003 -e HOME=/tmp -e USER=infra-ops -e LOGNAME=infra-ops \
-e PYTHONDONTWRITEBYTECODE=1 -e PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True -e UV_LINK_MODE=copy \
-v $ENV_DIR:/app/whisperx-env:ro -v $BUILD_DIR:/src:ro -v $TOOLS:/tools:ro --entrypoint bash"
UVRUN="uv run --native-tls --project /app/whisperx-env/parakeet"
# ── clone ──────────────────────────────────────────────────────────────────
if out=$(remote "$BUILD_DIR" "$UPSTREAM" "$SHA" "$TOOLS" "${PATCHED_PATHS[@]}" 2>&1 <<'EOF'
set -euo pipefail
dir=$1 upstream=$2 sha=$3 tools=$4
shift 4
sudo -n install -d -o infra-ops -g infra-ops "$tools" "$tools/fixtures" | cat
if [ -e "$dir" ]; then
[ "$(git -C "$dir" rev-parse HEAD 2>/dev/null)" = "$sha" ] \
|| { echo "$dir exists but is not a checkout of $sha; remove it by hand (sudo -n rm -rf $dir) or pick --suffix"; exit 1; }
extra=$({ git -C "$dir" diff --name-only HEAD; git -C "$dir" ls-files --others --exclude-standard; } \
| sort -u | grep -vxF -f <(printf '%s\n' "$@") || true)
[ -z "$extra" ] || { echo "$dir has changes outside the patches ($extra); remove it by hand or pick --suffix"; exit 1; }
echo "reusing $dir"
exit 0
fi
sudo -n install -d -o infra-ops -g infra-ops "$dir" | cat
cd "$dir"
git init -q
git remote add origin "$upstream"
git fetch -q --depth 1 origin "$sha" \
|| { echo "fetch of $sha failed; $dir is left empty, remove it by hand (sudo -n rm -rf $dir)"; exit 1; }
git checkout -q --detach FETCH_HEAD
[ "$(git rev-parse HEAD)" = "$sha" ] || { echo "checked out $(git rev-parse HEAD), wanted $sha"; exit 1; }
echo "cloned $sha into $dir"
EOF
); then pass clone "$out"; else fail clone "$out"; fi
if [[ $out == reusing* ]]; then REUSED=1; else REUSED=0; record create "$BUILD_DIR" "scriberr-rebuild: clean clone of upstream $SHA7"; fi
# ── patch ──────────────────────────────────────────────────────────────────
for p in "${PATCHES[@]}"; do
name=$(basename "$p")
# Already applied (a reused dir)? `apply --reverse --check` succeeds only then.
if "${SSH[@]}" "cd $BUILD_DIR && git apply --reverse --check -" <"$p" >/dev/null 2>&1; then
pass patch "$name already applied"
continue
fi
if ! out=$("${SSH[@]}" "cd $BUILD_DIR && git apply --check -" <"$p" 2>&1); then
if [ "$REUSED" = 1 ]; then
fail patch "$name does not apply to the REUSED $BUILD_DIR, which holds an older state of the patches; pick a new --suffix or remove the dir by hand (sudo -n rm -rf $BUILD_DIR):
$out"
fi
fail patch "$name DOES NOT APPLY to upstream $SHA7 — rebase the patch before upgrading:
$out"
fi
"${SSH[@]}" "cd $BUILD_DIR && git apply -" <"$p" || fail patch "$name: git apply failed after a clean --check"
record patch "$BUILD_DIR" "scriberr-rebuild: git apply $name"
pass patch "$name applied"
done
# ── build ──────────────────────────────────────────────────────────────────
if "${SSH[@]}" "docker image inspect $TAG >/dev/null 2>&1"; then
[ "$REUSE_IMAGE" = 1 ] || fail build "$TAG already exists; pass --reuse-image to test it, or pick a new --suffix"
pass build "reusing existing $TAG"
else
free=$("${SSH[@]}" "df -BG --output=avail \$(docker info -f '{{.DockerRootDir}}') | tail -1") \
|| fail build "could not read free space on fv-ml1"
free=${free//[!0-9]/}
[ "${free:-0}" -ge "$MIN_FREE_GB" ] \
|| fail build "only ${free:-?} GB free under Docker's root dir, need $MIN_FREE_GB; remove superseded scriberr tags first (patches/README.md)"
echo "building $TAG (several minutes; log on fv-ml1 at $BUILD_LOG)"
if "${SSH[@]}" "docker build -f $BUILD_DIR/Dockerfile.cuda.12.9 -t $TAG \
--label org.phasefinal.scriberr.upstream=$SHA --label org.phasefinal.scriberr.patches=$PATCH_SUM \
$BUILD_DIR >$BUILD_LOG 2>&1"; then
record build "$TAG" "scriberr-rebuild: upstream $SHA7 + patches $PATCH_SUM; old images kept"
pass build "$TAG ($free GB was free)"
else
"${SSH[@]}" "tail -25 $BUILD_LOG" >&2 || true
fail build "docker build failed (log tail above)"
fi
fi
# ── embed ──────────────────────────────────────────────────────────────────
if out=$(remote "$TAG" "$BUILD_DIR" "$SCRIPT_REL" 2>&1 <<'EOF'
docker run --rm -v "$2":/src:ro --entrypoint python3 "$1" -c "
import sys
script = open('/src/$3', 'rb').read()
sys.exit(0 if script in open('/app/scriberr', 'rb').read() else 1)"
EOF
); then
pass embed "patched $(basename "$SCRIPT_REL") is byte-identical inside /app/scriberr"
else
fail embed "the binary does not embed the patched script ${out:+($out)}"
fi
# ── unit ───────────────────────────────────────────────────────────────────
if out=$("${SSH[@]}" "$DOCKER_RUN $TAG -c 'cd /tmp && $UVRUN --with pytest \
python -m pytest -q -p no:cacheprovider /src/$TEST_REL 2>&1 | tail -3'" 2>&1) \
&& grep -q ' passed' <<<"$out" && ! grep -Eq 'failed|error' <<<"$out"; then
pass unit "$(tail -1 <<<"$out")"
else
fail unit "$out"
fi
# The checker travels with this script, so the host copy is refreshed each run.
"${SCP[@]}" "$REPO_ROOT/scripts/scriberr-seam-check.py" "$HOST:$TOOLS/seam-check.py" \
|| fail seam "could not copy the seam checker to fv-ml1:$TOOLS"
record update "$TOOLS/seam-check.py" "scriberr-rebuild: refreshed the seam checker"
# One GPU run of the patched script under the production invocation, output
# kept inside the container, validated there. Prints the seam checker's line.
gpu_run() { # $1 audio path on host, $2 --chunk-len, $3 --min-chunks
local audio_dir; audio_dir=$(dirname "$1")
"${SSH[@]}" "$DOCKER_RUN --name $CNAME --gpus '\"device=$GPU\"' -e NVIDIA_VISIBLE_DEVICES=$GPU \
-v $audio_dir:/audio:ro $TAG -c 'cd /tmp && $UVRUN python /src/$SCRIPT_REL /audio/$(basename "$1") \
--output /tmp/out.json --chunk-len $2 >/tmp/run.log 2>&1 || { tail -5 /tmp/run.log; exit 1; }; \
python3 /tools/seam-check.py /tmp/out.json --min-chunks $3'"
}
gpu_idle() {
local used
used=$("${SSH[@]}" "nvidia-smi -i $GPU --query-gpu=memory.used --format=csv,noheader,nounits") \
|| fail "$1" "could not read GPU $GPU memory on fv-ml1"
used=${used//[!0-9]/}
[ -n "$used" ] && [ "$used" -lt 100 ] \
|| fail "$1" "GPU $GPU is not idle (${used:-?} MiB used); refusing to share a live card"
}
# ── seam ───────────────────────────────────────────────────────────────────
gpu_idle seam
if out=$(gpu_run "$BUILD_DIR/$SEAM_AUDIO_REL" 10 2 2>&1); then pass seam "$(tail -1 <<<"$out")"
else fail seam "$out"; fi
# ── memory ─────────────────────────────────────────────────────────────────
"${SSH[@]}" "test -s $MEM_AUDIO" || fail memory "memory audio $MEM_AUDIO not found on fv-ml1"
gpu_idle memory
"${SSH[@]}" "nohup nvidia-smi -i $GPU --query-compute-apps=pid,used_memory \
--format=csv,noheader,nounits -lms 200 </dev/null >$SAMPLES 2>/dev/null & echo \$! >$SAMPLES.pid" \
|| fail memory "could not start the nvidia-smi sampler"
record run "$CNAME" "transient $TAG on GPU $GPU, --chunk-len 120, env ro; removed on exit"
rc=0; out=$(gpu_run "$MEM_AUDIO" 120 2 2>&1) || rc=$? # `||`, not set +e: keeps the ERR trap quiet
"${SSH[@]}" "kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid" || true
read -r pids peak < <("${SSH[@]}" \
"awk -F', *' 'NF==2 {if (!(\$1 in p)) {p[\$1]=1; n++}; if (\$2+0>m) m=\$2+0} END {print n+0, m+0}' $SAMPLES") \
|| fail memory "no GPU samples could be read back from $SAMPLES"
[ "$rc" = 0 ] || fail memory "memory run failed: $out"
[ "$pids" = 1 ] || fail memory "saw $pids processes on GPU $GPU during the run; the peak is not attributable"
if [ "$peak" -le "$BUDGET" ]; then
pass memory "peak $peak MiB <= budget $BUDGET MiB (0.2 s samples, GPU $GPU); $(tail -1 <<<"$out")"
else
fail memory "peak $peak MiB > budget $BUDGET MiB — do NOT deploy beside intern-decision"
fi
summary
echo
echo "VERDICT: PASS. Deploy is manual: stacks/scriberr/patches/README.md § Deploy (SCRIBERR_IMAGE=$TAG)."