#!/usr/bin/env bash # scriberr-rebuild — rebuild Scriberr's Blackwell image at a PINNED upstream sha # with our patches applied, then prove the result before anyone deploys it. # # Runs on nh3-dev and drives fv-ml1 over ssh. Deploy is a SEPARATE manual step # (stacks/scriberr/patches/README.md § Deploy); this script never touches the # live container, its .env, or GPU 1. # # Why this exists: upstream publishes no sm_120 image, so we build from source, # and we carry a patch to the Parakeet slicer (stacks/scriberr/patches/). Upstream # moves slowly, so an upgrade should be one command plus a verdict. # # Stages (each prints PASS/FAIL; the first FAIL stops the run): # clone clean shallow clone of upstream at the pinned sha, in a NEW dir # /opt/docker/src/scriberr-- (reused only if it already # holds that sha with every patch applied) # patch `git apply --check` then `git apply`, patch by patch; a conflict # stops the run loudly and names the patch # build docker build -f Dockerfile.cuda.12.9 -t scriberr:local-blackwell-- # — a DISTINCT tag, so the running image is never overwritten # embed the patched script's exact bytes are inside the new Go binary # (Scriberr rewrites the env's copy from this embed on every start) # unit the slicer's pure-function tests, under the live env's numpy/librosa # seam patched script, production invocation, short fixture at # --chunk-len 10, JSON validated against the Go struct # memory same on a long recording at --chunk-len 120 on a GPU with room # (>= 20 GB free); nvidia-smi sampled every 0.2 s, and only this # run's own container PIDs count, so a Scriberr job on the same # card does not pollute the peak. Per-process peak <= the budget # # Usage: # scripts/scriberr-rebuild [--sha SHA40] [--suffix NAME] [--gpu N] # [--budget MIB] [--memory-audio PATH] [--reuse-image] # # Defaults: --sha PINNED_SHA below, --suffix slicer1, --gpu 3, --budget 5496, # --memory-audio the public 30-min SCOTUS fixture. The GPU must have >= 20 GB # free, which in practice means GPU 3 (Scriberr's own card since 2026-09-30, # idle at 0 MiB, ~5.5 GB during a job); GPUs 0-2 are full of vLLM seats. set -euo pipefail PINNED_SHA=a353078fd96b8aca4002681813524b7397c90df1 # upstream HEAD 2026-09-20 UPSTREAM=https://github.com/rishikanthc/Scriberr.git HOST=${SCRIBERR_REBUILD_HOST:-infra-ops@10.251.50.54} ENV_DIR=/tank/scriberr/whisperx-env # live env, always mounted READ-ONLY TOOLS=/opt/docker/src/scriberr-rebuild # fixtures + seam checker on fv-ml1 SCRIPT_REL=internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py TEST_REL=internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py SEAM_AUDIO_REL=tests/data/AMI-Corpus-IB4002.Mix-Headset-clip.wav SHA=$PINNED_SHA SUFFIX=slicer1 GPU=3 BUDGET=5496 REUSE_IMAGE=0 MEM_AUDIO=$TOOLS/fixtures/scotus-22-451-first30m.wav while [ $# -gt 0 ]; do case $1 in --sha) SHA=$2; shift 2 ;; --suffix) SUFFIX=$2; shift 2 ;; --gpu) GPU=$2; shift 2 ;; --budget) BUDGET=$2; shift 2 ;; --memory-audio) MEM_AUDIO=$2; shift 2 ;; --reuse-image) REUSE_IMAGE=1; shift ;; -h|--help) sed -n '2,/^set -euo/p' "$0" | sed '$d; s/^# \{0,1\}//'; exit 0 ;; *) echo "unknown argument: $1 (see --help)" >&2; exit 2 ;; esac done [[ $SHA =~ ^[0-9a-f]{40}$ ]] || { echo "--sha must be a full 40-char sha (GitHub fetches by full sha)" >&2; exit 2; } [[ $SUFFIX =~ ^[a-z0-9][a-z0-9.-]*$ ]] || { echo "--suffix must be [a-z0-9.-]" >&2; exit 2; } [[ $GPU =~ ^[0-9]+$ && $BUDGET =~ ^[0-9]+$ ]] || { echo "--gpu and --budget must be integers" >&2; exit 2; } REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" PATCH_DIR=$REPO_ROOT/stacks/scriberr/patches mapfile -t PATCHES < <(find "$PATCH_DIR" -maxdepth 1 -name '*.patch' | sort) [ ${#PATCHES[@]} -gt 0 ] || { echo "no patches in $PATCH_DIR" >&2; exit 2; } # The only paths a reused build dir may differ from upstream in. mapfile -t PATCHED_PATHS < <(sed -n 's#^+++ b/##p' "${PATCHES[@]}" | sort -u) SHA7=${SHA:0:7} TAG=scriberr:local-blackwell-$SHA7-$SUFFIX BUILD_DIR=/opt/docker/src/scriberr-$SHA7-$SUFFIX PATCH_SUM=$(cat "${PATCHES[@]}" | sha256sum | cut -c1-16) CNAME=scriberr-rebuild-$SHA7-$SUFFIX # every GPU container, so cleanup can find it SAMPLES=/tmp/$CNAME.mem.csv BUILD_LOG=/tmp/$CNAME.build.log MIN_FREE_GB=20 # under Docker's root dir (zroot); an image adds ~0.1-6 GB # One multiplexed ssh connection serves the ~15 remote calls of a run. CM=(-o BatchMode=yes -o ControlMaster=auto -o "ControlPath=$HOME/.ssh/cm-%C" -o ControlPersist=120) SSH=(ssh "${CM[@]}" "$HOST") SCP=(scp -q "${CM[@]}") RESULTS=() pass() { RESULTS+=("PASS $1 $2"); echo "== PASS $1: $2"; } fail() { RESULTS+=("FAIL $1 $2"); echo "== FAIL $1: $2" >&2 summary; exit 1 } summary() { echo; echo "scriberr-rebuild upstream=$SHA7 patches=$PATCH_SUM tag=$TAG" printf ' %s\n' "${RESULTS[@]}" } record() { # best effort: the audit trail must not become a failure point "$REPO_ROOT/scripts/ops-log" record --host fv-ml1 --action "$1" --target "$2" \ --detail "$3" >/dev/null 2>&1 || echo "(ops-log record failed; continuing)" >&2 } remote() { "${SSH[@]}" bash -s -- "$@"; } # Whatever happens (a FAIL, Ctrl-C, a dropped connection), never leave the # nvidia-smi sampler or a transient GPU container behind on fv-ml1. cleanup() { "${SSH[@]}" "[ -s $SAMPLES.pid ] && kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid; \ docker rm -f $CNAME >/dev/null 2>&1; true" 2>/dev/null || true } trap cleanup EXIT trap 'exit 130' INT TERM # An unguarded remote call that fails must still say so and print the table. set -E trap 'echo "== ABORT: unexpected failure at line $LINENO (see output above)" >&2; summary' ERR # Every container run: the new image, the live env READ-ONLY, the build tree # read-only, and nothing else writable but the container's own /tmp. DOCKER_RUN="docker run --rm --user 1002:1003 -e HOME=/tmp -e USER=infra-ops -e LOGNAME=infra-ops \ -e PYTHONDONTWRITEBYTECODE=1 -e PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True -e UV_LINK_MODE=copy \ -v $ENV_DIR:/app/whisperx-env:ro -v $BUILD_DIR:/src:ro -v $TOOLS:/tools:ro --entrypoint bash" UVRUN="uv run --native-tls --project /app/whisperx-env/parakeet" # ── clone ────────────────────────────────────────────────────────────────── if out=$(remote "$BUILD_DIR" "$UPSTREAM" "$SHA" "$TOOLS" "${PATCHED_PATHS[@]}" 2>&1 <<'EOF' set -euo pipefail dir=$1 upstream=$2 sha=$3 tools=$4 shift 4 sudo -n install -d -o infra-ops -g infra-ops "$tools" "$tools/fixtures" | cat if [ -e "$dir" ]; then [ "$(git -C "$dir" rev-parse HEAD 2>/dev/null)" = "$sha" ] \ || { echo "$dir exists but is not a checkout of $sha; remove it by hand (sudo -n rm -rf $dir) or pick --suffix"; exit 1; } extra=$({ git -C "$dir" diff --name-only HEAD; git -C "$dir" ls-files --others --exclude-standard; } \ | sort -u | grep -vxF -f <(printf '%s\n' "$@") || true) [ -z "$extra" ] || { echo "$dir has changes outside the patches ($extra); remove it by hand or pick --suffix"; exit 1; } echo "reusing $dir" exit 0 fi sudo -n install -d -o infra-ops -g infra-ops "$dir" | cat cd "$dir" git init -q git remote add origin "$upstream" git fetch -q --depth 1 origin "$sha" \ || { echo "fetch of $sha failed; $dir is left empty, remove it by hand (sudo -n rm -rf $dir)"; exit 1; } git checkout -q --detach FETCH_HEAD [ "$(git rev-parse HEAD)" = "$sha" ] || { echo "checked out $(git rev-parse HEAD), wanted $sha"; exit 1; } echo "cloned $sha into $dir" EOF ); then pass clone "$out"; else fail clone "$out"; fi if [[ $out == reusing* ]]; then REUSED=1; else REUSED=0; record create "$BUILD_DIR" "scriberr-rebuild: clean clone of upstream $SHA7"; fi # ── patch ────────────────────────────────────────────────────────────────── for p in "${PATCHES[@]}"; do name=$(basename "$p") # Already applied (a reused dir)? `apply --reverse --check` succeeds only then. if "${SSH[@]}" "cd $BUILD_DIR && git apply --reverse --check -" <"$p" >/dev/null 2>&1; then pass patch "$name already applied" continue fi if ! out=$("${SSH[@]}" "cd $BUILD_DIR && git apply --check -" <"$p" 2>&1); then if [ "$REUSED" = 1 ]; then fail patch "$name does not apply to the REUSED $BUILD_DIR, which holds an older state of the patches; pick a new --suffix or remove the dir by hand (sudo -n rm -rf $BUILD_DIR): $out" fi fail patch "$name DOES NOT APPLY to upstream $SHA7 — rebase the patch before upgrading: $out" fi "${SSH[@]}" "cd $BUILD_DIR && git apply -" <"$p" || fail patch "$name: git apply failed after a clean --check" record patch "$BUILD_DIR" "scriberr-rebuild: git apply $name" pass patch "$name applied" done # ── build ────────────────────────────────────────────────────────────────── if "${SSH[@]}" "docker image inspect $TAG >/dev/null 2>&1"; then [ "$REUSE_IMAGE" = 1 ] || fail build "$TAG already exists; pass --reuse-image to test it, or pick a new --suffix" pass build "reusing existing $TAG" else free=$("${SSH[@]}" "df -BG --output=avail \$(docker info -f '{{.DockerRootDir}}') | tail -1") \ || fail build "could not read free space on fv-ml1" free=${free//[!0-9]/} [ "${free:-0}" -ge "$MIN_FREE_GB" ] \ || fail build "only ${free:-?} GB free under Docker's root dir, need $MIN_FREE_GB; remove superseded scriberr tags first (patches/README.md)" echo "building $TAG (several minutes; log on fv-ml1 at $BUILD_LOG)" if "${SSH[@]}" "docker build -f $BUILD_DIR/Dockerfile.cuda.12.9 -t $TAG \ --label org.phasefinal.scriberr.upstream=$SHA --label org.phasefinal.scriberr.patches=$PATCH_SUM \ $BUILD_DIR >$BUILD_LOG 2>&1"; then record build "$TAG" "scriberr-rebuild: upstream $SHA7 + patches $PATCH_SUM; old images kept" pass build "$TAG ($free GB was free)" else "${SSH[@]}" "tail -25 $BUILD_LOG" >&2 || true fail build "docker build failed (log tail above)" fi fi # ── embed ────────────────────────────────────────────────────────────────── if out=$(remote "$TAG" "$BUILD_DIR" "$SCRIPT_REL" 2>&1 <<'EOF' docker run --rm -v "$2":/src:ro --entrypoint python3 "$1" -c " import sys script = open('/src/$3', 'rb').read() sys.exit(0 if script in open('/app/scriberr', 'rb').read() else 1)" EOF ); then pass embed "patched $(basename "$SCRIPT_REL") is byte-identical inside /app/scriberr" else fail embed "the binary does not embed the patched script ${out:+($out)}" fi # ── unit ─────────────────────────────────────────────────────────────────── if out=$("${SSH[@]}" "$DOCKER_RUN $TAG -c 'cd /tmp && $UVRUN --with pytest \ python -m pytest -q -p no:cacheprovider /src/$TEST_REL 2>&1 | tail -3'" 2>&1) \ && grep -q ' passed' <<<"$out" && ! grep -Eq 'failed|error' <<<"$out"; then pass unit "$(tail -1 <<<"$out")" else fail unit "$out" fi # The checker travels with this script, so the host copy is refreshed each run. "${SCP[@]}" "$REPO_ROOT/scripts/scriberr-seam-check.py" "$HOST:$TOOLS/seam-check.py" \ || fail seam "could not copy the seam checker to fv-ml1:$TOOLS" record update "$TOOLS/seam-check.py" "scriberr-rebuild: refreshed the seam checker" # One GPU run of the patched script under the production invocation, output # kept inside the container, validated there. Prints the seam checker's line. gpu_run() { # $1 audio path on host, $2 --chunk-len, $3 --min-chunks local audio_dir; audio_dir=$(dirname "$1") "${SSH[@]}" "$DOCKER_RUN --name $CNAME --gpus '\"device=$GPU\"' -e NVIDIA_VISIBLE_DEVICES=$GPU \ -v $audio_dir:/audio:ro $TAG -c 'cd /tmp && $UVRUN python /src/$SCRIPT_REL /audio/$(basename "$1") \ --output /tmp/out.json --chunk-len $2 >/tmp/run.log 2>&1 || { tail -5 /tmp/run.log; exit 1; }; \ python3 /tools/seam-check.py /tmp/out.json --min-chunks $3'" } gpu_idle() { # "room", not "idle": Scriberr itself may be running a job on this card local free free=$("${SSH[@]}" "nvidia-smi -i $GPU --query-gpu=memory.free --format=csv,noheader,nounits") \ || fail "$1" "could not read GPU $GPU memory on fv-ml1" free=${free//[!0-9]/} [ -n "$free" ] && [ "$free" -ge 20000 ] \ || fail "$1" "GPU $GPU has only ${free:-?} MiB free, need 20000; refusing to crowd a live card" } # ── seam ─────────────────────────────────────────────────────────────────── gpu_idle seam if out=$(gpu_run "$BUILD_DIR/$SEAM_AUDIO_REL" 10 2 2>&1); then pass seam "$(tail -1 <<<"$out")" else fail seam "$out"; fi # ── memory ───────────────────────────────────────────────────────────────── "${SSH[@]}" "test -s $MEM_AUDIO" || fail memory "memory audio $MEM_AUDIO not found on fv-ml1" gpu_idle memory # Two host-side loops: every compute process on the card (0.2 s), and the host # PIDs inside this run's container (0.5 s). Only rows whose PID was ours count. "${SSH[@]}" "nohup nvidia-smi -i $GPU --query-compute-apps=pid,used_memory \ --format=csv,noheader,nounits -lms 200 $SAMPLES 2>/dev/null & echo \$! >$SAMPLES.pid; \ : >$SAMPLES.ours; nohup bash -c 'while true; do docker top $CNAME -eo pid 2>/dev/null | tail -n +2 >>$SAMPLES.ours; sleep 0.5; done' \ /dev/null 2>&1 & echo \$! >>$SAMPLES.pid" \ || fail memory "could not start the samplers" record run "$CNAME" "transient $TAG on GPU $GPU, --chunk-len 120, env ro; removed on exit" rc=0; out=$(gpu_run "$MEM_AUDIO" 120 2 2>&1) || rc=$? # `||`, not set +e: keeps the ERR trap quiet "${SSH[@]}" "kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid" || true read -r pids peak < <("${SSH[@]}" \ "awk -F', *' 'FNR==NR {ours[\$1+0]=1; next} NF==2 && ((\$1+0) in ours) {if (!(\$1 in p)) {p[\$1]=1; n++}; if (\$2+0>m) m=\$2+0} END {print n+0, m+0}' $SAMPLES.ours $SAMPLES") \ || fail memory "no GPU samples could be read back from $SAMPLES" [ "$rc" = 0 ] || fail memory "memory run failed: $out" [ "$pids" = 1 ] || fail memory "saw $pids of this run's own processes on GPU $GPU (expected 1); the peak is not attributable" if [ "$peak" -le "$BUDGET" ]; then pass memory "peak $peak MiB <= budget $BUDGET MiB (0.2 s samples, GPU $GPU); $(tail -1 <<<"$out")" else fail memory "peak $peak MiB > budget $BUDGET MiB — do NOT deploy beside intern-decision" fi summary echo echo "VERDICT: PASS. Deploy is manual: stacks/scriberr/patches/README.md § Deploy (SCRIBERR_IMAGE=$TAG)."