feat(scriberr): overlap-and-stitch Parakeet slicer patch, rebuild script, bench

Carry patches/0001 on our Scriberr build (upstream a353078): adjacent
buffered chunks overlap by 4 s inside --chunk-len and hand over at a word
both chunks transcribed alike, instead of cutting at fixed marks with no
overlap. Pause-aware cutting is included as an opt-in (--pause-search);
it measured neutral once the stitch was right. The Go<->Python CLI and
JSON seam is unchanged.

Bench (4 recordings, 118 min, 3 cut placements each, against a no-cut
whole-file reference; metrics only, private audio stays on fv-ml1):
cuts with an error within +-3 s fall from 52% (93/179) to 22% (41/184)
against a 19% background; floor +-0.08. Positive control: upstream's
cutter +0.33 over background. A-vs-A byte-identical in-process and
across CLI processes. Peak GPU memory unchanged at 5,496 MiB (n=3).
Also found: Parakeet skips runs of >=10 words mid-chunk with any
slicer, upstream's included; not addressed here.

scripts/scriberr-rebuild clones a pinned upstream sha into a new
/opt/docker/src dir, git-apply-checks the patches, builds a distinct
tag, and checks embed, unit tests, the JSON seam (scriberr-seam-check.py)
and the memory budget on idle GPU 3. Deploy stays manual. The upstream
PR is prepared under patches/upstream-pr/ and not opened.
This commit is contained in:
vh
2026-09-30 12:10:32 -07:00
parent ab62644315
commit ee3db68db1
8 changed files with 1556 additions and 7 deletions
+263
View File
@@ -0,0 +1,263 @@
#!/usr/bin/env bash
# scriberr-rebuild — rebuild Scriberr's Blackwell image at a PINNED upstream sha
# with our patches applied, then prove the result before anyone deploys it.
#
# Runs on nh3-dev and drives fv-ml1 over ssh. Deploy is a SEPARATE manual step
# (stacks/scriberr/patches/README.md § Deploy); this script never touches the
# live container, its .env, or GPU 1.
#
# Why this exists: upstream publishes no sm_120 image, so we build from source,
# and we carry a patch to the Parakeet slicer (stacks/scriberr/patches/). Upstream
# moves slowly, so an upgrade should be one command plus a verdict.
#
# Stages (each prints PASS/FAIL; the first FAIL stops the run):
# clone clean shallow clone of upstream at the pinned sha, in a NEW dir
# /opt/docker/src/scriberr-<sha7>-<suffix> (reused only if it already
# holds that sha with every patch applied)
# patch `git apply --check` then `git apply`, patch by patch; a conflict
# stops the run loudly and names the patch
# build docker build -f Dockerfile.cuda.12.9 -t scriberr:local-blackwell-<sha7>-<suffix>
# — a DISTINCT tag, so the running image is never overwritten
# embed the patched script's exact bytes are inside the new Go binary
# (Scriberr rewrites the env's copy from this embed on every start)
# unit the slicer's pure-function tests, under the live env's numpy/librosa
# seam patched script, production invocation, short fixture at
# --chunk-len 10, JSON validated against the Go struct
# memory same on a long recording at --chunk-len 120 on an IDLE GPU,
# nvidia-smi sampled every 0.2 s; per-process peak <= the budget
#
# Usage:
# scripts/scriberr-rebuild [--sha SHA40] [--suffix NAME] [--gpu N]
# [--budget MIB] [--memory-audio PATH] [--reuse-image]
#
# Defaults: --sha PINNED_SHA below, --suffix slicer1, --gpu 3, --budget 5496,
# --memory-audio the public 30-min SCOTUS fixture. The GPU must be idle
# (< 100 MiB used), which in practice means GPU 3; GPUs 0-2 run live seats.
set -euo pipefail
PINNED_SHA=a353078fd96b8aca4002681813524b7397c90df1 # upstream HEAD 2026-09-20
UPSTREAM=https://github.com/rishikanthc/Scriberr.git
HOST=${SCRIBERR_REBUILD_HOST:-infra-ops@10.251.50.54}
ENV_DIR=/tank/scriberr/whisperx-env # live env, always mounted READ-ONLY
TOOLS=/opt/docker/src/scriberr-rebuild # fixtures + seam checker on fv-ml1
SCRIPT_REL=internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py
TEST_REL=internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py
SEAM_AUDIO_REL=tests/data/AMI-Corpus-IB4002.Mix-Headset-clip.wav
SHA=$PINNED_SHA SUFFIX=slicer1 GPU=3 BUDGET=5496 REUSE_IMAGE=0
MEM_AUDIO=$TOOLS/fixtures/scotus-22-451-first30m.wav
while [ $# -gt 0 ]; do
case $1 in
--sha) SHA=$2; shift 2 ;;
--suffix) SUFFIX=$2; shift 2 ;;
--gpu) GPU=$2; shift 2 ;;
--budget) BUDGET=$2; shift 2 ;;
--memory-audio) MEM_AUDIO=$2; shift 2 ;;
--reuse-image) REUSE_IMAGE=1; shift ;;
-h|--help) sed -n '2,/^set -euo/p' "$0" | sed '$d; s/^# \{0,1\}//'; exit 0 ;;
*) echo "unknown argument: $1 (see --help)" >&2; exit 2 ;;
esac
done
[[ $SHA =~ ^[0-9a-f]{40}$ ]] || { echo "--sha must be a full 40-char sha (GitHub fetches by full sha)" >&2; exit 2; }
[[ $SUFFIX =~ ^[a-z0-9][a-z0-9.-]*$ ]] || { echo "--suffix must be [a-z0-9.-]" >&2; exit 2; }
[[ $GPU =~ ^[0-9]+$ && $BUDGET =~ ^[0-9]+$ ]] || { echo "--gpu and --budget must be integers" >&2; exit 2; }
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
PATCH_DIR=$REPO_ROOT/stacks/scriberr/patches
mapfile -t PATCHES < <(find "$PATCH_DIR" -maxdepth 1 -name '*.patch' | sort)
[ ${#PATCHES[@]} -gt 0 ] || { echo "no patches in $PATCH_DIR" >&2; exit 2; }
# The only paths a reused build dir may differ from upstream in.
mapfile -t PATCHED_PATHS < <(sed -n 's#^+++ b/##p' "${PATCHES[@]}" | sort -u)
SHA7=${SHA:0:7}
TAG=scriberr:local-blackwell-$SHA7-$SUFFIX
BUILD_DIR=/opt/docker/src/scriberr-$SHA7-$SUFFIX
PATCH_SUM=$(cat "${PATCHES[@]}" | sha256sum | cut -c1-16)
CNAME=scriberr-rebuild-$SHA7-$SUFFIX # every GPU container, so cleanup can find it
SAMPLES=/tmp/$CNAME.mem.csv
BUILD_LOG=/tmp/$CNAME.build.log
MIN_FREE_GB=20 # under Docker's root dir (zroot); an image adds ~0.1-6 GB
# One multiplexed ssh connection serves the ~15 remote calls of a run.
CM=(-o BatchMode=yes -o ControlMaster=auto -o "ControlPath=$HOME/.ssh/cm-%C" -o ControlPersist=120)
SSH=(ssh "${CM[@]}" "$HOST")
SCP=(scp -q "${CM[@]}")
RESULTS=()
pass() { RESULTS+=("PASS $1 $2"); echo "== PASS $1: $2"; }
fail() {
RESULTS+=("FAIL $1 $2"); echo "== FAIL $1: $2" >&2
summary; exit 1
}
summary() {
echo; echo "scriberr-rebuild upstream=$SHA7 patches=$PATCH_SUM tag=$TAG"
printf ' %s\n' "${RESULTS[@]}"
}
record() { # best effort: the audit trail must not become a failure point
"$REPO_ROOT/scripts/ops-log" record --host fv-ml1 --action "$1" --target "$2" \
--detail "$3" >/dev/null 2>&1 || echo "(ops-log record failed; continuing)" >&2
}
remote() { "${SSH[@]}" bash -s -- "$@"; }
# Whatever happens (a FAIL, Ctrl-C, a dropped connection), never leave the
# nvidia-smi sampler or a transient GPU container behind on fv-ml1.
cleanup() {
"${SSH[@]}" "[ -s $SAMPLES.pid ] && kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid; \
docker rm -f $CNAME >/dev/null 2>&1; true" 2>/dev/null || true
}
trap cleanup EXIT
trap 'exit 130' INT TERM
# An unguarded remote call that fails must still say so and print the table.
set -E
trap 'echo "== ABORT: unexpected failure at line $LINENO (see output above)" >&2; summary' ERR
# Every container run: the new image, the live env READ-ONLY, the build tree
# read-only, and nothing else writable but the container's own /tmp.
DOCKER_RUN="docker run --rm --user 1002:1003 -e HOME=/tmp -e USER=infra-ops -e LOGNAME=infra-ops \
-e PYTHONDONTWRITEBYTECODE=1 -e PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True -e UV_LINK_MODE=copy \
-v $ENV_DIR:/app/whisperx-env:ro -v $BUILD_DIR:/src:ro -v $TOOLS:/tools:ro --entrypoint bash"
UVRUN="uv run --native-tls --project /app/whisperx-env/parakeet"
# ── clone ──────────────────────────────────────────────────────────────────
if out=$(remote "$BUILD_DIR" "$UPSTREAM" "$SHA" "$TOOLS" "${PATCHED_PATHS[@]}" 2>&1 <<'EOF'
set -euo pipefail
dir=$1 upstream=$2 sha=$3 tools=$4
shift 4
sudo -n install -d -o infra-ops -g infra-ops "$tools" "$tools/fixtures" | cat
if [ -e "$dir" ]; then
[ "$(git -C "$dir" rev-parse HEAD 2>/dev/null)" = "$sha" ] \
|| { echo "$dir exists but is not a checkout of $sha; remove it by hand (sudo -n rm -rf $dir) or pick --suffix"; exit 1; }
extra=$({ git -C "$dir" diff --name-only HEAD; git -C "$dir" ls-files --others --exclude-standard; } \
| sort -u | grep -vxF -f <(printf '%s\n' "$@") || true)
[ -z "$extra" ] || { echo "$dir has changes outside the patches ($extra); remove it by hand or pick --suffix"; exit 1; }
echo "reusing $dir"
exit 0
fi
sudo -n install -d -o infra-ops -g infra-ops "$dir" | cat
cd "$dir"
git init -q
git remote add origin "$upstream"
git fetch -q --depth 1 origin "$sha" \
|| { echo "fetch of $sha failed; $dir is left empty, remove it by hand (sudo -n rm -rf $dir)"; exit 1; }
git checkout -q --detach FETCH_HEAD
[ "$(git rev-parse HEAD)" = "$sha" ] || { echo "checked out $(git rev-parse HEAD), wanted $sha"; exit 1; }
echo "cloned $sha into $dir"
EOF
); then pass clone "$out"; else fail clone "$out"; fi
if [[ $out == reusing* ]]; then REUSED=1; else REUSED=0; record create "$BUILD_DIR" "scriberr-rebuild: clean clone of upstream $SHA7"; fi
# ── patch ──────────────────────────────────────────────────────────────────
for p in "${PATCHES[@]}"; do
name=$(basename "$p")
# Already applied (a reused dir)? `apply --reverse --check` succeeds only then.
if "${SSH[@]}" "cd $BUILD_DIR && git apply --reverse --check -" <"$p" >/dev/null 2>&1; then
pass patch "$name already applied"
continue
fi
if ! out=$("${SSH[@]}" "cd $BUILD_DIR && git apply --check -" <"$p" 2>&1); then
if [ "$REUSED" = 1 ]; then
fail patch "$name does not apply to the REUSED $BUILD_DIR, which holds an older state of the patches; pick a new --suffix or remove the dir by hand (sudo -n rm -rf $BUILD_DIR):
$out"
fi
fail patch "$name DOES NOT APPLY to upstream $SHA7 — rebase the patch before upgrading:
$out"
fi
"${SSH[@]}" "cd $BUILD_DIR && git apply -" <"$p" || fail patch "$name: git apply failed after a clean --check"
record patch "$BUILD_DIR" "scriberr-rebuild: git apply $name"
pass patch "$name applied"
done
# ── build ──────────────────────────────────────────────────────────────────
if "${SSH[@]}" "docker image inspect $TAG >/dev/null 2>&1"; then
[ "$REUSE_IMAGE" = 1 ] || fail build "$TAG already exists; pass --reuse-image to test it, or pick a new --suffix"
pass build "reusing existing $TAG"
else
free=$("${SSH[@]}" "df -BG --output=avail \$(docker info -f '{{.DockerRootDir}}') | tail -1") \
|| fail build "could not read free space on fv-ml1"
free=${free//[!0-9]/}
[ "${free:-0}" -ge "$MIN_FREE_GB" ] \
|| fail build "only ${free:-?} GB free under Docker's root dir, need $MIN_FREE_GB; remove superseded scriberr tags first (patches/README.md)"
echo "building $TAG (several minutes; log on fv-ml1 at $BUILD_LOG)"
if "${SSH[@]}" "docker build -f $BUILD_DIR/Dockerfile.cuda.12.9 -t $TAG \
--label org.phasefinal.scriberr.upstream=$SHA --label org.phasefinal.scriberr.patches=$PATCH_SUM \
$BUILD_DIR >$BUILD_LOG 2>&1"; then
record build "$TAG" "scriberr-rebuild: upstream $SHA7 + patches $PATCH_SUM; old images kept"
pass build "$TAG ($free GB was free)"
else
"${SSH[@]}" "tail -25 $BUILD_LOG" >&2 || true
fail build "docker build failed (log tail above)"
fi
fi
# ── embed ──────────────────────────────────────────────────────────────────
if out=$(remote "$TAG" "$BUILD_DIR" "$SCRIPT_REL" 2>&1 <<'EOF'
docker run --rm -v "$2":/src:ro --entrypoint python3 "$1" -c "
import sys
script = open('/src/$3', 'rb').read()
sys.exit(0 if script in open('/app/scriberr', 'rb').read() else 1)"
EOF
); then
pass embed "patched $(basename "$SCRIPT_REL") is byte-identical inside /app/scriberr"
else
fail embed "the binary does not embed the patched script ${out:+($out)}"
fi
# ── unit ───────────────────────────────────────────────────────────────────
if out=$("${SSH[@]}" "$DOCKER_RUN $TAG -c 'cd /tmp && $UVRUN --with pytest \
python -m pytest -q -p no:cacheprovider /src/$TEST_REL 2>&1 | tail -3'" 2>&1) \
&& grep -q ' passed' <<<"$out" && ! grep -Eq 'failed|error' <<<"$out"; then
pass unit "$(tail -1 <<<"$out")"
else
fail unit "$out"
fi
# The checker travels with this script, so the host copy is refreshed each run.
"${SCP[@]}" "$REPO_ROOT/scripts/scriberr-seam-check.py" "$HOST:$TOOLS/seam-check.py" \
|| fail seam "could not copy the seam checker to fv-ml1:$TOOLS"
record update "$TOOLS/seam-check.py" "scriberr-rebuild: refreshed the seam checker"
# One GPU run of the patched script under the production invocation, output
# kept inside the container, validated there. Prints the seam checker's line.
gpu_run() { # $1 audio path on host, $2 --chunk-len, $3 --min-chunks
local audio_dir; audio_dir=$(dirname "$1")
"${SSH[@]}" "$DOCKER_RUN --name $CNAME --gpus '\"device=$GPU\"' -e NVIDIA_VISIBLE_DEVICES=$GPU \
-v $audio_dir:/audio:ro $TAG -c 'cd /tmp && $UVRUN python /src/$SCRIPT_REL /audio/$(basename "$1") \
--output /tmp/out.json --chunk-len $2 >/tmp/run.log 2>&1 || { tail -5 /tmp/run.log; exit 1; }; \
python3 /tools/seam-check.py /tmp/out.json --min-chunks $3'"
}
gpu_idle() {
local used
used=$("${SSH[@]}" "nvidia-smi -i $GPU --query-gpu=memory.used --format=csv,noheader,nounits") \
|| fail "$1" "could not read GPU $GPU memory on fv-ml1"
used=${used//[!0-9]/}
[ -n "$used" ] && [ "$used" -lt 100 ] \
|| fail "$1" "GPU $GPU is not idle (${used:-?} MiB used); refusing to share a live card"
}
# ── seam ───────────────────────────────────────────────────────────────────
gpu_idle seam
if out=$(gpu_run "$BUILD_DIR/$SEAM_AUDIO_REL" 10 2 2>&1); then pass seam "$(tail -1 <<<"$out")"
else fail seam "$out"; fi
# ── memory ─────────────────────────────────────────────────────────────────
"${SSH[@]}" "test -s $MEM_AUDIO" || fail memory "memory audio $MEM_AUDIO not found on fv-ml1"
gpu_idle memory
"${SSH[@]}" "nohup nvidia-smi -i $GPU --query-compute-apps=pid,used_memory \
--format=csv,noheader,nounits -lms 200 </dev/null >$SAMPLES 2>/dev/null & echo \$! >$SAMPLES.pid" \
|| fail memory "could not start the nvidia-smi sampler"
record run "$CNAME" "transient $TAG on GPU $GPU, --chunk-len 120, env ro; removed on exit"
rc=0; out=$(gpu_run "$MEM_AUDIO" 120 2 2>&1) || rc=$? # `||`, not set +e: keeps the ERR trap quiet
"${SSH[@]}" "kill \$(cat $SAMPLES.pid) 2>/dev/null; : >$SAMPLES.pid" || true
read -r pids peak < <("${SSH[@]}" \
"awk -F', *' 'NF==2 {if (!(\$1 in p)) {p[\$1]=1; n++}; if (\$2+0>m) m=\$2+0} END {print n+0, m+0}' $SAMPLES") \
|| fail memory "no GPU samples could be read back from $SAMPLES"
[ "$rc" = 0 ] || fail memory "memory run failed: $out"
[ "$pids" = 1 ] || fail memory "saw $pids processes on GPU $GPU during the run; the peak is not attributable"
if [ "$peak" -le "$BUDGET" ]; then
pass memory "peak $peak MiB <= budget $BUDGET MiB (0.2 s samples, GPU $GPU); $(tail -1 <<<"$out")"
else
fail memory "peak $peak MiB > budget $BUDGET MiB — do NOT deploy beside intern-decision"
fi
summary
echo
echo "VERDICT: PASS. Deploy is manual: stacks/scriberr/patches/README.md § Deploy (SCRIBERR_IMAGE=$TAG)."
+86
View File
@@ -0,0 +1,86 @@
#!/usr/bin/env python3
"""Validate a parakeet_transcribe_buffered.py result against the Go seam.
Scriberr's parakeet_adapter.go (parseResult) unmarshals this JSON into a struct
with typed fields; a float where Go expects an int, or a missing key, fails the
job. This checks the shape Go reads plus the stitching invariants the slicer
patch promises. Stdlib only, so it runs under any python3. Prints counts, never
transcript text.
usage: scriberr-seam-check.py RESULT.json [--min-chunks N]
"""
import argparse
import json
import sys
NUMBER = (int, float)
def fail(msg):
print(f"SEAM FAIL: {msg}")
sys.exit(1)
def check_items(items, text_key, name):
for i, item in enumerate(items):
if not isinstance(item, dict):
fail(f"{name}[{i}] is not an object")
if not isinstance(item.get(text_key), str):
fail(f"{name}[{i}].{text_key} is not a string")
for key in ("start_offset", "end_offset"):
if type(item.get(key)) is not int:
fail(f"{name}[{i}].{key} is not an integer (Go field is int)")
for key in ("start", "end"):
if not isinstance(item.get(key), NUMBER) or isinstance(item.get(key), bool):
fail(f"{name}[{i}].{key} is not a number")
if item["start"] > item["end"]:
fail(f"{name}[{i}] starts after it ends")
def main():
parser = argparse.ArgumentParser(description="Validate a buffered Parakeet result for Go.")
parser.add_argument("result", help="result JSON written by parakeet_transcribe_buffered.py")
parser.add_argument("--min-chunks", type=int, default=1,
help="fail unless the run used at least this many chunks")
args = parser.parse_args()
min_chunks = args.min_chunks
try:
data = json.load(open(args.result, encoding="utf-8"))
except (OSError, ValueError) as e:
fail(f"cannot read {args.result}: {e}")
if not isinstance(data, dict):
fail("the result is not a JSON object")
required = {"transcription": str, "language": str, "word_timestamps": list,
"segment_timestamps": list, "audio_file": str, "model": str}
for key, kind in required.items():
if not isinstance(data.get(key), kind):
fail(f"'{key}' missing or not {kind.__name__}")
if data.get("buffered") is not True:
fail("'buffered' is not true")
if not isinstance(data.get("chunk_duration_secs"), NUMBER):
fail("'chunk_duration_secs' is not a number")
if type(data.get("num_chunks")) is not int or data["num_chunks"] < min_chunks:
fail(f"'num_chunks' is not an integer >= {min_chunks}")
words, segments = data["word_timestamps"], data["segment_timestamps"]
if not words or not data["transcription"].strip():
fail("empty transcript")
check_items(words, "word", "word_timestamps")
check_items(segments, "segment", "segment_timestamps")
starts = [w["start"] for w in words]
if starts != sorted(starts):
fail("word start times go backwards (a stitch repeated or reordered words)")
joined = " ".join(w["word"] for w in words)
if data["transcription"] != joined:
fail("'transcription' is not the stitched words joined by spaces")
if " ".join(s["segment"] for s in segments) != joined:
fail("segments do not cover the stitched words exactly once, in order")
print(f"SEAM OK: {len(words)} words, {len(segments)} segments, "
f"{data['num_chunks']} chunks, cuts at {len(data.get('cut_times', []))} points")
if __name__ == "__main__":
main()