docs(scriberr): Parakeet dropout investigation; proposed 0002 (gap retry + model path)
Prime's ask (via the coordinator): investigate the "Parakeet skips stretches of speech" finding, including other Parakeet weights. Investigation only; nothing deployed. Against ground truth (official SCOTUS transcript, Gutenberg #38916) the drops are real: production v3 loses 140 / 66 clean words per transcript on the two public files and ~50 on each private one (Whisper-referenced, Canary-confirmed; adjudicator 129/129 correct on the calibration). Cause: the v2/v3 0.6B weights collapse deep inside long full-attention windows; the encoder output is degraded, the audio alone transcribes fine, and 1.1B TDT/RNNT/CTC and CTC-0.6B never do it. Decoding (CUDA graphs, greedy variants, max_symbols, beam), slice length, local attention, loudness, resampling and a noise floor do not fix it. Controls: A-vs-A, silence positive control (>=15 words 36/36), null control, bootstrap floor. Proposed patch 0002 re-transcribes >=3 s stretches where the audio holds speech but no word came out (-80 to -90 % lost words on all four recordings, lower WER, no invented text, +10 MiB) and adds an explicit PARAKEET_MODEL_PATH with the loaded model recorded in JSON and ModelUsed. Reviewed at high effort, all findings fixed; built and tested as scriberr:local-blackwell-a353078-dropout2, not deployed. scriberr-rebuild: --patches takes DIR[:DIR...]; embeds and seam-checks both Parakeet scripts (seam-check --standard for the short-audio one).
This commit is contained in:
@@ -31,6 +31,12 @@
|
||||
# Usage:
|
||||
# scripts/scriberr-rebuild [--sha SHA40] [--suffix NAME] [--gpu N]
|
||||
# [--budget MIB] [--memory-audio PATH] [--reuse-image]
|
||||
# [--patches DIR]
|
||||
#
|
||||
# --patches DIR[:DIR...] applies every *.patch in those directories, sorted by file
|
||||
# name, instead of the carried set in stacks/scriberr/patches. To test-build a
|
||||
# proposed patch on top of the carried ones, under its own --suffix:
|
||||
# --patches stacks/scriberr/patches:stacks/scriberr/patches/proposed
|
||||
#
|
||||
# Defaults: --sha PINNED_SHA below, --suffix slicer1, --gpu 3, --budget 5496,
|
||||
# --memory-audio the public 30-min SCOTUS fixture. The GPU must have >= 20 GB
|
||||
@@ -44,10 +50,11 @@ HOST=${SCRIBERR_REBUILD_HOST:-infra-ops@10.251.50.54}
|
||||
ENV_DIR=/tank/scriberr/whisperx-env # live env, always mounted READ-ONLY
|
||||
TOOLS=/opt/docker/src/scriberr-rebuild # fixtures + seam checker on fv-ml1
|
||||
SCRIPT_REL=internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py
|
||||
STD_REL=internal/transcription/adapters/py/nvidia/parakeet_transcribe.py
|
||||
TEST_REL=internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py
|
||||
SEAM_AUDIO_REL=tests/data/AMI-Corpus-IB4002.Mix-Headset-clip.wav
|
||||
|
||||
SHA=$PINNED_SHA SUFFIX=slicer1 GPU=3 BUDGET=5496 REUSE_IMAGE=0
|
||||
SHA=$PINNED_SHA SUFFIX=slicer1 GPU=3 BUDGET=5496 REUSE_IMAGE=0 PATCH_DIR_ARG=""
|
||||
MEM_AUDIO=$TOOLS/fixtures/scotus-22-451-first30m.wav
|
||||
while [ $# -gt 0 ]; do
|
||||
case $1 in
|
||||
@@ -57,6 +64,7 @@ while [ $# -gt 0 ]; do
|
||||
--budget) BUDGET=$2; shift 2 ;;
|
||||
--memory-audio) MEM_AUDIO=$2; shift 2 ;;
|
||||
--reuse-image) REUSE_IMAGE=1; shift ;;
|
||||
--patches) PATCH_DIR_ARG=$2; shift 2 ;;
|
||||
-h|--help) sed -n '2,/^set -euo/p' "$0" | sed '$d; s/^# \{0,1\}//'; exit 0 ;;
|
||||
*) echo "unknown argument: $1 (see --help)" >&2; exit 2 ;;
|
||||
esac
|
||||
@@ -66,8 +74,10 @@ done
|
||||
[[ $GPU =~ ^[0-9]+$ && $BUDGET =~ ^[0-9]+$ ]] || { echo "--gpu and --budget must be integers" >&2; exit 2; }
|
||||
|
||||
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
PATCH_DIR=$REPO_ROOT/stacks/scriberr/patches
|
||||
mapfile -t PATCHES < <(find "$PATCH_DIR" -maxdepth 1 -name '*.patch' | sort)
|
||||
PATCH_DIR=${PATCH_DIR_ARG:-$REPO_ROOT/stacks/scriberr/patches}
|
||||
IFS=: read -r -a PATCH_DIRS <<<"$PATCH_DIR"
|
||||
mapfile -t PATCHES < <(for d in "${PATCH_DIRS[@]}"; do find "$d" -maxdepth 1 -name '*.patch'; done \
|
||||
| awk -F/ '{print $NF "\t" $0}' | sort | cut -f2-)
|
||||
[ ${#PATCHES[@]} -gt 0 ] || { echo "no patches in $PATCH_DIR" >&2; exit 2; }
|
||||
# The only paths a reused build dir may differ from upstream in.
|
||||
mapfile -t PATCHED_PATHS < <(sed -n 's#^+++ b/##p' "${PATCHES[@]}" | sort -u)
|
||||
@@ -192,14 +202,14 @@ else
|
||||
fi
|
||||
|
||||
# ── embed ──────────────────────────────────────────────────────────────────
|
||||
if out=$(remote "$TAG" "$BUILD_DIR" "$SCRIPT_REL" 2>&1 <<'EOF'
|
||||
if out=$(remote "$TAG" "$BUILD_DIR" "$SCRIPT_REL" "$STD_REL" 2>&1 <<'EOF'
|
||||
docker run --rm -v "$2":/src:ro --entrypoint python3 "$1" -c "
|
||||
import sys
|
||||
script = open('/src/$3', 'rb').read()
|
||||
sys.exit(0 if script in open('/app/scriberr', 'rb').read() else 1)"
|
||||
binary = open('/app/scriberr', 'rb').read()
|
||||
sys.exit(0 if all(open('/src/' + f, 'rb').read() in binary for f in ('$3', '$4')) else 1)"
|
||||
EOF
|
||||
); then
|
||||
pass embed "patched $(basename "$SCRIPT_REL") is byte-identical inside /app/scriberr"
|
||||
pass embed "both Parakeet scripts are byte-identical inside /app/scriberr"
|
||||
else
|
||||
fail embed "the binary does not embed the patched script ${out:+($out)}"
|
||||
fi
|
||||
@@ -240,6 +250,12 @@ gpu_idle() { # "room", not "idle": Scriberr itself may be running a job on this
|
||||
gpu_idle seam
|
||||
if out=$(gpu_run "$BUILD_DIR/$SEAM_AUDIO_REL" 10 2 2>&1); then pass seam "$(tail -1 <<<"$out")"
|
||||
else fail seam "$out"; fi
|
||||
# The short-audio script (files under PARAKEET_CHUNK_THRESHOLD_SECS) has its own entry point.
|
||||
if out=$("${SSH[@]}" "$DOCKER_RUN --name $CNAME --gpus '\"device=$GPU\"' -e NVIDIA_VISIBLE_DEVICES=$GPU \
|
||||
-v $BUILD_DIR/tests/data:/audio:ro $TAG -c 'cd /tmp && $UVRUN python /src/$STD_REL /audio/$(basename "$SEAM_AUDIO_REL") \
|
||||
--output /tmp/out.json --context-left 256 --context-right 256 >/tmp/run.log 2>&1 || { tail -5 /tmp/run.log; exit 1; }; \
|
||||
python3 /tools/seam-check.py /tmp/out.json --standard'" 2>&1); then pass seam-short "$(tail -1 <<<"$out")"
|
||||
else fail seam-short "$out"; fi
|
||||
|
||||
# ── memory ─────────────────────────────────────────────────────────────────
|
||||
"${SSH[@]}" "test -s $MEM_AUDIO" || fail memory "memory audio $MEM_AUDIO not found on fv-ml1"
|
||||
|
||||
@@ -7,7 +7,8 @@ job. This checks the shape Go reads plus the stitching invariants the slicer
|
||||
patch promises. Stdlib only, so it runs under any python3. Prints counts, never
|
||||
transcript text.
|
||||
|
||||
usage: scriberr-seam-check.py RESULT.json [--min-chunks N]
|
||||
usage: scriberr-seam-check.py RESULT.json [--min-chunks N] [--standard]
|
||||
--standard the short-audio script's result (no buffered/num_chunks keys)
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
@@ -42,6 +43,8 @@ def main():
|
||||
parser.add_argument("result", help="result JSON written by parakeet_transcribe_buffered.py")
|
||||
parser.add_argument("--min-chunks", type=int, default=1,
|
||||
help="fail unless the run used at least this many chunks")
|
||||
parser.add_argument("--standard", action="store_true",
|
||||
help="the short-audio script's result: no buffered/num_chunks keys")
|
||||
args = parser.parse_args()
|
||||
min_chunks = args.min_chunks
|
||||
try:
|
||||
@@ -56,12 +59,13 @@ def main():
|
||||
for key, kind in required.items():
|
||||
if not isinstance(data.get(key), kind):
|
||||
fail(f"'{key}' missing or not {kind.__name__}")
|
||||
if data.get("buffered") is not True:
|
||||
fail("'buffered' is not true")
|
||||
if not isinstance(data.get("chunk_duration_secs"), NUMBER):
|
||||
fail("'chunk_duration_secs' is not a number")
|
||||
if type(data.get("num_chunks")) is not int or data["num_chunks"] < min_chunks:
|
||||
fail(f"'num_chunks' is not an integer >= {min_chunks}")
|
||||
if not args.standard:
|
||||
if data.get("buffered") is not True:
|
||||
fail("'buffered' is not true")
|
||||
if not isinstance(data.get("chunk_duration_secs"), NUMBER):
|
||||
fail("'chunk_duration_secs' is not a number")
|
||||
if type(data.get("num_chunks")) is not int or data["num_chunks"] < min_chunks:
|
||||
fail(f"'num_chunks' is not an integer >= {min_chunks}")
|
||||
|
||||
words, segments = data["word_timestamps"], data["segment_timestamps"]
|
||||
if not words or not data["transcription"].strip():
|
||||
@@ -79,7 +83,8 @@ def main():
|
||||
fail("segments do not cover the stitched words exactly once, in order")
|
||||
|
||||
print(f"SEAM OK: {len(words)} words, {len(segments)} segments, "
|
||||
f"{data['num_chunks']} chunks, cuts at {len(data.get('cut_times', []))} points")
|
||||
f"{data.get('num_chunks', 1)} chunks, cuts at {len(data.get('cut_times', []))} points"
|
||||
f"{', model ' + data['model'] if data.get('model') else ''}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
Reference in New Issue
Block a user