A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image, k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights). - Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%). - unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54). - unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other, -3.2 to -4.4 pp AMI (paired CIs exclude 0). - Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only). - B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0. Raw requests, hypotheses, manifests and the full harness under services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart from 240 light test requests.
41 lines
2.6 KiB
Bash
Executable File
41 lines
2.6 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Start one A/B arm as a transient container on fv-ml1 GPU 3 (never 0), bound to 127.0.0.1 only.
|
|
# Prints: <name> <port> <t_started_epoch> <t_ready_epoch> <cold_s> once /healthz answers.
|
|
#
|
|
# arm.sh sherpa NAME PORT MODEL_DIR [NUM_THREADS=1] [SUFFIX=int8.onnx] [DELAY_MS=0] [PLAIN=0]
|
|
# The SEAT'S OWN IMAGE (local/parakeet:sherpa-onnx-v4) with the seat's env (PROVIDER=cuda). The
|
|
# harness app_timed.py is mounted over /app/app.py unless PLAIN=1 (image's app untouched).
|
|
# arm.sh nemo NAME PORT DTYPE PATH_MODE [CUDA_GRAPHS=1] [LOAD_CPU=0] [LOCAL_ATT=]
|
|
# parakeet-unified-en-0.6b (pinned fe53cd88) under NeMo 3.0.0 torch, serve_nemo.py.
|
|
set -euo pipefail
|
|
AB=/tank/spikes/parakeet-ab
|
|
kind=$1 name=$2 port=$3
|
|
common=(--rm -d --name "$name" --gpus '"device=3"' --label ab=parakeet-2026-09-30 -p 127.0.0.1:$port:8000)
|
|
if [ "$kind" = sherpa ]; then
|
|
mdir=$4 nt=${5:-1} suf=${6:-int8.onnx} delay=${7:-0} plain=${8:-0}
|
|
app=(-v "$AB/code/app_timed.py:/app/app.py:ro"); [ "$plain" = 1 ] && app=()
|
|
# The seat's entrypoint.sh only knows the *.int8.onnx names (and would try to download into the
|
|
# read-only mount), so a non-int8 export skips it and execs the same final command it would.
|
|
ep=(); cmd=(); [ "$suf" != int8.onnx ] && { ep=(--entrypoint /opt/venv/bin/uvicorn); cmd=(app:app --host 0.0.0.0 --port 8000); }
|
|
docker run "${common[@]}" -e MODEL_DIR=/models -e PROVIDER=cuda -e NUM_THREADS=$nt -e LOG_LEVEL=INFO \
|
|
-e AB_SUFFIX=$suf -e AB_DELAY_MS=$delay "${app[@]}" -v "$mdir":/models:ro "${ep[@]}" local/parakeet:sherpa-onnx-v4 "${cmd[@]}" >/dev/null
|
|
elif [ "$kind" = nemo ]; then
|
|
dt=$4 pm=$5 cg=${6:-1} lc=${7:-0}
|
|
docker run "${common[@]}" --user 1002:1003 -e HOME=/tmp -e USER=infra-ops -e LOGNAME=infra-ops -e HF_HUB_OFFLINE=1 \
|
|
-e MODEL_PATH=/hf/hub/models--nvidia--parakeet-unified-en-0.6b/snapshots/fe53cd885760c96b6a5f51a0bfd362cb4584a98b/parakeet-unified-en-0.6b.nemo \
|
|
-e DTYPE=$dt -e PATH_MODE=$pm -e CUDA_GRAPHS=$cg -e LOAD_CPU=$lc -e LOCAL_ATT=${8:-} \
|
|
-v /tank/aimodels/huggingface:/hf:ro -v $AB:/ab:ro -v $AB/tmp:/tmp \
|
|
--entrypoint /ab/envs/nemo300-serve/bin/python scriberr:local-blackwell \
|
|
-m uvicorn serve_nemo:app --app-dir /ab/code --host 0.0.0.0 --port 8000 >/dev/null
|
|
else
|
|
echo "usage: arm.sh sherpa|nemo ..." >&2; exit 2
|
|
fi
|
|
for i in $(seq 1 2400); do
|
|
curl -sf -m 2 http://127.0.0.1:$port/healthz >/dev/null 2>&1 && break
|
|
docker inspect "$name" >/dev/null 2>&1 || { echo "$name died" >&2; exit 1; }
|
|
sleep 0.25
|
|
done
|
|
t1=$(date +%s.%N)
|
|
t0=$(date -d "$(docker inspect -f '{{.State.StartedAt}}' "$name")" +%s.%N)
|
|
echo "$name $port $t0 $t1 $(python3 -c "print(round($t1-$t0,1))")"
|