Files
esh-pfi-infrastructure/services/parakeet-ab-2026-09-30/code/run_lat.py
T
vh a6c1d3c454 docs(parakeet): seat A/B vs parakeet-unified-en-0.6b - latency is the int8-on-CPU runtime; unified wins WER
A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against
nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image,
k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's
recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights).

- Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%).
- unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s
  (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54).
- unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other,
  -3.2 to -4.4 pp AMI (paired CIs exclude 0).
- Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after
  a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only).
- B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0.

Raw requests, hypotheses, manifests and the full harness under
services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart
from 240 light test requests.
2026-09-30 18:51:44 -07:00

113 lines
5.4 KiB
Python

#!/usr/bin/env python3
"""Latency block on fv-ml1 GPU 3. Host python3, stdlib only.
1. Cold: start each arm fresh and SEQUENTIALLY (so no two warm-ups compete for CPU), record
container-start -> /healthz, then a first-call probe: one never-seen clip per bin, ascending length.
2. Warm single stream: R rounds; in each round the arms run in a seeded shuffled order and each arm
sends all 80 clips once (seeded order, seeded 0-490 ms tail trim so every request is a new length).
Interleaving spreads drift (thermal, a Scriberr job on GPU 3) across arms instead of onto one.
3. Concurrency 4: per arm, per bin, 40 requests from 4 workers.
Writes out/raw/{arms,lat,conc,phases}.jsonl. Memory comes from the separate nvidia-smi sampler.
usage: run_lat.py ROUNDS [ARM ...]
"""
import json
import random
import subprocess
import sys
import time
AB = "/tank/spikes/parakeet-ab"
import os
RAW = os.environ.get("AB_RAW", f"{AB}/out/raw")
M = f"{AB}/models"
ARMS = { # name: (port, arm.sh args, description)
"ab-a1": (18301, ["sherpa", "/tank/parakeet/models", "1"], "A' #1: seat image, v3 int8, 1 thread"),
"ab-a2": (18302, ["sherpa", "/tank/parakeet/models", "1"], "A' #2: identical second instance (floor)"),
"ab-a50": (18303, ["sherpa", "/tank/parakeet/models", "1", "int8.onnx", "50"], "A' +50 ms injected (positive control)"),
"ab-apl": (18309, ["sherpa", "/tank/parakeet/models", "1", "int8.onnx", "0", "1"], "A' with the image's app untouched (null)"),
"ab-at16": (18326, ["sherpa", "/tank/parakeet/models", "16"], "A' with NUM_THREADS=16"),
"ab-c": (18304, ["sherpa", f"{M}/unified-int8", "1"], "C: unified-en int8 (k2-fsa), seat runtime"),
"ab-d": (18305, ["sherpa", f"{M}/v2-int8", "1"], "D: v2 int8 (k2-fsa), seat runtime"),
"ab-cf32": (18312, ["sherpa", f"{M}/unified-f32", "1", "onnx"], "C-fp32: unified-en fp32 ONNX (k2-fsa export recipe), seat runtime"),
"ab-cf16": (18310, ["sherpa", f"{M}/unified-f16", "1", "fp16.onnx"], "C-fp16: unified-en fp16 ONNX (pre_encode kept fp32), seat runtime"),
"ab-af32": (18311, ["sherpa", f"{M}/v3-f32", "1", "onnx"], "A-fp32: the seat's v3 as fp32 ONNX (k2-fsa export recipe), seat runtime"),
"ab-b32": (18306, ["nemo", "fp32", "direct", "1"], "B-fp32: unified-en, NeMo 3.0.0 torch"),
"ab-b16": (18307, ["nemo", "bf16", "direct", "1"], "B-bf16: unified-en, NeMo 3.0.0 torch, autocast bf16"),
"ab-b32c": (18314, ["nemo", "fp32", "direct", "1", "1"], "B-fp32, .nemo restored on CPU then moved to GPU"),
"ab-b16w": (18315, ["nemo", "bf16w", "direct", "1", "1"], "B-bf16w: encoder/decoder/joint weights in bf16, CPU load"),
}
BINS = ["b1_3", "b3_8", "b8_20", "b20_60"]
def sh(cmd, **kw):
return subprocess.run(cmd, capture_output=True, text=True, **kw)
def log(fn, d):
with open(f"{RAW}/{fn}", "a") as f:
f.write(json.dumps(d) + "\n")
def url(name):
return f"http://127.0.0.1:{ARMS[name][0]}/v1/audio/transcriptions"
def bench(*args):
r = sh(["python3", f"{AB}/code/bench.py", *args])
if r.returncode:
print("bench failed", args, r.stderr[-500:], flush=True)
def main():
R = int(sys.argv[1])
arms = sys.argv[2:] or list(ARMS)
sys.path.insert(0, f"{AB}/code")
from bench import post, hostpath
lat = {json.loads(l)["id"]: json.loads(l) for l in open(f"{AB}/data/lat.jsonl")}
# 1. cold
for name in arms:
port, args, desc = ARMS[name]
sh(["docker", "rm", "-f", name])
t = time.time()
r = sh([f"{AB}/code/arm.sh", args[0], name, str(port), *args[1:]])
if r.returncode:
print("arm failed", name, r.stderr, flush=True)
continue
_, _, t0, t1, cold = r.stdout.split()
pid = sh(["docker", "top", name, "-eo", "pid"]).stdout.split()[-1]
log("arms.jsonl", dict(arm=name, desc=desc, port=port, args=args, pid=int(pid), t_started=float(t0), t_ready=float(t1), cold_s=float(cold)))
time.sleep(3) # at-rest memory window
log("phases.jsonl", dict(arm=name, phase="rest-after-warmup", t0=time.time() - 3, t1=time.time()))
for tag in BINS: # first call at each length: clip _10 of each bin (never sent before to this container)
uid = f"{tag}_10"
res = post(url(name), open(hostpath(lat[uid]["wav"]), "rb").read())
res.pop("text", None)
log("first.jsonl", dict(arm=name, id=uid, bin=tag, dur=lat[uid]["dur"], t_wall=time.time(), **res))
print(f"cold {name} {cold}s", flush=True)
# 2. warm single stream, interleaved
t_ss = time.time()
for k in range(R):
order = list(arms)
random.Random(f"round-{k}").shuffle(order)
for name in order:
t0 = time.time()
bench("lat", url(name), name, f"{AB}/data/lat.jsonl", f"{RAW}/lat.jsonl", "--rounds", "1",
"--round-offset", str(k), "--jitter-ms", "490")
log("phases.jsonl", dict(arm=name, phase=f"lat-r{k}", t0=t0, t1=time.time()))
print(f"round {k} done {time.time() - t_ss:.0f}s", flush=True)
# 3. concurrency 4
for name in arms:
t0 = time.time()
bench("conc", url(name), name, f"{AB}/data/lat.jsonl", f"{RAW}/conc.jsonl", "--per-bin", "40",
"--workers", "4", "--jitter-ms", "490")
log("phases.jsonl", dict(arm=name, phase="conc4", t0=t0, t1=time.time()))
print(f"conc {name} done", flush=True)
print("LATENCY BLOCK DONE", flush=True)
if __name__ == "__main__":
main()