A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image, k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights). - Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%). - unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54). - unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other, -3.2 to -4.4 pp AMI (paired CIs exclude 0). - Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only). - B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0. Raw requests, hypotheses, manifests and the full harness under services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart from 240 light test requests.
113 lines
5.4 KiB
Python
113 lines
5.4 KiB
Python
#!/usr/bin/env python3
|
|
"""Latency block on fv-ml1 GPU 3. Host python3, stdlib only.
|
|
|
|
1. Cold: start each arm fresh and SEQUENTIALLY (so no two warm-ups compete for CPU), record
|
|
container-start -> /healthz, then a first-call probe: one never-seen clip per bin, ascending length.
|
|
2. Warm single stream: R rounds; in each round the arms run in a seeded shuffled order and each arm
|
|
sends all 80 clips once (seeded order, seeded 0-490 ms tail trim so every request is a new length).
|
|
Interleaving spreads drift (thermal, a Scriberr job on GPU 3) across arms instead of onto one.
|
|
3. Concurrency 4: per arm, per bin, 40 requests from 4 workers.
|
|
Writes out/raw/{arms,lat,conc,phases}.jsonl. Memory comes from the separate nvidia-smi sampler.
|
|
usage: run_lat.py ROUNDS [ARM ...]
|
|
"""
|
|
import json
|
|
import random
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
|
|
AB = "/tank/spikes/parakeet-ab"
|
|
import os
|
|
RAW = os.environ.get("AB_RAW", f"{AB}/out/raw")
|
|
M = f"{AB}/models"
|
|
ARMS = { # name: (port, arm.sh args, description)
|
|
"ab-a1": (18301, ["sherpa", "/tank/parakeet/models", "1"], "A' #1: seat image, v3 int8, 1 thread"),
|
|
"ab-a2": (18302, ["sherpa", "/tank/parakeet/models", "1"], "A' #2: identical second instance (floor)"),
|
|
"ab-a50": (18303, ["sherpa", "/tank/parakeet/models", "1", "int8.onnx", "50"], "A' +50 ms injected (positive control)"),
|
|
"ab-apl": (18309, ["sherpa", "/tank/parakeet/models", "1", "int8.onnx", "0", "1"], "A' with the image's app untouched (null)"),
|
|
"ab-at16": (18326, ["sherpa", "/tank/parakeet/models", "16"], "A' with NUM_THREADS=16"),
|
|
"ab-c": (18304, ["sherpa", f"{M}/unified-int8", "1"], "C: unified-en int8 (k2-fsa), seat runtime"),
|
|
"ab-d": (18305, ["sherpa", f"{M}/v2-int8", "1"], "D: v2 int8 (k2-fsa), seat runtime"),
|
|
"ab-cf32": (18312, ["sherpa", f"{M}/unified-f32", "1", "onnx"], "C-fp32: unified-en fp32 ONNX (k2-fsa export recipe), seat runtime"),
|
|
"ab-cf16": (18310, ["sherpa", f"{M}/unified-f16", "1", "fp16.onnx"], "C-fp16: unified-en fp16 ONNX (pre_encode kept fp32), seat runtime"),
|
|
"ab-af32": (18311, ["sherpa", f"{M}/v3-f32", "1", "onnx"], "A-fp32: the seat's v3 as fp32 ONNX (k2-fsa export recipe), seat runtime"),
|
|
"ab-b32": (18306, ["nemo", "fp32", "direct", "1"], "B-fp32: unified-en, NeMo 3.0.0 torch"),
|
|
"ab-b16": (18307, ["nemo", "bf16", "direct", "1"], "B-bf16: unified-en, NeMo 3.0.0 torch, autocast bf16"),
|
|
"ab-b32c": (18314, ["nemo", "fp32", "direct", "1", "1"], "B-fp32, .nemo restored on CPU then moved to GPU"),
|
|
"ab-b16w": (18315, ["nemo", "bf16w", "direct", "1", "1"], "B-bf16w: encoder/decoder/joint weights in bf16, CPU load"),
|
|
}
|
|
BINS = ["b1_3", "b3_8", "b8_20", "b20_60"]
|
|
|
|
|
|
def sh(cmd, **kw):
|
|
return subprocess.run(cmd, capture_output=True, text=True, **kw)
|
|
|
|
|
|
def log(fn, d):
|
|
with open(f"{RAW}/{fn}", "a") as f:
|
|
f.write(json.dumps(d) + "\n")
|
|
|
|
|
|
def url(name):
|
|
return f"http://127.0.0.1:{ARMS[name][0]}/v1/audio/transcriptions"
|
|
|
|
|
|
def bench(*args):
|
|
r = sh(["python3", f"{AB}/code/bench.py", *args])
|
|
if r.returncode:
|
|
print("bench failed", args, r.stderr[-500:], flush=True)
|
|
|
|
|
|
def main():
|
|
R = int(sys.argv[1])
|
|
arms = sys.argv[2:] or list(ARMS)
|
|
sys.path.insert(0, f"{AB}/code")
|
|
from bench import post, hostpath
|
|
lat = {json.loads(l)["id"]: json.loads(l) for l in open(f"{AB}/data/lat.jsonl")}
|
|
|
|
# 1. cold
|
|
for name in arms:
|
|
port, args, desc = ARMS[name]
|
|
sh(["docker", "rm", "-f", name])
|
|
t = time.time()
|
|
r = sh([f"{AB}/code/arm.sh", args[0], name, str(port), *args[1:]])
|
|
if r.returncode:
|
|
print("arm failed", name, r.stderr, flush=True)
|
|
continue
|
|
_, _, t0, t1, cold = r.stdout.split()
|
|
pid = sh(["docker", "top", name, "-eo", "pid"]).stdout.split()[-1]
|
|
log("arms.jsonl", dict(arm=name, desc=desc, port=port, args=args, pid=int(pid), t_started=float(t0), t_ready=float(t1), cold_s=float(cold)))
|
|
time.sleep(3) # at-rest memory window
|
|
log("phases.jsonl", dict(arm=name, phase="rest-after-warmup", t0=time.time() - 3, t1=time.time()))
|
|
for tag in BINS: # first call at each length: clip _10 of each bin (never sent before to this container)
|
|
uid = f"{tag}_10"
|
|
res = post(url(name), open(hostpath(lat[uid]["wav"]), "rb").read())
|
|
res.pop("text", None)
|
|
log("first.jsonl", dict(arm=name, id=uid, bin=tag, dur=lat[uid]["dur"], t_wall=time.time(), **res))
|
|
print(f"cold {name} {cold}s", flush=True)
|
|
|
|
# 2. warm single stream, interleaved
|
|
t_ss = time.time()
|
|
for k in range(R):
|
|
order = list(arms)
|
|
random.Random(f"round-{k}").shuffle(order)
|
|
for name in order:
|
|
t0 = time.time()
|
|
bench("lat", url(name), name, f"{AB}/data/lat.jsonl", f"{RAW}/lat.jsonl", "--rounds", "1",
|
|
"--round-offset", str(k), "--jitter-ms", "490")
|
|
log("phases.jsonl", dict(arm=name, phase=f"lat-r{k}", t0=t0, t1=time.time()))
|
|
print(f"round {k} done {time.time() - t_ss:.0f}s", flush=True)
|
|
|
|
# 3. concurrency 4
|
|
for name in arms:
|
|
t0 = time.time()
|
|
bench("conc", url(name), name, f"{AB}/data/lat.jsonl", f"{RAW}/conc.jsonl", "--per-bin", "40",
|
|
"--workers", "4", "--jitter-ms", "490")
|
|
log("phases.jsonl", dict(arm=name, phase="conc4", t0=t0, t1=time.time()))
|
|
print(f"conc {name} done", flush=True)
|
|
print("LATENCY BLOCK DONE", flush=True)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|