Files
esh-pfi-infrastructure/services/parakeet-ab-2026-09-30/code/analyze_lat.py
T
vh a6c1d3c454 docs(parakeet): seat A/B vs parakeet-unified-en-0.6b - latency is the int8-on-CPU runtime; unified wins WER
A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against
nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image,
k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's
recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights).

- Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%).
- unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s
  (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54).
- unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other,
  -3.2 to -4.4 pp AMI (paired CIs exclude 0).
- Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after
  a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only).
- B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0.

Raw requests, hypotheses, manifests and the full harness under
services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart
from 240 light test requests.
2026-09-30 18:51:44 -07:00

180 lines
8.0 KiB
Python

"""Latency + memory analysis. numpy; run with envs/score.
Single stream: per arm x bin, p50/p90/p99 of server-side (x-ab-decode-ms) and end-to-end (client wall,
new connection per request, fv-ml1 loopback). Deltas vs a reference arm are PAIRED: in round k every arm
got byte-identical input for a clip (the tail trim is seeded by round and clip, never by arm), so the
statistic is the median of per-request differences, with a 95 % bootstrap CI over requests.
Floor = ab-a2 vs ab-a1 (identical second instance). Positive control = ab-a50 (+50 ms injected) vs ab-a1.
Null = ab-apl (the image's own app, no timing header) vs ab-a1 (e2e only).
usage: analyze_lat.py RAW_DIR [REF_ARM]
"""
import collections
import json
import sys
import numpy as np
RAW = sys.argv[1]
REF = sys.argv[2] if len(sys.argv) > 2 else "ab-a1"
BINS = ["b1_3", "b3_8", "b8_20", "b20_60"]
rng = np.random.default_rng(7)
def rows(fn, mode=None):
out = []
try:
for l in open(f"{RAW}/{fn}"):
r = json.loads(l)
if mode is None or r.get("mode") == mode:
out.append(r)
except FileNotFoundError:
pass
return out
def pct(x, q):
return float(np.percentile(np.asarray(x, float), q)) if len(x) else float("nan")
def boot_ci(x, stat=np.median, n=2000):
x = np.asarray(x, float)
if len(x) < 3:
return (float("nan"), float("nan"))
b = [stat(x[rng.integers(0, len(x), len(x))]) for _ in range(n)]
return (float(np.percentile(b, 2.5)), float(np.percentile(b, 97.5)))
def fmt(v):
return "nan" if v != v else f"{v:.1f}"
def single_stream(files, arms_order=None):
lat = [r for f in files for r in rows(f, "lat") if r["status"] == 200]
bad = [r for f in files for r in rows(f, "lat") if r["status"] != 200]
by = collections.defaultdict(list)
for r in lat:
by[(r["arm"], r["bin"])].append(r)
arms = arms_order or sorted({a for a, _ in by})
table = []
for a in arms:
for b in BINS:
xs = by.get((a, b), [])
if not xs:
continue
e = [r["e2e_ms"] for r in xs]
s = [r["server_ms"] for r in xs if r.get("server_ms") is not None]
lo, hi = boot_ci(e)
table.append(dict(arm=a, bin=b, n=len(xs), rounds=len({r["round"] for r in xs}),
e2e_p50=pct(e, 50), e2e_p50_ci=[lo, hi], e2e_p90=pct(e, 90), e2e_p99=pct(e, 99), e2e_max=max(e),
srv_p50=pct(s, 50), srv_p90=pct(s, 90), srv_p99=pct(s, 99),
overhead_p50=pct([r["e2e_ms"] - r["server_ms"] for r in xs if r.get("server_ms") is not None], 50)))
return table, by, bad
def paired(by, a, ref, b, key="e2e_ms"):
ra = {(r["round"], r["id"]): r for r in by.get((a, b), [])}
rr = {(r["round"], r["id"]): r for r in by.get((ref, b), [])}
ks = sorted(set(ra) & set(rr))
if not ks:
return None
for k in ks: # the pairing is only valid if the two requests carried identical input
assert ra[k].get("trim_ms") == rr[k].get("trim_ms"), (a, ref, k)
d = [ra[k][key] - rr[k][key] for k in ks]
lo, hi = boot_ci(d)
return dict(arm=a, ref=ref, bin=b, n_pairs=len(ks), median_diff=float(np.median(d)), ci95=[lo, hi],
p50_arm=pct([ra[k][key] for k in ks], 50), p50_ref=pct([rr[k][key] for k in ks], 50))
def conc():
rs = [r for r in rows("conc.jsonl", "conc")]
bins = {(r["arm"], r["bin"]): r for r in rows("conc.jsonl", "conc-bin")}
by = collections.defaultdict(list)
for r in rs:
by[(r["arm"], r["bin"])].append(r)
out = []
for (a, b), xs in sorted(by.items()):
ok = [r for r in xs if r["status"] == 200]
e = [r["e2e_ms"] for r in ok]
wall = bins[(a, b)]["wall_s"]
audio = sum(r["dur"] - r.get("trim_ms", 0) / 1000 for r in ok)
out.append(dict(arm=a, bin=b, n=len(xs), errors=len(xs) - len(ok), e2e_p50=pct(e, 50), e2e_p90=pct(e, 90),
e2e_p99=pct(e, 99), srv_p50=pct([r["server_ms"] for r in ok if r.get("server_ms")], 50),
req_per_s=round(len(ok) / wall, 2), audio_x_realtime=round(audio / wall, 1)))
return out
def memory():
arms = {r["arm"]: r for r in rows("arms.jsonl")}
phases = rows("phases.jsonl")
# every PID of each container (NeMo forks workers; the CUDA context is in the first python process)
try:
allp = json.load(open(f"{RAW}/pids.json"))
except FileNotFoundError:
allp = {a: [r["pid"]] for a, r in arms.items()}
pid2arm = {p: a for a, ps in allp.items() for p in ps}
for a in arms:
arms[a]["pid"] = next((p for p in allp.get(a, []) if p in pid2arm), arms[a]["pid"])
series = collections.defaultdict(list) # pid -> [(t, MiB)]
import datetime
try:
for l in open(f"{RAW}/mem.csv"):
p = [x.strip() for x in l.split(",")]
if len(p) != 4 or not p[2].isdigit():
continue
pid = int(p[2])
if pid in pid2arm:
t = datetime.datetime.strptime(p[0], "%Y/%m/%d %H:%M:%S.%f").timestamp()
series[pid].append((t, int(p[3])))
except FileNotFoundError:
pass
out = []
for a, r in arms.items():
s = sorted(x for p in allp.get(a, [r["pid"]]) for x in series.get(p, []))
rest = [m for t, m in s if any(ph["arm"] == a and ph["phase"] == "rest-after-warmup" and ph["t0"] <= t <= ph["t1"] + 1 for ph in phases)]
lat_w = [ph for ph in phases if ph["arm"] == a and (ph["phase"].startswith("lat-") or ph["phase"] == "conc4")]
t_lat0 = min((ph["t0"] for ph in lat_w), default=None)
t_lat1 = max((ph["t1"] for ph in lat_w), default=None)
lat_peak = max((m for t, m in s if t_lat0 and t_lat0 <= t <= t_lat1 + 2), default=None)
out.append(dict(arm=a, pid=r["pid"], cold_s=r["cold_s"], rest_after_warmup_mib=max(rest) if rest else None,
peak_utterance_block_mib=lat_peak, peak_whole_life_mib=max((m for _, m in s), default=None),
last_mib=s[-1][1] if s else None, samples=len(s)))
return out
if __name__ == "__main__":
what = sys.argv[3] if len(sys.argv) > 3 else "all"
files = ["lat.jsonl"] + (["lat-live.jsonl"] if what in ("all", "live") else [])
t, by, bad = single_stream(files)
print("## single stream (ms)")
for r in t:
print(f"{r['arm']:10s} {r['bin']:7s} n={r['n']:3d} e2e p50 {fmt(r['e2e_p50'])} [{fmt(r['e2e_p50_ci'][0])},{fmt(r['e2e_p50_ci'][1])}] "
f"p90 {fmt(r['e2e_p90'])} p99 {fmt(r['e2e_p99'])} max {fmt(r['e2e_max'])} | server p50 {fmt(r['srv_p50'])} p90 {fmt(r['srv_p90'])} "
f"p99 {fmt(r['srv_p99'])} | http overhead p50 {fmt(r['overhead_p50'])}")
if bad:
print("NON-200:", collections.Counter((r["arm"], r["status"]) for r in bad))
print(f"\n## paired median difference vs {REF} (e2e ms; 95% bootstrap CI)")
pairs = []
for a in sorted({a for a, _ in by}):
if a == REF:
continue
for b in BINS:
p = paired(by, a, REF, b)
if p:
pairs.append(p)
print(f"{a:10s} {b:7s} pairs={p['n_pairs']:3d} diff {p['median_diff']:+8.1f} [{p['ci95'][0]:+8.1f},{p['ci95'][1]:+8.1f}]")
c = conc()
print("\n## concurrency 4 (ms)")
for r in c:
print(f"{r['arm']:10s} {r['bin']:7s} n={r['n']} err={r['errors']} e2e p50 {fmt(r['e2e_p50'])} p90 {fmt(r['e2e_p90'])} p99 {fmt(r['e2e_p99'])} "
f"| server p50 {fmt(r['srv_p50'])} | {r['req_per_s']} req/s, {r['audio_x_realtime']}x realtime")
first = rows("first.jsonl")
print("\n## first call per bin after cold start (ms) vs that arm's warm e2e p50")
warm = {(r["arm"], r["bin"]): r["e2e_p50"] for r in t}
for r in first:
print(f"{r['arm']:10s} {r['bin']:7s} first {r['e2e_ms']:8.1f} warm p50 {fmt(warm.get((r['arm'], r['bin']), float('nan')))}")
m = memory()
print("\n## GPU memory per process (MiB)")
for r in m:
print(json.dumps(r))
json.dump(dict(single=t, paired=pairs, conc=c, first=first, memory=m), open(f"{RAW}/../latency-summary.json", "w"), indent=1)