A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image, k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights). - Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%). - unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54). - unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other, -3.2 to -4.4 pp AMI (paired CIs exclude 0). - Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only). - B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0. Raw requests, hypotheses, manifests and the full harness under services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart from 240 light test requests.
180 lines
8.0 KiB
Python
180 lines
8.0 KiB
Python
"""Latency + memory analysis. numpy; run with envs/score.
|
|
|
|
Single stream: per arm x bin, p50/p90/p99 of server-side (x-ab-decode-ms) and end-to-end (client wall,
|
|
new connection per request, fv-ml1 loopback). Deltas vs a reference arm are PAIRED: in round k every arm
|
|
got byte-identical input for a clip (the tail trim is seeded by round and clip, never by arm), so the
|
|
statistic is the median of per-request differences, with a 95 % bootstrap CI over requests.
|
|
Floor = ab-a2 vs ab-a1 (identical second instance). Positive control = ab-a50 (+50 ms injected) vs ab-a1.
|
|
Null = ab-apl (the image's own app, no timing header) vs ab-a1 (e2e only).
|
|
usage: analyze_lat.py RAW_DIR [REF_ARM]
|
|
"""
|
|
import collections
|
|
import json
|
|
import sys
|
|
|
|
import numpy as np
|
|
|
|
RAW = sys.argv[1]
|
|
REF = sys.argv[2] if len(sys.argv) > 2 else "ab-a1"
|
|
BINS = ["b1_3", "b3_8", "b8_20", "b20_60"]
|
|
rng = np.random.default_rng(7)
|
|
|
|
|
|
def rows(fn, mode=None):
|
|
out = []
|
|
try:
|
|
for l in open(f"{RAW}/{fn}"):
|
|
r = json.loads(l)
|
|
if mode is None or r.get("mode") == mode:
|
|
out.append(r)
|
|
except FileNotFoundError:
|
|
pass
|
|
return out
|
|
|
|
|
|
def pct(x, q):
|
|
return float(np.percentile(np.asarray(x, float), q)) if len(x) else float("nan")
|
|
|
|
|
|
def boot_ci(x, stat=np.median, n=2000):
|
|
x = np.asarray(x, float)
|
|
if len(x) < 3:
|
|
return (float("nan"), float("nan"))
|
|
b = [stat(x[rng.integers(0, len(x), len(x))]) for _ in range(n)]
|
|
return (float(np.percentile(b, 2.5)), float(np.percentile(b, 97.5)))
|
|
|
|
|
|
def fmt(v):
|
|
return "nan" if v != v else f"{v:.1f}"
|
|
|
|
|
|
def single_stream(files, arms_order=None):
|
|
lat = [r for f in files for r in rows(f, "lat") if r["status"] == 200]
|
|
bad = [r for f in files for r in rows(f, "lat") if r["status"] != 200]
|
|
by = collections.defaultdict(list)
|
|
for r in lat:
|
|
by[(r["arm"], r["bin"])].append(r)
|
|
arms = arms_order or sorted({a for a, _ in by})
|
|
table = []
|
|
for a in arms:
|
|
for b in BINS:
|
|
xs = by.get((a, b), [])
|
|
if not xs:
|
|
continue
|
|
e = [r["e2e_ms"] for r in xs]
|
|
s = [r["server_ms"] for r in xs if r.get("server_ms") is not None]
|
|
lo, hi = boot_ci(e)
|
|
table.append(dict(arm=a, bin=b, n=len(xs), rounds=len({r["round"] for r in xs}),
|
|
e2e_p50=pct(e, 50), e2e_p50_ci=[lo, hi], e2e_p90=pct(e, 90), e2e_p99=pct(e, 99), e2e_max=max(e),
|
|
srv_p50=pct(s, 50), srv_p90=pct(s, 90), srv_p99=pct(s, 99),
|
|
overhead_p50=pct([r["e2e_ms"] - r["server_ms"] for r in xs if r.get("server_ms") is not None], 50)))
|
|
return table, by, bad
|
|
|
|
|
|
def paired(by, a, ref, b, key="e2e_ms"):
|
|
ra = {(r["round"], r["id"]): r for r in by.get((a, b), [])}
|
|
rr = {(r["round"], r["id"]): r for r in by.get((ref, b), [])}
|
|
ks = sorted(set(ra) & set(rr))
|
|
if not ks:
|
|
return None
|
|
for k in ks: # the pairing is only valid if the two requests carried identical input
|
|
assert ra[k].get("trim_ms") == rr[k].get("trim_ms"), (a, ref, k)
|
|
d = [ra[k][key] - rr[k][key] for k in ks]
|
|
lo, hi = boot_ci(d)
|
|
return dict(arm=a, ref=ref, bin=b, n_pairs=len(ks), median_diff=float(np.median(d)), ci95=[lo, hi],
|
|
p50_arm=pct([ra[k][key] for k in ks], 50), p50_ref=pct([rr[k][key] for k in ks], 50))
|
|
|
|
|
|
def conc():
|
|
rs = [r for r in rows("conc.jsonl", "conc")]
|
|
bins = {(r["arm"], r["bin"]): r for r in rows("conc.jsonl", "conc-bin")}
|
|
by = collections.defaultdict(list)
|
|
for r in rs:
|
|
by[(r["arm"], r["bin"])].append(r)
|
|
out = []
|
|
for (a, b), xs in sorted(by.items()):
|
|
ok = [r for r in xs if r["status"] == 200]
|
|
e = [r["e2e_ms"] for r in ok]
|
|
wall = bins[(a, b)]["wall_s"]
|
|
audio = sum(r["dur"] - r.get("trim_ms", 0) / 1000 for r in ok)
|
|
out.append(dict(arm=a, bin=b, n=len(xs), errors=len(xs) - len(ok), e2e_p50=pct(e, 50), e2e_p90=pct(e, 90),
|
|
e2e_p99=pct(e, 99), srv_p50=pct([r["server_ms"] for r in ok if r.get("server_ms")], 50),
|
|
req_per_s=round(len(ok) / wall, 2), audio_x_realtime=round(audio / wall, 1)))
|
|
return out
|
|
|
|
|
|
def memory():
|
|
arms = {r["arm"]: r for r in rows("arms.jsonl")}
|
|
phases = rows("phases.jsonl")
|
|
# every PID of each container (NeMo forks workers; the CUDA context is in the first python process)
|
|
try:
|
|
allp = json.load(open(f"{RAW}/pids.json"))
|
|
except FileNotFoundError:
|
|
allp = {a: [r["pid"]] for a, r in arms.items()}
|
|
pid2arm = {p: a for a, ps in allp.items() for p in ps}
|
|
for a in arms:
|
|
arms[a]["pid"] = next((p for p in allp.get(a, []) if p in pid2arm), arms[a]["pid"])
|
|
series = collections.defaultdict(list) # pid -> [(t, MiB)]
|
|
import datetime
|
|
try:
|
|
for l in open(f"{RAW}/mem.csv"):
|
|
p = [x.strip() for x in l.split(",")]
|
|
if len(p) != 4 or not p[2].isdigit():
|
|
continue
|
|
pid = int(p[2])
|
|
if pid in pid2arm:
|
|
t = datetime.datetime.strptime(p[0], "%Y/%m/%d %H:%M:%S.%f").timestamp()
|
|
series[pid].append((t, int(p[3])))
|
|
except FileNotFoundError:
|
|
pass
|
|
out = []
|
|
for a, r in arms.items():
|
|
s = sorted(x for p in allp.get(a, [r["pid"]]) for x in series.get(p, []))
|
|
rest = [m for t, m in s if any(ph["arm"] == a and ph["phase"] == "rest-after-warmup" and ph["t0"] <= t <= ph["t1"] + 1 for ph in phases)]
|
|
lat_w = [ph for ph in phases if ph["arm"] == a and (ph["phase"].startswith("lat-") or ph["phase"] == "conc4")]
|
|
t_lat0 = min((ph["t0"] for ph in lat_w), default=None)
|
|
t_lat1 = max((ph["t1"] for ph in lat_w), default=None)
|
|
lat_peak = max((m for t, m in s if t_lat0 and t_lat0 <= t <= t_lat1 + 2), default=None)
|
|
out.append(dict(arm=a, pid=r["pid"], cold_s=r["cold_s"], rest_after_warmup_mib=max(rest) if rest else None,
|
|
peak_utterance_block_mib=lat_peak, peak_whole_life_mib=max((m for _, m in s), default=None),
|
|
last_mib=s[-1][1] if s else None, samples=len(s)))
|
|
return out
|
|
|
|
|
|
if __name__ == "__main__":
|
|
what = sys.argv[3] if len(sys.argv) > 3 else "all"
|
|
files = ["lat.jsonl"] + (["lat-live.jsonl"] if what in ("all", "live") else [])
|
|
t, by, bad = single_stream(files)
|
|
print("## single stream (ms)")
|
|
for r in t:
|
|
print(f"{r['arm']:10s} {r['bin']:7s} n={r['n']:3d} e2e p50 {fmt(r['e2e_p50'])} [{fmt(r['e2e_p50_ci'][0])},{fmt(r['e2e_p50_ci'][1])}] "
|
|
f"p90 {fmt(r['e2e_p90'])} p99 {fmt(r['e2e_p99'])} max {fmt(r['e2e_max'])} | server p50 {fmt(r['srv_p50'])} p90 {fmt(r['srv_p90'])} "
|
|
f"p99 {fmt(r['srv_p99'])} | http overhead p50 {fmt(r['overhead_p50'])}")
|
|
if bad:
|
|
print("NON-200:", collections.Counter((r["arm"], r["status"]) for r in bad))
|
|
print(f"\n## paired median difference vs {REF} (e2e ms; 95% bootstrap CI)")
|
|
pairs = []
|
|
for a in sorted({a for a, _ in by}):
|
|
if a == REF:
|
|
continue
|
|
for b in BINS:
|
|
p = paired(by, a, REF, b)
|
|
if p:
|
|
pairs.append(p)
|
|
print(f"{a:10s} {b:7s} pairs={p['n_pairs']:3d} diff {p['median_diff']:+8.1f} [{p['ci95'][0]:+8.1f},{p['ci95'][1]:+8.1f}]")
|
|
c = conc()
|
|
print("\n## concurrency 4 (ms)")
|
|
for r in c:
|
|
print(f"{r['arm']:10s} {r['bin']:7s} n={r['n']} err={r['errors']} e2e p50 {fmt(r['e2e_p50'])} p90 {fmt(r['e2e_p90'])} p99 {fmt(r['e2e_p99'])} "
|
|
f"| server p50 {fmt(r['srv_p50'])} | {r['req_per_s']} req/s, {r['audio_x_realtime']}x realtime")
|
|
first = rows("first.jsonl")
|
|
print("\n## first call per bin after cold start (ms) vs that arm's warm e2e p50")
|
|
warm = {(r["arm"], r["bin"]): r["e2e_p50"] for r in t}
|
|
for r in first:
|
|
print(f"{r['arm']:10s} {r['bin']:7s} first {r['e2e_ms']:8.1f} warm p50 {fmt(warm.get((r['arm'], r['bin']), float('nan')))}")
|
|
m = memory()
|
|
print("\n## GPU memory per process (MiB)")
|
|
for r in m:
|
|
print(json.dumps(r))
|
|
json.dump(dict(single=t, paired=pairs, conc=c, first=first, memory=m), open(f"{RAW}/../latency-summary.json", "w"), indent=1)
|