"""Latency + memory analysis. numpy; run with envs/score. Single stream: per arm x bin, p50/p90/p99 of server-side (x-ab-decode-ms) and end-to-end (client wall, new connection per request, fv-ml1 loopback). Deltas vs a reference arm are PAIRED: in round k every arm got byte-identical input for a clip (the tail trim is seeded by round and clip, never by arm), so the statistic is the median of per-request differences, with a 95 % bootstrap CI over requests. Floor = ab-a2 vs ab-a1 (identical second instance). Positive control = ab-a50 (+50 ms injected) vs ab-a1. Null = ab-apl (the image's own app, no timing header) vs ab-a1 (e2e only). usage: analyze_lat.py RAW_DIR [REF_ARM] """ import collections import json import sys import numpy as np RAW = sys.argv[1] REF = sys.argv[2] if len(sys.argv) > 2 else "ab-a1" BINS = ["b1_3", "b3_8", "b8_20", "b20_60"] rng = np.random.default_rng(7) def rows(fn, mode=None): out = [] try: for l in open(f"{RAW}/{fn}"): r = json.loads(l) if mode is None or r.get("mode") == mode: out.append(r) except FileNotFoundError: pass return out def pct(x, q): return float(np.percentile(np.asarray(x, float), q)) if len(x) else float("nan") def boot_ci(x, stat=np.median, n=2000): x = np.asarray(x, float) if len(x) < 3: return (float("nan"), float("nan")) b = [stat(x[rng.integers(0, len(x), len(x))]) for _ in range(n)] return (float(np.percentile(b, 2.5)), float(np.percentile(b, 97.5))) def fmt(v): return "nan" if v != v else f"{v:.1f}" def single_stream(files, arms_order=None): lat = [r for f in files for r in rows(f, "lat") if r["status"] == 200] bad = [r for f in files for r in rows(f, "lat") if r["status"] != 200] by = collections.defaultdict(list) for r in lat: by[(r["arm"], r["bin"])].append(r) arms = arms_order or sorted({a for a, _ in by}) table = [] for a in arms: for b in BINS: xs = by.get((a, b), []) if not xs: continue e = [r["e2e_ms"] for r in xs] s = [r["server_ms"] for r in xs if r.get("server_ms") is not None] lo, hi = boot_ci(e) table.append(dict(arm=a, bin=b, n=len(xs), rounds=len({r["round"] for r in xs}), e2e_p50=pct(e, 50), e2e_p50_ci=[lo, hi], e2e_p90=pct(e, 90), e2e_p99=pct(e, 99), e2e_max=max(e), srv_p50=pct(s, 50), srv_p90=pct(s, 90), srv_p99=pct(s, 99), overhead_p50=pct([r["e2e_ms"] - r["server_ms"] for r in xs if r.get("server_ms") is not None], 50))) return table, by, bad def paired(by, a, ref, b, key="e2e_ms"): ra = {(r["round"], r["id"]): r for r in by.get((a, b), [])} rr = {(r["round"], r["id"]): r for r in by.get((ref, b), [])} ks = sorted(set(ra) & set(rr)) if not ks: return None for k in ks: # the pairing is only valid if the two requests carried identical input assert ra[k].get("trim_ms") == rr[k].get("trim_ms"), (a, ref, k) d = [ra[k][key] - rr[k][key] for k in ks] lo, hi = boot_ci(d) return dict(arm=a, ref=ref, bin=b, n_pairs=len(ks), median_diff=float(np.median(d)), ci95=[lo, hi], p50_arm=pct([ra[k][key] for k in ks], 50), p50_ref=pct([rr[k][key] for k in ks], 50)) def conc(): rs = [r for r in rows("conc.jsonl", "conc")] bins = {(r["arm"], r["bin"]): r for r in rows("conc.jsonl", "conc-bin")} by = collections.defaultdict(list) for r in rs: by[(r["arm"], r["bin"])].append(r) out = [] for (a, b), xs in sorted(by.items()): ok = [r for r in xs if r["status"] == 200] e = [r["e2e_ms"] for r in ok] wall = bins[(a, b)]["wall_s"] audio = sum(r["dur"] - r.get("trim_ms", 0) / 1000 for r in ok) out.append(dict(arm=a, bin=b, n=len(xs), errors=len(xs) - len(ok), e2e_p50=pct(e, 50), e2e_p90=pct(e, 90), e2e_p99=pct(e, 99), srv_p50=pct([r["server_ms"] for r in ok if r.get("server_ms")], 50), req_per_s=round(len(ok) / wall, 2), audio_x_realtime=round(audio / wall, 1))) return out def memory(): arms = {r["arm"]: r for r in rows("arms.jsonl")} phases = rows("phases.jsonl") # every PID of each container (NeMo forks workers; the CUDA context is in the first python process) try: allp = json.load(open(f"{RAW}/pids.json")) except FileNotFoundError: allp = {a: [r["pid"]] for a, r in arms.items()} pid2arm = {p: a for a, ps in allp.items() for p in ps} for a in arms: arms[a]["pid"] = next((p for p in allp.get(a, []) if p in pid2arm), arms[a]["pid"]) series = collections.defaultdict(list) # pid -> [(t, MiB)] import datetime try: for l in open(f"{RAW}/mem.csv"): p = [x.strip() for x in l.split(",")] if len(p) != 4 or not p[2].isdigit(): continue pid = int(p[2]) if pid in pid2arm: t = datetime.datetime.strptime(p[0], "%Y/%m/%d %H:%M:%S.%f").timestamp() series[pid].append((t, int(p[3]))) except FileNotFoundError: pass out = [] for a, r in arms.items(): s = sorted(x for p in allp.get(a, [r["pid"]]) for x in series.get(p, [])) rest = [m for t, m in s if any(ph["arm"] == a and ph["phase"] == "rest-after-warmup" and ph["t0"] <= t <= ph["t1"] + 1 for ph in phases)] lat_w = [ph for ph in phases if ph["arm"] == a and (ph["phase"].startswith("lat-") or ph["phase"] == "conc4")] t_lat0 = min((ph["t0"] for ph in lat_w), default=None) t_lat1 = max((ph["t1"] for ph in lat_w), default=None) lat_peak = max((m for t, m in s if t_lat0 and t_lat0 <= t <= t_lat1 + 2), default=None) out.append(dict(arm=a, pid=r["pid"], cold_s=r["cold_s"], rest_after_warmup_mib=max(rest) if rest else None, peak_utterance_block_mib=lat_peak, peak_whole_life_mib=max((m for _, m in s), default=None), last_mib=s[-1][1] if s else None, samples=len(s))) return out if __name__ == "__main__": what = sys.argv[3] if len(sys.argv) > 3 else "all" files = ["lat.jsonl"] + (["lat-live.jsonl"] if what in ("all", "live") else []) t, by, bad = single_stream(files) print("## single stream (ms)") for r in t: print(f"{r['arm']:10s} {r['bin']:7s} n={r['n']:3d} e2e p50 {fmt(r['e2e_p50'])} [{fmt(r['e2e_p50_ci'][0])},{fmt(r['e2e_p50_ci'][1])}] " f"p90 {fmt(r['e2e_p90'])} p99 {fmt(r['e2e_p99'])} max {fmt(r['e2e_max'])} | server p50 {fmt(r['srv_p50'])} p90 {fmt(r['srv_p90'])} " f"p99 {fmt(r['srv_p99'])} | http overhead p50 {fmt(r['overhead_p50'])}") if bad: print("NON-200:", collections.Counter((r["arm"], r["status"]) for r in bad)) print(f"\n## paired median difference vs {REF} (e2e ms; 95% bootstrap CI)") pairs = [] for a in sorted({a for a, _ in by}): if a == REF: continue for b in BINS: p = paired(by, a, REF, b) if p: pairs.append(p) print(f"{a:10s} {b:7s} pairs={p['n_pairs']:3d} diff {p['median_diff']:+8.1f} [{p['ci95'][0]:+8.1f},{p['ci95'][1]:+8.1f}]") c = conc() print("\n## concurrency 4 (ms)") for r in c: print(f"{r['arm']:10s} {r['bin']:7s} n={r['n']} err={r['errors']} e2e p50 {fmt(r['e2e_p50'])} p90 {fmt(r['e2e_p90'])} p99 {fmt(r['e2e_p99'])} " f"| server p50 {fmt(r['srv_p50'])} | {r['req_per_s']} req/s, {r['audio_x_realtime']}x realtime") first = rows("first.jsonl") print("\n## first call per bin after cold start (ms) vs that arm's warm e2e p50") warm = {(r["arm"], r["bin"]): r["e2e_p50"] for r in t} for r in first: print(f"{r['arm']:10s} {r['bin']:7s} first {r['e2e_ms']:8.1f} warm p50 {fmt(warm.get((r['arm'], r['bin']), float('nan')))}") m = memory() print("\n## GPU memory per process (MiB)") for r in m: print(json.dumps(r)) json.dump(dict(single=t, paired=pairs, conc=c, first=first, memory=m), open(f"{RAW}/../latency-summary.json", "w"), indent=1)