#!/usr/bin/env python3 """Latency block on fv-ml1 GPU 3. Host python3, stdlib only. 1. Cold: start each arm fresh and SEQUENTIALLY (so no two warm-ups compete for CPU), record container-start -> /healthz, then a first-call probe: one never-seen clip per bin, ascending length. 2. Warm single stream: R rounds; in each round the arms run in a seeded shuffled order and each arm sends all 80 clips once (seeded order, seeded 0-490 ms tail trim so every request is a new length). Interleaving spreads drift (thermal, a Scriberr job on GPU 3) across arms instead of onto one. 3. Concurrency 4: per arm, per bin, 40 requests from 4 workers. Writes out/raw/{arms,lat,conc,phases}.jsonl. Memory comes from the separate nvidia-smi sampler. usage: run_lat.py ROUNDS [ARM ...] """ import json import random import subprocess import sys import time AB = "/tank/spikes/parakeet-ab" import os RAW = os.environ.get("AB_RAW", f"{AB}/out/raw") M = f"{AB}/models" ARMS = { # name: (port, arm.sh args, description) "ab-a1": (18301, ["sherpa", "/tank/parakeet/models", "1"], "A' #1: seat image, v3 int8, 1 thread"), "ab-a2": (18302, ["sherpa", "/tank/parakeet/models", "1"], "A' #2: identical second instance (floor)"), "ab-a50": (18303, ["sherpa", "/tank/parakeet/models", "1", "int8.onnx", "50"], "A' +50 ms injected (positive control)"), "ab-apl": (18309, ["sherpa", "/tank/parakeet/models", "1", "int8.onnx", "0", "1"], "A' with the image's app untouched (null)"), "ab-at16": (18326, ["sherpa", "/tank/parakeet/models", "16"], "A' with NUM_THREADS=16"), "ab-c": (18304, ["sherpa", f"{M}/unified-int8", "1"], "C: unified-en int8 (k2-fsa), seat runtime"), "ab-d": (18305, ["sherpa", f"{M}/v2-int8", "1"], "D: v2 int8 (k2-fsa), seat runtime"), "ab-cf32": (18312, ["sherpa", f"{M}/unified-f32", "1", "onnx"], "C-fp32: unified-en fp32 ONNX (k2-fsa export recipe), seat runtime"), "ab-cf16": (18310, ["sherpa", f"{M}/unified-f16", "1", "fp16.onnx"], "C-fp16: unified-en fp16 ONNX (pre_encode kept fp32), seat runtime"), "ab-af32": (18311, ["sherpa", f"{M}/v3-f32", "1", "onnx"], "A-fp32: the seat's v3 as fp32 ONNX (k2-fsa export recipe), seat runtime"), "ab-b32": (18306, ["nemo", "fp32", "direct", "1"], "B-fp32: unified-en, NeMo 3.0.0 torch"), "ab-b16": (18307, ["nemo", "bf16", "direct", "1"], "B-bf16: unified-en, NeMo 3.0.0 torch, autocast bf16"), "ab-b32c": (18314, ["nemo", "fp32", "direct", "1", "1"], "B-fp32, .nemo restored on CPU then moved to GPU"), "ab-b16w": (18315, ["nemo", "bf16w", "direct", "1", "1"], "B-bf16w: encoder/decoder/joint weights in bf16, CPU load"), } BINS = ["b1_3", "b3_8", "b8_20", "b20_60"] def sh(cmd, **kw): return subprocess.run(cmd, capture_output=True, text=True, **kw) def log(fn, d): with open(f"{RAW}/{fn}", "a") as f: f.write(json.dumps(d) + "\n") def url(name): return f"http://127.0.0.1:{ARMS[name][0]}/v1/audio/transcriptions" def bench(*args): r = sh(["python3", f"{AB}/code/bench.py", *args]) if r.returncode: print("bench failed", args, r.stderr[-500:], flush=True) def main(): R = int(sys.argv[1]) arms = sys.argv[2:] or list(ARMS) sys.path.insert(0, f"{AB}/code") from bench import post, hostpath lat = {json.loads(l)["id"]: json.loads(l) for l in open(f"{AB}/data/lat.jsonl")} # 1. cold for name in arms: port, args, desc = ARMS[name] sh(["docker", "rm", "-f", name]) t = time.time() r = sh([f"{AB}/code/arm.sh", args[0], name, str(port), *args[1:]]) if r.returncode: print("arm failed", name, r.stderr, flush=True) continue _, _, t0, t1, cold = r.stdout.split() pid = sh(["docker", "top", name, "-eo", "pid"]).stdout.split()[-1] log("arms.jsonl", dict(arm=name, desc=desc, port=port, args=args, pid=int(pid), t_started=float(t0), t_ready=float(t1), cold_s=float(cold))) time.sleep(3) # at-rest memory window log("phases.jsonl", dict(arm=name, phase="rest-after-warmup", t0=time.time() - 3, t1=time.time())) for tag in BINS: # first call at each length: clip _10 of each bin (never sent before to this container) uid = f"{tag}_10" res = post(url(name), open(hostpath(lat[uid]["wav"]), "rb").read()) res.pop("text", None) log("first.jsonl", dict(arm=name, id=uid, bin=tag, dur=lat[uid]["dur"], t_wall=time.time(), **res)) print(f"cold {name} {cold}s", flush=True) # 2. warm single stream, interleaved t_ss = time.time() for k in range(R): order = list(arms) random.Random(f"round-{k}").shuffle(order) for name in order: t0 = time.time() bench("lat", url(name), name, f"{AB}/data/lat.jsonl", f"{RAW}/lat.jsonl", "--rounds", "1", "--round-offset", str(k), "--jitter-ms", "490") log("phases.jsonl", dict(arm=name, phase=f"lat-r{k}", t0=t0, t1=time.time())) print(f"round {k} done {time.time() - t_ss:.0f}s", flush=True) # 3. concurrency 4 for name in arms: t0 = time.time() bench("conc", url(name), name, f"{AB}/data/lat.jsonl", f"{RAW}/conc.jsonl", "--per-bin", "40", "--workers", "4", "--jitter-ms", "490") log("phases.jsonl", dict(arm=name, phase="conc4", t0=t0, t1=time.time())) print(f"conc {name} done", flush=True) print("LATENCY BLOCK DONE", flush=True) if __name__ == "__main__": main()