Files
esh-pfi-infrastructure/services/parakeet-ab-2026-09-30/code/acc_summary.py
T
vh a6c1d3c454 docs(parakeet): seat A/B vs parakeet-unified-en-0.6b - latency is the int8-on-CPU runtime; unified wins WER
A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against
nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image,
k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's
recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights).

- Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%).
- unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s
  (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54).
- unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other,
  -3.2 to -4.4 pp AMI (paired CIs exclude 0).
- Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after
  a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only).
- B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0.

Raw requests, hypotheses, manifests and the full harness under
services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart
from 240 light test requests.
2026-09-30 18:51:44 -07:00

91 lines
4.4 KiB
Python

"""Accuracy summary over out/raw/acc/<set>--<arm>.jsonl. Run with envs/score.
- WER per arm per set (score.py: whole-string Whisper normaliser, exact Levenshtein).
- Paired bootstrap deltas vs a reference arm (default ab-a1 = the seat's runtime + weights).
- Positive control (pc: 1.5 s of digital silence inside 40 utterances, reference unchanged): the arm
must register extra DELETIONS vs the same 40 utterances in ls-clean.
- Null control (null: ls-clean at -0.5 dB): the WER change vs ls-clean must sit inside the paired CI.
- Determinism: identical text across two instances / thread counts.
- Output features: share of outputs with sentence punctuation, with an upper-case letter, all-lower.
usage: acc_summary.py ACC_DIR DATA_DIR [REF_ARM]
"""
import glob
import json
import os
import re
import sys
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import score # noqa: E402
ACC, DATA = sys.argv[1], sys.argv[2]
REF = sys.argv[3] if len(sys.argv) > 3 else "ab-a1"
SETS = ["ls-clean", "ls-other", "ami", "pc", "null"]
def load(set_name):
ref = {json.loads(l)["id"]: json.loads(l) for l in open(f"{DATA}/{set_name}.jsonl")}
by = {}
for p in glob.glob(f"{ACC}/{set_name}--*.jsonl"):
arm = p.rsplit("--", 1)[1][:-6]
by[arm] = {json.loads(l)["id"]: json.loads(l) for l in open(p)}
return ref, by
out = dict(wer=[], delta=[], pc=[], null=[], same=[], features=[])
pu = {}
for s in SETS:
ref, by = load(s)
for arm, rows in sorted(by.items()):
pu[(s, arm)] = score.per_utt(ref, rows)
out["wer"].append(dict(set=s, arm=arm, **score.summary(pu[(s, arm)])))
if REF in by:
for arm in sorted(by):
if arm != REF:
out["delta"].append(dict(set=s, x=arm, y=REF, **score.boot(pu[(s, arm)], pu[(s, REF)])))
if "ab-b32" in by:
for arm in sorted(by):
if arm not in ("ab-b32", REF):
out["delta"].append(dict(set=s, x=arm, y="ab-b32", **score.boot(pu[(s, arm)], pu[(s, "ab-b32")])))
# positive control: deletions on the silenced copies vs the same utterances unsilenced
pc_ref = {json.loads(l)["id"]: json.loads(l) for l in open(f"{DATA}/pc.jsonl")}
for (s, arm), v in pu.items():
if s != "pc" or ("ls-clean", arm) not in pu:
continue
base = pu[("ls-clean", arm)]
ids = [k for k in pc_ref if v.get(k) and base.get(k)]
d_pc = sum(v[k][1] for k in ids); d_base = sum(base[k][1] for k in ids)
e_pc = sum(sum(v[k][:3]) for k in ids); e_base = sum(sum(base[k][:3]) for k in ids)
n = sum(base[k][3] for k in ids)
hit = sum(1 for k in ids if v[k][1] > base[k][1])
out["pc"].append(dict(arm=arm, utts=len(ids), deletions_clean=d_base, deletions_silenced=d_pc,
errors_clean=e_base, errors_silenced=e_pc, wer_clean=round(100 * e_base / n, 2),
wer_silenced=round(100 * e_pc / n, 2), utts_with_more_deletions=hit))
# null control
for (s, arm), v in pu.items():
if s == "null" and ("ls-clean", arm) in pu:
b = score.boot(v, pu[("ls-clean", arm)])
out["null"].append(dict(arm=arm, **b))
# determinism: identical text between pairs of arms on ls-clean
ref, by = load("ls-clean")
for a, b in (("ab-a1", "ab-a2"), ("ab-a1", "ab-at16"), ("ab-b32", "ab-b32t"), ("ab-a1", "A-live")):
if a in by and b in by:
ids = [k for k in ref if k in by[a] and k in by[b]]
out["same"].append(dict(a=a, b=b, utts=len(ids), identical_text=sum(by[a][k]["text"] == by[b][k]["text"] for k in ids)))
# features
for arm, rows in sorted(by.items()):
txt = [r["text"] or "" for r in rows.values() if r.get("status") == 200]
out["features"].append(dict(arm=arm, outputs=len(txt),
with_sentence_punct=round(100 * sum(bool(re.search(r"[.?!]", t)) for t in txt) / len(txt), 1),
with_comma=round(100 * sum("," in t for t in txt) / len(txt), 1),
with_upper=round(100 * sum(any(c.isupper() for c in t) for t in txt) / len(txt), 1),
all_lower=round(100 * sum(t == t.lower() and t.strip() != "" for t in txt) / len(txt), 1),
empty=sum(t.strip() == "" for t in txt),
non_ascii=sum(any(ord(c) > 127 for c in t) for t in txt)))
for k, v in out.items():
print(f"## {k}")
for r in v:
print(json.dumps(r))
json.dump(out, open(f"{ACC}/../../accuracy-summary.json", "w"), indent=1)