A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image, k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights). - Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%). - unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54). - unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other, -3.2 to -4.4 pp AMI (paired CIs exclude 0). - Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only). - B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0. Raw requests, hypotheses, manifests and the full harness under services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart from 240 light test requests.
91 lines
4.4 KiB
Python
91 lines
4.4 KiB
Python
"""Accuracy summary over out/raw/acc/<set>--<arm>.jsonl. Run with envs/score.
|
|
|
|
- WER per arm per set (score.py: whole-string Whisper normaliser, exact Levenshtein).
|
|
- Paired bootstrap deltas vs a reference arm (default ab-a1 = the seat's runtime + weights).
|
|
- Positive control (pc: 1.5 s of digital silence inside 40 utterances, reference unchanged): the arm
|
|
must register extra DELETIONS vs the same 40 utterances in ls-clean.
|
|
- Null control (null: ls-clean at -0.5 dB): the WER change vs ls-clean must sit inside the paired CI.
|
|
- Determinism: identical text across two instances / thread counts.
|
|
- Output features: share of outputs with sentence punctuation, with an upper-case letter, all-lower.
|
|
usage: acc_summary.py ACC_DIR DATA_DIR [REF_ARM]
|
|
"""
|
|
import glob
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
import score # noqa: E402
|
|
|
|
ACC, DATA = sys.argv[1], sys.argv[2]
|
|
REF = sys.argv[3] if len(sys.argv) > 3 else "ab-a1"
|
|
SETS = ["ls-clean", "ls-other", "ami", "pc", "null"]
|
|
|
|
|
|
def load(set_name):
|
|
ref = {json.loads(l)["id"]: json.loads(l) for l in open(f"{DATA}/{set_name}.jsonl")}
|
|
by = {}
|
|
for p in glob.glob(f"{ACC}/{set_name}--*.jsonl"):
|
|
arm = p.rsplit("--", 1)[1][:-6]
|
|
by[arm] = {json.loads(l)["id"]: json.loads(l) for l in open(p)}
|
|
return ref, by
|
|
|
|
|
|
out = dict(wer=[], delta=[], pc=[], null=[], same=[], features=[])
|
|
pu = {}
|
|
for s in SETS:
|
|
ref, by = load(s)
|
|
for arm, rows in sorted(by.items()):
|
|
pu[(s, arm)] = score.per_utt(ref, rows)
|
|
out["wer"].append(dict(set=s, arm=arm, **score.summary(pu[(s, arm)])))
|
|
if REF in by:
|
|
for arm in sorted(by):
|
|
if arm != REF:
|
|
out["delta"].append(dict(set=s, x=arm, y=REF, **score.boot(pu[(s, arm)], pu[(s, REF)])))
|
|
if "ab-b32" in by:
|
|
for arm in sorted(by):
|
|
if arm not in ("ab-b32", REF):
|
|
out["delta"].append(dict(set=s, x=arm, y="ab-b32", **score.boot(pu[(s, arm)], pu[(s, "ab-b32")])))
|
|
|
|
# positive control: deletions on the silenced copies vs the same utterances unsilenced
|
|
pc_ref = {json.loads(l)["id"]: json.loads(l) for l in open(f"{DATA}/pc.jsonl")}
|
|
for (s, arm), v in pu.items():
|
|
if s != "pc" or ("ls-clean", arm) not in pu:
|
|
continue
|
|
base = pu[("ls-clean", arm)]
|
|
ids = [k for k in pc_ref if v.get(k) and base.get(k)]
|
|
d_pc = sum(v[k][1] for k in ids); d_base = sum(base[k][1] for k in ids)
|
|
e_pc = sum(sum(v[k][:3]) for k in ids); e_base = sum(sum(base[k][:3]) for k in ids)
|
|
n = sum(base[k][3] for k in ids)
|
|
hit = sum(1 for k in ids if v[k][1] > base[k][1])
|
|
out["pc"].append(dict(arm=arm, utts=len(ids), deletions_clean=d_base, deletions_silenced=d_pc,
|
|
errors_clean=e_base, errors_silenced=e_pc, wer_clean=round(100 * e_base / n, 2),
|
|
wer_silenced=round(100 * e_pc / n, 2), utts_with_more_deletions=hit))
|
|
# null control
|
|
for (s, arm), v in pu.items():
|
|
if s == "null" and ("ls-clean", arm) in pu:
|
|
b = score.boot(v, pu[("ls-clean", arm)])
|
|
out["null"].append(dict(arm=arm, **b))
|
|
# determinism: identical text between pairs of arms on ls-clean
|
|
ref, by = load("ls-clean")
|
|
for a, b in (("ab-a1", "ab-a2"), ("ab-a1", "ab-at16"), ("ab-b32", "ab-b32t"), ("ab-a1", "A-live")):
|
|
if a in by and b in by:
|
|
ids = [k for k in ref if k in by[a] and k in by[b]]
|
|
out["same"].append(dict(a=a, b=b, utts=len(ids), identical_text=sum(by[a][k]["text"] == by[b][k]["text"] for k in ids)))
|
|
# features
|
|
for arm, rows in sorted(by.items()):
|
|
txt = [r["text"] or "" for r in rows.values() if r.get("status") == 200]
|
|
out["features"].append(dict(arm=arm, outputs=len(txt),
|
|
with_sentence_punct=round(100 * sum(bool(re.search(r"[.?!]", t)) for t in txt) / len(txt), 1),
|
|
with_comma=round(100 * sum("," in t for t in txt) / len(txt), 1),
|
|
with_upper=round(100 * sum(any(c.isupper() for c in t) for t in txt) / len(txt), 1),
|
|
all_lower=round(100 * sum(t == t.lower() and t.strip() != "" for t in txt) / len(txt), 1),
|
|
empty=sum(t.strip() == "" for t in txt),
|
|
non_ascii=sum(any(ord(c) > 127 for c in t) for t in txt)))
|
|
for k, v in out.items():
|
|
print(f"## {k}")
|
|
for r in v:
|
|
print(json.dumps(r))
|
|
json.dump(out, open(f"{ACC}/../../accuracy-summary.json", "w"), indent=1)
|