Files
esh-pfi-infrastructure/services/parakeet-ab-2026-09-30/code/score.py
T
vh a6c1d3c454 docs(parakeet): seat A/B vs parakeet-unified-en-0.6b - latency is the int8-on-CPU runtime; unified wins WER
A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against
nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image,
k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's
recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights).

- Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%).
- unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s
  (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54).
- unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other,
  -3.2 to -4.4 pp AMI (paired CIs exclude 0).
- Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after
  a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only).
- B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0.

Raw requests, hypotheses, manifests and the full harness under
services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart
from 240 light test requests.
2026-09-30 18:51:44 -07:00

134 lines
5.3 KiB
Python

"""WER for the utterance sets, with paired-bootstrap deltas between arms.
Normaliser: Whisper's EnglishTextNormalizer, transformers 4.53.3 (the investigation's version), no
spelling map, applied to the WHOLE utterance string on both sides (the standard Open-ASR-leaderboard
use; the investigation applied it word by word only because it had to keep per-word timestamps on
long-form, which is scored with its own gtscore.py, unchanged). Edit counts: exact Levenshtein with a
backtrace for S/D/I. Corpus WER = sum(S+D+I) / sum(ref words).
usage:
score.py wer MANIFEST RESULTS.jsonl [ARM ...] -> per-arm WER table (json lines)
score.py delta MANIFEST RESULTS.jsonl ARM_X ARM_Y -> paired bootstrap of WER_X - WER_Y
score.py selftest -> positive/null checks of the scorer itself
"""
import json
import random
import sys
from transformers.models.whisper.english_normalizer import EnglishTextNormalizer
_norm = EnglishTextNormalizer({})
def toks(text):
return _norm(text or "").split()
def edits(ref, hyp):
n, m = len(ref), len(hyp)
d = [[0] * (m + 1) for _ in range(n + 1)]
for i in range(n + 1):
d[i][0] = i
for j in range(m + 1):
d[0][j] = j
for i in range(1, n + 1):
ri = ref[i - 1]
row, prev = d[i], d[i - 1]
for j in range(1, m + 1):
row[j] = min(prev[j] + 1, row[j - 1] + 1, prev[j - 1] + (ri != hyp[j - 1]))
s = ins = de = 0
i, j = n, m
while i > 0 or j > 0:
if i > 0 and j > 0 and d[i][j] == d[i - 1][j - 1] + (ref[i - 1] != hyp[j - 1]):
s += ref[i - 1] != hyp[j - 1]
i, j = i - 1, j - 1
elif i > 0 and d[i][j] == d[i - 1][j] + 1:
de += 1
i -= 1
else:
ins += 1
j -= 1
return s, de, ins
def load(manifest, results, arms=None):
ref = {}
for l in open(manifest):
r = json.loads(l)
ref[r["id"]] = r
by = {}
for l in open(results):
r = json.loads(l)
if r.get("mode") != "acc" or (arms and r["arm"] not in arms):
continue
by.setdefault(r["arm"], {})[r["id"]] = r # last write wins (a rerun replaces)
return ref, by
def per_utt(ref, rows):
out = {}
for uid, rr in ref.items():
h = rows.get(uid)
R = toks(rr["ref"])
H = toks(h["text"]) if h and h.get("status") == 200 else None
if H is None:
out[uid] = None
continue
s, de, ins = edits(R, H)
out[uid] = (s, de, ins, len(R), len(H) == 0)
return out
def summary(pu):
ok = [v for v in pu.values() if v is not None]
S = sum(v[0] for v in ok); D = sum(v[1] for v in ok); I = sum(v[2] for v in ok); N = sum(v[3] for v in ok)
return dict(utts=len(ok), missing=sum(v is None for v in pu.values()), ref_words=N, S=S, D=D, I=I,
wer=round(100 * (S + D + I) / N, 3), empty_outputs=sum(v[4] for v in ok))
def boot(pu_x, pu_y, n=4000, seed=7):
ids = [k for k in pu_x if pu_x[k] is not None and pu_y.get(k) is not None]
ex = [sum(pu_x[k][:3]) for k in ids]; ey = [sum(pu_y[k][:3]) for k in ids]; nw = [pu_x[k][3] for k in ids]
point = 100 * (sum(ex) - sum(ey)) / sum(nw)
rng = random.Random(seed)
ds = []
for _ in range(n):
idx = [rng.randrange(len(ids)) for _ in ids]
N = sum(nw[i] for i in idx)
ds.append(100 * (sum(ex[i] for i in idx) - sum(ey[i] for i in idx)) / N)
ds.sort()
same = sum(1 for k in ids if pu_x[k][:3] == pu_y[k][:3])
return dict(n_utts=len(ids), delta_pp=round(point, 3), ci95=[round(ds[int(.025 * n)], 3), round(ds[int(.975 * n)], 3)],
utts_with_identical_edit_counts=same)
def selftest():
# scorer positive controls: one deleted word, one wrong word, one inserted word must each register exactly once
ref = "the quick brown fox jumps over the lazy dog"
cases = {"identical": (ref, (0, 0, 0)), "one deletion": ("the quick fox jumps over the lazy dog", (0, 1, 0)),
"one substitution": ("the quick brown box jumps over the lazy dog", (1, 0, 0)),
"one insertion": ("the quick brown fox jumps right over the lazy dog", (0, 0, 1)),
"casing+punct only (null)": ("The quick, brown fox jumps over the lazy dog.", (0, 0, 0)),
"filler only (null)": ("the quick brown fox uh jumps over the lazy dog", (0, 0, 0))}
ok = True
for name, (hyp, want) in cases.items():
got = edits(toks(ref), toks(hyp))
ok &= got == want
print(f"{name:26s} want S/D/I {want} got {got} {'OK' if got == want else 'FAIL'}")
print("SELFTEST", "PASS" if ok else "FAIL")
if __name__ == "__main__":
cmd = sys.argv[1]
if cmd == "selftest":
selftest()
elif cmd == "wer":
ref, by = load(sys.argv[2], sys.argv[3], sys.argv[4:] or None)
for arm in sorted(by):
print(json.dumps(dict(arm=arm, set=sys.argv[2].rsplit("/", 1)[-1].replace(".jsonl", ""), **summary(per_utt(ref, by[arm])))))
elif cmd == "delta":
ref, by = load(sys.argv[2], sys.argv[3], sys.argv[4:6])
x, y = sys.argv[4], sys.argv[5]
print(json.dumps(dict(set=sys.argv[2].rsplit("/", 1)[-1].replace(".jsonl", ""), x=x, y=y,
**boot(per_utt(ref, by[x]), per_utt(ref, by[y])))))