Files
esh-pfi-infrastructure/services/parakeet-ab-2026-09-30/code/gtscore.py
T
vh a6c1d3c454 docs(parakeet): seat A/B vs parakeet-unified-en-0.6b - latency is the int8-on-CPU runtime; unified wins WER
A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against
nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image,
k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's
recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights).

- Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%).
- unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s
  (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54).
- unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other,
  -3.2 to -4.4 pp AMI (paired CIs exclude 0).
- Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after
  a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only).
- B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0.

Raw requests, hypotheses, manifests and the full harness under
services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart
from 240 light test requests.
2026-09-30 18:51:44 -07:00

116 lines
4.3 KiB
Python

"""Score transcripts against ground truth (or any reference) and find dropped stretches.
Both sides are normalised word by word with Whisper's English normaliser
(transformers 4.53.3, no spelling map), so a hypothesis token keeps the time of
the word it came from. Alignment: difflib opcodes, then exact Levenshtein inside
each non-matching block.
A GAP is a maximal stretch between matching blocks of >= ISLAND tokens (shorter
matching islands inside a mismatch are treated as part of it, so one spurious
"the" cannot split a skipped paragraph in two). A gap is a
DROPOUT if it holds >= RUN reference words and the hypothesis emitted fewer
than half as many there (the hypothesis skipped speech), and an
INSERTION run if the mirror holds (the hypothesis added >= RUN words the
reference does not have).
Prints counts only.
"""
import difflib
import json
import sys
from transformers.models.whisper.english_normalizer import EnglishTextNormalizer
RUN = 10
ISLAND = 3
_norm = EnglishTextNormalizer({})
def tokens_from_text(text):
return [t for w in text.split() for t in _norm(w).split()]
def tokens_from_words(words):
"""[(token, start, end)] from word dicts, one entry per normalised token."""
out = []
for w in words:
for t in _norm(w["word"]).split():
out.append((t, float(w["start"]), float(w["end"])))
return out
def _edit_counts(a, b):
n, m = len(a), len(b)
d = [[0] * (m + 1) for _ in range(n + 1)]
for i in range(n + 1):
d[i][0] = i
for j in range(m + 1):
d[0][j] = j
for i in range(1, n + 1):
for j in range(1, m + 1):
d[i][j] = min(d[i - 1][j] + 1, d[i][j - 1] + 1, d[i - 1][j - 1] + (a[i - 1] != b[j - 1]))
s = i_ = de = 0
i, j = n, m
while i > 0 or j > 0:
if i > 0 and j > 0 and d[i][j] == d[i - 1][j - 1] + (a[i - 1] != b[j - 1]):
s += a[i - 1] != b[j - 1]
i, j = i - 1, j - 1
elif i > 0 and d[i][j] == d[i - 1][j] + 1:
de += 1
i -= 1
else:
i_ += 1
j -= 1
return s, i_, de
def score(ref, hyp, hyp_times=None):
"""ref, hyp: token lists. hyp_times: [(start, end)] per hyp token, optional."""
sm = difflib.SequenceMatcher(None, ref, hyp, autojunk=False)
ops = sm.get_opcodes()
S = I = D = 0
for tag, i1, i2, j1, j2 in ops:
if tag != "equal":
s, i_, de = _edit_counts(ref[i1:i2], hyp[j1:j2])
S, I, D = S + s, I + i_, D + de
# gaps between solid matching blocks
solid = [(i1, i2, j1, j2) for tag, i1, i2, j1, j2 in ops if tag == "equal" and i2 - i1 >= ISLAND]
bounds = [(0, 0, 0, 0)] + solid + [(len(ref), len(ref), len(hyp), len(hyp))]
drops, ins = [], []
for (_, a_i2, _, a_j2), (b_i1, _, b_j1, _) in zip(bounds, bounds[1:]):
gl, hl = b_i1 - a_i2, b_j1 - a_j2
t0 = t1 = None
if hyp_times:
t0 = hyp_times[a_j2 - 1][1] if a_j2 > 0 else 0.0
t1 = hyp_times[b_j1][0] if b_j1 < len(hyp) else hyp_times[-1][1]
gap = dict(ref_start=a_i2, ref_words=gl, hyp_start=a_j2, hyp_words=hl, t0=t0, t1=t1)
if gl >= RUN and hl < 0.5 * gl:
drops.append(gap)
elif hl >= RUN and gl < 0.5 * hl:
ins.append(gap)
n = len(ref)
return dict(ref_words=n, hyp_words=len(hyp), S=S, I=I, D=D,
wer=round((S + I + D) / n, 4) if n else None,
dropouts=len(drops), dropout_words=sum(g["ref_words"] for g in drops),
insertion_runs=len(ins), insertion_words=sum(g["hyp_words"] for g in ins),
drop_gaps=drops, ins_gaps=ins)
def load_hyp(path):
doc = json.load(open(path))
if "word_timestamps" in doc and doc["word_timestamps"]:
toks = tokens_from_words(doc["word_timestamps"])
return [t for t, _, _ in toks], [(s, e) for _, s, e in toks]
return tokens_from_text(doc.get("transcription") or doc.get("text", "")), None
if __name__ == "__main__":
ref_path, hyp_path = sys.argv[1], sys.argv[2]
ref = tokens_from_text(open(ref_path).read())
hyp, times = load_hyp(hyp_path)
r = score(ref, hyp, times)
print(json.dumps({k: v for k, v in r.items() if not k.endswith("_gaps")}))
for g in r["drop_gaps"]:
print(" DROP", g)
for g in r["ins_gaps"]:
print(" INS ", g)