A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image, k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights). - Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%). - unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54). - unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other, -3.2 to -4.4 pp AMI (paired CIs exclude 0). - Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only). - B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0. Raw requests, hypotheses, manifests and the full harness under services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart from 240 light test requests.
116 lines
4.3 KiB
Python
116 lines
4.3 KiB
Python
"""Score transcripts against ground truth (or any reference) and find dropped stretches.
|
|
|
|
Both sides are normalised word by word with Whisper's English normaliser
|
|
(transformers 4.53.3, no spelling map), so a hypothesis token keeps the time of
|
|
the word it came from. Alignment: difflib opcodes, then exact Levenshtein inside
|
|
each non-matching block.
|
|
|
|
A GAP is a maximal stretch between matching blocks of >= ISLAND tokens (shorter
|
|
matching islands inside a mismatch are treated as part of it, so one spurious
|
|
"the" cannot split a skipped paragraph in two). A gap is a
|
|
DROPOUT if it holds >= RUN reference words and the hypothesis emitted fewer
|
|
than half as many there (the hypothesis skipped speech), and an
|
|
INSERTION run if the mirror holds (the hypothesis added >= RUN words the
|
|
reference does not have).
|
|
Prints counts only.
|
|
"""
|
|
import difflib
|
|
import json
|
|
import sys
|
|
|
|
from transformers.models.whisper.english_normalizer import EnglishTextNormalizer
|
|
|
|
RUN = 10
|
|
ISLAND = 3
|
|
_norm = EnglishTextNormalizer({})
|
|
|
|
|
|
def tokens_from_text(text):
|
|
return [t for w in text.split() for t in _norm(w).split()]
|
|
|
|
|
|
def tokens_from_words(words):
|
|
"""[(token, start, end)] from word dicts, one entry per normalised token."""
|
|
out = []
|
|
for w in words:
|
|
for t in _norm(w["word"]).split():
|
|
out.append((t, float(w["start"]), float(w["end"])))
|
|
return out
|
|
|
|
|
|
def _edit_counts(a, b):
|
|
n, m = len(a), len(b)
|
|
d = [[0] * (m + 1) for _ in range(n + 1)]
|
|
for i in range(n + 1):
|
|
d[i][0] = i
|
|
for j in range(m + 1):
|
|
d[0][j] = j
|
|
for i in range(1, n + 1):
|
|
for j in range(1, m + 1):
|
|
d[i][j] = min(d[i - 1][j] + 1, d[i][j - 1] + 1, d[i - 1][j - 1] + (a[i - 1] != b[j - 1]))
|
|
s = i_ = de = 0
|
|
i, j = n, m
|
|
while i > 0 or j > 0:
|
|
if i > 0 and j > 0 and d[i][j] == d[i - 1][j - 1] + (a[i - 1] != b[j - 1]):
|
|
s += a[i - 1] != b[j - 1]
|
|
i, j = i - 1, j - 1
|
|
elif i > 0 and d[i][j] == d[i - 1][j] + 1:
|
|
de += 1
|
|
i -= 1
|
|
else:
|
|
i_ += 1
|
|
j -= 1
|
|
return s, i_, de
|
|
|
|
|
|
def score(ref, hyp, hyp_times=None):
|
|
"""ref, hyp: token lists. hyp_times: [(start, end)] per hyp token, optional."""
|
|
sm = difflib.SequenceMatcher(None, ref, hyp, autojunk=False)
|
|
ops = sm.get_opcodes()
|
|
S = I = D = 0
|
|
for tag, i1, i2, j1, j2 in ops:
|
|
if tag != "equal":
|
|
s, i_, de = _edit_counts(ref[i1:i2], hyp[j1:j2])
|
|
S, I, D = S + s, I + i_, D + de
|
|
# gaps between solid matching blocks
|
|
solid = [(i1, i2, j1, j2) for tag, i1, i2, j1, j2 in ops if tag == "equal" and i2 - i1 >= ISLAND]
|
|
bounds = [(0, 0, 0, 0)] + solid + [(len(ref), len(ref), len(hyp), len(hyp))]
|
|
drops, ins = [], []
|
|
for (_, a_i2, _, a_j2), (b_i1, _, b_j1, _) in zip(bounds, bounds[1:]):
|
|
gl, hl = b_i1 - a_i2, b_j1 - a_j2
|
|
t0 = t1 = None
|
|
if hyp_times:
|
|
t0 = hyp_times[a_j2 - 1][1] if a_j2 > 0 else 0.0
|
|
t1 = hyp_times[b_j1][0] if b_j1 < len(hyp) else hyp_times[-1][1]
|
|
gap = dict(ref_start=a_i2, ref_words=gl, hyp_start=a_j2, hyp_words=hl, t0=t0, t1=t1)
|
|
if gl >= RUN and hl < 0.5 * gl:
|
|
drops.append(gap)
|
|
elif hl >= RUN and gl < 0.5 * hl:
|
|
ins.append(gap)
|
|
n = len(ref)
|
|
return dict(ref_words=n, hyp_words=len(hyp), S=S, I=I, D=D,
|
|
wer=round((S + I + D) / n, 4) if n else None,
|
|
dropouts=len(drops), dropout_words=sum(g["ref_words"] for g in drops),
|
|
insertion_runs=len(ins), insertion_words=sum(g["hyp_words"] for g in ins),
|
|
drop_gaps=drops, ins_gaps=ins)
|
|
|
|
|
|
def load_hyp(path):
|
|
doc = json.load(open(path))
|
|
if "word_timestamps" in doc and doc["word_timestamps"]:
|
|
toks = tokens_from_words(doc["word_timestamps"])
|
|
return [t for t, _, _ in toks], [(s, e) for _, s, e in toks]
|
|
return tokens_from_text(doc.get("transcription") or doc.get("text", "")), None
|
|
|
|
|
|
if __name__ == "__main__":
|
|
ref_path, hyp_path = sys.argv[1], sys.argv[2]
|
|
ref = tokens_from_text(open(ref_path).read())
|
|
hyp, times = load_hyp(hyp_path)
|
|
r = score(ref, hyp, times)
|
|
print(json.dumps({k: v for k, v in r.items() if not k.endswith("_gaps")}))
|
|
for g in r["drop_gaps"]:
|
|
print(" DROP", g)
|
|
for g in r["ins_gaps"]:
|
|
print(" INS ", g)
|