docs(parakeet): seat A/B vs parakeet-unified-en-0.6b - latency is the int8-on-CPU runtime; unified wins WER
A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image, k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights). - Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%). - unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54). - unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other, -3.2 to -4.4 pp AMI (paired CIs exclude 0). - Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only). - B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0. Raw requests, hypotheses, manifests and the full harness under services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart from 240 light test requests.
This commit is contained in:
@@ -0,0 +1,115 @@
|
||||
"""Score transcripts against ground truth (or any reference) and find dropped stretches.
|
||||
|
||||
Both sides are normalised word by word with Whisper's English normaliser
|
||||
(transformers 4.53.3, no spelling map), so a hypothesis token keeps the time of
|
||||
the word it came from. Alignment: difflib opcodes, then exact Levenshtein inside
|
||||
each non-matching block.
|
||||
|
||||
A GAP is a maximal stretch between matching blocks of >= ISLAND tokens (shorter
|
||||
matching islands inside a mismatch are treated as part of it, so one spurious
|
||||
"the" cannot split a skipped paragraph in two). A gap is a
|
||||
DROPOUT if it holds >= RUN reference words and the hypothesis emitted fewer
|
||||
than half as many there (the hypothesis skipped speech), and an
|
||||
INSERTION run if the mirror holds (the hypothesis added >= RUN words the
|
||||
reference does not have).
|
||||
Prints counts only.
|
||||
"""
|
||||
import difflib
|
||||
import json
|
||||
import sys
|
||||
|
||||
from transformers.models.whisper.english_normalizer import EnglishTextNormalizer
|
||||
|
||||
RUN = 10
|
||||
ISLAND = 3
|
||||
_norm = EnglishTextNormalizer({})
|
||||
|
||||
|
||||
def tokens_from_text(text):
|
||||
return [t for w in text.split() for t in _norm(w).split()]
|
||||
|
||||
|
||||
def tokens_from_words(words):
|
||||
"""[(token, start, end)] from word dicts, one entry per normalised token."""
|
||||
out = []
|
||||
for w in words:
|
||||
for t in _norm(w["word"]).split():
|
||||
out.append((t, float(w["start"]), float(w["end"])))
|
||||
return out
|
||||
|
||||
|
||||
def _edit_counts(a, b):
|
||||
n, m = len(a), len(b)
|
||||
d = [[0] * (m + 1) for _ in range(n + 1)]
|
||||
for i in range(n + 1):
|
||||
d[i][0] = i
|
||||
for j in range(m + 1):
|
||||
d[0][j] = j
|
||||
for i in range(1, n + 1):
|
||||
for j in range(1, m + 1):
|
||||
d[i][j] = min(d[i - 1][j] + 1, d[i][j - 1] + 1, d[i - 1][j - 1] + (a[i - 1] != b[j - 1]))
|
||||
s = i_ = de = 0
|
||||
i, j = n, m
|
||||
while i > 0 or j > 0:
|
||||
if i > 0 and j > 0 and d[i][j] == d[i - 1][j - 1] + (a[i - 1] != b[j - 1]):
|
||||
s += a[i - 1] != b[j - 1]
|
||||
i, j = i - 1, j - 1
|
||||
elif i > 0 and d[i][j] == d[i - 1][j] + 1:
|
||||
de += 1
|
||||
i -= 1
|
||||
else:
|
||||
i_ += 1
|
||||
j -= 1
|
||||
return s, i_, de
|
||||
|
||||
|
||||
def score(ref, hyp, hyp_times=None):
|
||||
"""ref, hyp: token lists. hyp_times: [(start, end)] per hyp token, optional."""
|
||||
sm = difflib.SequenceMatcher(None, ref, hyp, autojunk=False)
|
||||
ops = sm.get_opcodes()
|
||||
S = I = D = 0
|
||||
for tag, i1, i2, j1, j2 in ops:
|
||||
if tag != "equal":
|
||||
s, i_, de = _edit_counts(ref[i1:i2], hyp[j1:j2])
|
||||
S, I, D = S + s, I + i_, D + de
|
||||
# gaps between solid matching blocks
|
||||
solid = [(i1, i2, j1, j2) for tag, i1, i2, j1, j2 in ops if tag == "equal" and i2 - i1 >= ISLAND]
|
||||
bounds = [(0, 0, 0, 0)] + solid + [(len(ref), len(ref), len(hyp), len(hyp))]
|
||||
drops, ins = [], []
|
||||
for (_, a_i2, _, a_j2), (b_i1, _, b_j1, _) in zip(bounds, bounds[1:]):
|
||||
gl, hl = b_i1 - a_i2, b_j1 - a_j2
|
||||
t0 = t1 = None
|
||||
if hyp_times:
|
||||
t0 = hyp_times[a_j2 - 1][1] if a_j2 > 0 else 0.0
|
||||
t1 = hyp_times[b_j1][0] if b_j1 < len(hyp) else hyp_times[-1][1]
|
||||
gap = dict(ref_start=a_i2, ref_words=gl, hyp_start=a_j2, hyp_words=hl, t0=t0, t1=t1)
|
||||
if gl >= RUN and hl < 0.5 * gl:
|
||||
drops.append(gap)
|
||||
elif hl >= RUN and gl < 0.5 * hl:
|
||||
ins.append(gap)
|
||||
n = len(ref)
|
||||
return dict(ref_words=n, hyp_words=len(hyp), S=S, I=I, D=D,
|
||||
wer=round((S + I + D) / n, 4) if n else None,
|
||||
dropouts=len(drops), dropout_words=sum(g["ref_words"] for g in drops),
|
||||
insertion_runs=len(ins), insertion_words=sum(g["hyp_words"] for g in ins),
|
||||
drop_gaps=drops, ins_gaps=ins)
|
||||
|
||||
|
||||
def load_hyp(path):
|
||||
doc = json.load(open(path))
|
||||
if "word_timestamps" in doc and doc["word_timestamps"]:
|
||||
toks = tokens_from_words(doc["word_timestamps"])
|
||||
return [t for t, _, _ in toks], [(s, e) for _, s, e in toks]
|
||||
return tokens_from_text(doc.get("transcription") or doc.get("text", "")), None
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
ref_path, hyp_path = sys.argv[1], sys.argv[2]
|
||||
ref = tokens_from_text(open(ref_path).read())
|
||||
hyp, times = load_hyp(hyp_path)
|
||||
r = score(ref, hyp, times)
|
||||
print(json.dumps({k: v for k, v in r.items() if not k.endswith("_gaps")}))
|
||||
for g in r["drop_gaps"]:
|
||||
print(" DROP", g)
|
||||
for g in r["ins_gaps"]:
|
||||
print(" INS ", g)
|
||||
Reference in New Issue
Block a user