docs(parakeet): seat A/B vs parakeet-unified-en-0.6b - latency is the int8-on-CPU runtime; unified wins WER

A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against
nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image,
k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's
recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights).

- Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%).
- unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s
  (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54).
- unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other,
  -3.2 to -4.4 pp AMI (paired CIs exclude 0).
- Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after
  a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only).
- B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0.

Raw requests, hypotheses, manifests and the full harness under
services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart
from 240 light test requests.
This commit is contained in:
vh
2026-09-30 18:51:44 -07:00
parent 11174ffea1
commit a6c1d3c454
64 changed files with 23022 additions and 0 deletions
@@ -0,0 +1,115 @@
"""Score transcripts against ground truth (or any reference) and find dropped stretches.
Both sides are normalised word by word with Whisper's English normaliser
(transformers 4.53.3, no spelling map), so a hypothesis token keeps the time of
the word it came from. Alignment: difflib opcodes, then exact Levenshtein inside
each non-matching block.
A GAP is a maximal stretch between matching blocks of >= ISLAND tokens (shorter
matching islands inside a mismatch are treated as part of it, so one spurious
"the" cannot split a skipped paragraph in two). A gap is a
DROPOUT if it holds >= RUN reference words and the hypothesis emitted fewer
than half as many there (the hypothesis skipped speech), and an
INSERTION run if the mirror holds (the hypothesis added >= RUN words the
reference does not have).
Prints counts only.
"""
import difflib
import json
import sys
from transformers.models.whisper.english_normalizer import EnglishTextNormalizer
RUN = 10
ISLAND = 3
_norm = EnglishTextNormalizer({})
def tokens_from_text(text):
return [t for w in text.split() for t in _norm(w).split()]
def tokens_from_words(words):
"""[(token, start, end)] from word dicts, one entry per normalised token."""
out = []
for w in words:
for t in _norm(w["word"]).split():
out.append((t, float(w["start"]), float(w["end"])))
return out
def _edit_counts(a, b):
n, m = len(a), len(b)
d = [[0] * (m + 1) for _ in range(n + 1)]
for i in range(n + 1):
d[i][0] = i
for j in range(m + 1):
d[0][j] = j
for i in range(1, n + 1):
for j in range(1, m + 1):
d[i][j] = min(d[i - 1][j] + 1, d[i][j - 1] + 1, d[i - 1][j - 1] + (a[i - 1] != b[j - 1]))
s = i_ = de = 0
i, j = n, m
while i > 0 or j > 0:
if i > 0 and j > 0 and d[i][j] == d[i - 1][j - 1] + (a[i - 1] != b[j - 1]):
s += a[i - 1] != b[j - 1]
i, j = i - 1, j - 1
elif i > 0 and d[i][j] == d[i - 1][j] + 1:
de += 1
i -= 1
else:
i_ += 1
j -= 1
return s, i_, de
def score(ref, hyp, hyp_times=None):
"""ref, hyp: token lists. hyp_times: [(start, end)] per hyp token, optional."""
sm = difflib.SequenceMatcher(None, ref, hyp, autojunk=False)
ops = sm.get_opcodes()
S = I = D = 0
for tag, i1, i2, j1, j2 in ops:
if tag != "equal":
s, i_, de = _edit_counts(ref[i1:i2], hyp[j1:j2])
S, I, D = S + s, I + i_, D + de
# gaps between solid matching blocks
solid = [(i1, i2, j1, j2) for tag, i1, i2, j1, j2 in ops if tag == "equal" and i2 - i1 >= ISLAND]
bounds = [(0, 0, 0, 0)] + solid + [(len(ref), len(ref), len(hyp), len(hyp))]
drops, ins = [], []
for (_, a_i2, _, a_j2), (b_i1, _, b_j1, _) in zip(bounds, bounds[1:]):
gl, hl = b_i1 - a_i2, b_j1 - a_j2
t0 = t1 = None
if hyp_times:
t0 = hyp_times[a_j2 - 1][1] if a_j2 > 0 else 0.0
t1 = hyp_times[b_j1][0] if b_j1 < len(hyp) else hyp_times[-1][1]
gap = dict(ref_start=a_i2, ref_words=gl, hyp_start=a_j2, hyp_words=hl, t0=t0, t1=t1)
if gl >= RUN and hl < 0.5 * gl:
drops.append(gap)
elif hl >= RUN and gl < 0.5 * hl:
ins.append(gap)
n = len(ref)
return dict(ref_words=n, hyp_words=len(hyp), S=S, I=I, D=D,
wer=round((S + I + D) / n, 4) if n else None,
dropouts=len(drops), dropout_words=sum(g["ref_words"] for g in drops),
insertion_runs=len(ins), insertion_words=sum(g["hyp_words"] for g in ins),
drop_gaps=drops, ins_gaps=ins)
def load_hyp(path):
doc = json.load(open(path))
if "word_timestamps" in doc and doc["word_timestamps"]:
toks = tokens_from_words(doc["word_timestamps"])
return [t for t, _, _ in toks], [(s, e) for _, s, e in toks]
return tokens_from_text(doc.get("transcription") or doc.get("text", "")), None
if __name__ == "__main__":
ref_path, hyp_path = sys.argv[1], sys.argv[2]
ref = tokens_from_text(open(ref_path).read())
hyp, times = load_hyp(hyp_path)
r = score(ref, hyp, times)
print(json.dumps({k: v for k, v in r.items() if not k.endswith("_gaps")}))
for g in r["drop_gaps"]:
print(" DROP", g)
for g in r["ins_gaps"]:
print(" INS ", g)