docs(parakeet): seat A/B vs parakeet-unified-en-0.6b - latency is the int8-on-CPU runtime; unified wins WER
A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image, k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights). - Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%). - unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54). - unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other, -3.2 to -4.4 pp AMI (paired CIs exclude 0). - Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only). - B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0. Raw requests, hypotheses, manifests and the full harness under services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart from 240 light test requests.
This commit is contained in:
@@ -0,0 +1,149 @@
|
||||
"""Build the A/B test sets from the pinned parquets. Seeded; writes 16 kHz mono PCM16 WAVs + manifests.
|
||||
|
||||
ls-clean, ls-other : 400 utterances each, uniform random sample (seed 20260930) of LibriSpeech test
|
||||
ami : 400 AMI IHM test utterances >= 1.0 s, uniform random sample (same seed)
|
||||
lat : latency clips, 20 per bin, from ALL of test-clean:
|
||||
1-3 s, 3-8 s, 8-20 s: single utterances at evenly spaced duration quantiles of the bin
|
||||
20-60 s: consecutive utterances of one chapter joined with 0.25 s silence, targets 20..58 s
|
||||
pc : positive control, 40 ls-clean sample utterances >= 6 s with 1.5 s of digital silence
|
||||
placed at 40 % of the utterance (the reference still holds the words)
|
||||
null : the 400 ls-clean utterances at -0.5 dB gain (a should-not-matter perturbation)
|
||||
Manifests: data/<set>.jsonl rows {id, wav, dur, ref, ...}. Prints counts only.
|
||||
"""
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import random
|
||||
|
||||
import numpy as np
|
||||
import pyarrow.parquet as pq
|
||||
import soundfile as sf
|
||||
|
||||
DL = "/ab/dl"
|
||||
OUT = "/ab/data"
|
||||
SEED = 20260930
|
||||
SR = 16000
|
||||
|
||||
|
||||
def audio_of(cell):
|
||||
a, sr = sf.read(io.BytesIO(cell["bytes"]), dtype="float32")
|
||||
if a.ndim > 1:
|
||||
a = a.mean(axis=1)
|
||||
assert sr == SR, sr
|
||||
return a.astype(np.float32)
|
||||
|
||||
|
||||
def write(set_name, uid, a):
|
||||
d = f"{OUT}/{set_name}"
|
||||
os.makedirs(d, exist_ok=True)
|
||||
p = f"{d}/{uid}.wav"
|
||||
sf.write(p, np.clip(a, -1, 1), SR, subtype="PCM_16")
|
||||
return p
|
||||
|
||||
|
||||
def dump(set_name, rows):
|
||||
with open(f"{OUT}/{set_name}.jsonl", "w") as f:
|
||||
for r in rows:
|
||||
f.write(json.dumps(r) + "\n")
|
||||
print(set_name, len(rows), "utts", round(sum(r["dur"] for r in rows) / 60, 1), "min")
|
||||
|
||||
|
||||
def libri(split):
|
||||
t = pq.read_table(f"{DL}/openslr__librispeech_asr__all__test.{split}__0000.parquet").to_pylist()
|
||||
return t
|
||||
|
||||
|
||||
clean = libri("clean")
|
||||
other = libri("other")
|
||||
print("test-clean", len(clean), "test-other", len(other))
|
||||
|
||||
samples = {}
|
||||
for name, tab in (("ls-clean", clean), ("ls-other", other)):
|
||||
rng = random.Random(SEED)
|
||||
pick = rng.sample(range(len(tab)), 400)
|
||||
rows = []
|
||||
for i in pick:
|
||||
r = tab[i]
|
||||
a = audio_of(r["audio"])
|
||||
rows.append(dict(id=r["id"], wav=write(name, r["id"], a), dur=round(len(a) / SR, 3), ref=r["text"],
|
||||
speaker=r["speaker_id"], chapter=r["chapter_id"]))
|
||||
samples.setdefault(name, []).append((r, a))
|
||||
dump(name, rows)
|
||||
|
||||
# AMI IHM test
|
||||
ami = []
|
||||
for k in range(4):
|
||||
for r in pq.read_table(f"{DL}/edinburghcstr__ami__ihm__test-0000{k}-of-00004.parquet").to_pylist():
|
||||
dur = float(r["end_time"]) - float(r["begin_time"])
|
||||
if dur >= 1.0 and r["text"].strip():
|
||||
ami.append(r)
|
||||
rng = random.Random(SEED)
|
||||
rows = []
|
||||
for r in rng.sample(ami, 400):
|
||||
a = audio_of(r["audio"])
|
||||
rows.append(dict(id=r["audio_id"], wav=write("ami", r["audio_id"], a), dur=round(len(a) / SR, 3), ref=r["text"],
|
||||
meeting=r["meeting_id"], speaker=r["speaker_id"]))
|
||||
print("ami eligible (>=1 s)", len(ami), "meetings in sample", len({r["meeting"] for r in rows}))
|
||||
dump("ami", rows)
|
||||
|
||||
# latency clips
|
||||
durs = [(len(audio_of(r["audio"])) / SR, i) for i, r in enumerate(clean)]
|
||||
lat = []
|
||||
for lo, hi, tag in ((1, 3, "b1_3"), (3, 8, "b3_8"), (8, 20, "b8_20")):
|
||||
inbin = sorted((d, i) for d, i in durs if lo <= d < hi)
|
||||
for k in range(20):
|
||||
d, i = inbin[int((k + 0.5) / 20 * len(inbin))]
|
||||
r = clean[i]
|
||||
a = audio_of(r["audio"])
|
||||
uid = f"{tag}_{k:02d}"
|
||||
lat.append(dict(id=uid, wav=write("lat", uid, a), dur=round(len(a) / SR, 3), ref=r["text"], bin=tag, src=[r["id"]]))
|
||||
# 20-60 s: join consecutive utterances within a chapter (ordered by id)
|
||||
by_ch = {}
|
||||
for r in clean:
|
||||
by_ch.setdefault((r["speaker_id"], r["chapter_id"]), []).append(r)
|
||||
chapters = sorted(by_ch)
|
||||
random.Random(SEED).shuffle(chapters)
|
||||
gap = np.zeros(int(0.25 * SR), dtype=np.float32)
|
||||
targets = [20 + (58 - 20) * k / 19 for k in range(20)]
|
||||
ci = 0
|
||||
for k, tgt in enumerate(targets):
|
||||
while True:
|
||||
utts = sorted(by_ch[chapters[ci % len(chapters)]], key=lambda r: r["id"])
|
||||
ci += 1
|
||||
parts, refs, ids, n = [], [], [], 0
|
||||
for r in utts:
|
||||
a = audio_of(r["audio"])
|
||||
parts += [a, gap]
|
||||
refs.append(r["text"])
|
||||
ids.append(r["id"])
|
||||
n += len(a) + len(gap)
|
||||
if n / SR >= tgt:
|
||||
break
|
||||
if tgt <= n / SR <= 60:
|
||||
break
|
||||
a = np.concatenate(parts[:-1])
|
||||
uid = f"b20_60_{k:02d}"
|
||||
lat.append(dict(id=uid, wav=write("lat", uid, a), dur=round(len(a) / SR, 3), ref=" ".join(refs), bin="b20_60", src=ids))
|
||||
dump("lat", lat)
|
||||
for tag in ("b1_3", "b3_8", "b8_20", "b20_60"):
|
||||
ds = [r["dur"] for r in lat if r["bin"] == tag]
|
||||
print(" ", tag, "n", len(ds), "min", min(ds), "median", sorted(ds)[len(ds) // 2], "max", max(ds))
|
||||
|
||||
# positive control: 1.5 s of digital silence at 40 % of the utterance
|
||||
pc = []
|
||||
for r, a in samples["ls-clean"]:
|
||||
if len(a) / SR >= 6.0 and len(pc) < 40:
|
||||
b = a.copy()
|
||||
s0 = int(0.40 * len(a))
|
||||
s1 = s0 + int(1.5 * SR)
|
||||
b[s0:s1] = 0.0
|
||||
pc.append(dict(id=r["id"], wav=write("pc", r["id"], b), dur=round(len(b) / SR, 3), ref=r["text"],
|
||||
silence=[round(s0 / SR, 3), round(s1 / SR, 3)]))
|
||||
dump("pc", pc)
|
||||
|
||||
# null control: -0.5 dB gain
|
||||
g = 10 ** (-0.5 / 20)
|
||||
nul = []
|
||||
for r, a in samples["ls-clean"]:
|
||||
nul.append(dict(id=r["id"], wav=write("null", r["id"], a * g), dur=round(len(a) / SR, 3), ref=r["text"]))
|
||||
dump("null", nul)
|
||||
Reference in New Issue
Block a user