A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image, k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights). - Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%). - unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54). - unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other, -3.2 to -4.4 pp AMI (paired CIs exclude 0). - Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only). - B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0. Raw requests, hypotheses, manifests and the full harness under services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart from 240 light test requests.
150 lines
5.5 KiB
Python
150 lines
5.5 KiB
Python
"""Build the A/B test sets from the pinned parquets. Seeded; writes 16 kHz mono PCM16 WAVs + manifests.
|
|
|
|
ls-clean, ls-other : 400 utterances each, uniform random sample (seed 20260930) of LibriSpeech test
|
|
ami : 400 AMI IHM test utterances >= 1.0 s, uniform random sample (same seed)
|
|
lat : latency clips, 20 per bin, from ALL of test-clean:
|
|
1-3 s, 3-8 s, 8-20 s: single utterances at evenly spaced duration quantiles of the bin
|
|
20-60 s: consecutive utterances of one chapter joined with 0.25 s silence, targets 20..58 s
|
|
pc : positive control, 40 ls-clean sample utterances >= 6 s with 1.5 s of digital silence
|
|
placed at 40 % of the utterance (the reference still holds the words)
|
|
null : the 400 ls-clean utterances at -0.5 dB gain (a should-not-matter perturbation)
|
|
Manifests: data/<set>.jsonl rows {id, wav, dur, ref, ...}. Prints counts only.
|
|
"""
|
|
import io
|
|
import json
|
|
import os
|
|
import random
|
|
|
|
import numpy as np
|
|
import pyarrow.parquet as pq
|
|
import soundfile as sf
|
|
|
|
DL = "/ab/dl"
|
|
OUT = "/ab/data"
|
|
SEED = 20260930
|
|
SR = 16000
|
|
|
|
|
|
def audio_of(cell):
|
|
a, sr = sf.read(io.BytesIO(cell["bytes"]), dtype="float32")
|
|
if a.ndim > 1:
|
|
a = a.mean(axis=1)
|
|
assert sr == SR, sr
|
|
return a.astype(np.float32)
|
|
|
|
|
|
def write(set_name, uid, a):
|
|
d = f"{OUT}/{set_name}"
|
|
os.makedirs(d, exist_ok=True)
|
|
p = f"{d}/{uid}.wav"
|
|
sf.write(p, np.clip(a, -1, 1), SR, subtype="PCM_16")
|
|
return p
|
|
|
|
|
|
def dump(set_name, rows):
|
|
with open(f"{OUT}/{set_name}.jsonl", "w") as f:
|
|
for r in rows:
|
|
f.write(json.dumps(r) + "\n")
|
|
print(set_name, len(rows), "utts", round(sum(r["dur"] for r in rows) / 60, 1), "min")
|
|
|
|
|
|
def libri(split):
|
|
t = pq.read_table(f"{DL}/openslr__librispeech_asr__all__test.{split}__0000.parquet").to_pylist()
|
|
return t
|
|
|
|
|
|
clean = libri("clean")
|
|
other = libri("other")
|
|
print("test-clean", len(clean), "test-other", len(other))
|
|
|
|
samples = {}
|
|
for name, tab in (("ls-clean", clean), ("ls-other", other)):
|
|
rng = random.Random(SEED)
|
|
pick = rng.sample(range(len(tab)), 400)
|
|
rows = []
|
|
for i in pick:
|
|
r = tab[i]
|
|
a = audio_of(r["audio"])
|
|
rows.append(dict(id=r["id"], wav=write(name, r["id"], a), dur=round(len(a) / SR, 3), ref=r["text"],
|
|
speaker=r["speaker_id"], chapter=r["chapter_id"]))
|
|
samples.setdefault(name, []).append((r, a))
|
|
dump(name, rows)
|
|
|
|
# AMI IHM test
|
|
ami = []
|
|
for k in range(4):
|
|
for r in pq.read_table(f"{DL}/edinburghcstr__ami__ihm__test-0000{k}-of-00004.parquet").to_pylist():
|
|
dur = float(r["end_time"]) - float(r["begin_time"])
|
|
if dur >= 1.0 and r["text"].strip():
|
|
ami.append(r)
|
|
rng = random.Random(SEED)
|
|
rows = []
|
|
for r in rng.sample(ami, 400):
|
|
a = audio_of(r["audio"])
|
|
rows.append(dict(id=r["audio_id"], wav=write("ami", r["audio_id"], a), dur=round(len(a) / SR, 3), ref=r["text"],
|
|
meeting=r["meeting_id"], speaker=r["speaker_id"]))
|
|
print("ami eligible (>=1 s)", len(ami), "meetings in sample", len({r["meeting"] for r in rows}))
|
|
dump("ami", rows)
|
|
|
|
# latency clips
|
|
durs = [(len(audio_of(r["audio"])) / SR, i) for i, r in enumerate(clean)]
|
|
lat = []
|
|
for lo, hi, tag in ((1, 3, "b1_3"), (3, 8, "b3_8"), (8, 20, "b8_20")):
|
|
inbin = sorted((d, i) for d, i in durs if lo <= d < hi)
|
|
for k in range(20):
|
|
d, i = inbin[int((k + 0.5) / 20 * len(inbin))]
|
|
r = clean[i]
|
|
a = audio_of(r["audio"])
|
|
uid = f"{tag}_{k:02d}"
|
|
lat.append(dict(id=uid, wav=write("lat", uid, a), dur=round(len(a) / SR, 3), ref=r["text"], bin=tag, src=[r["id"]]))
|
|
# 20-60 s: join consecutive utterances within a chapter (ordered by id)
|
|
by_ch = {}
|
|
for r in clean:
|
|
by_ch.setdefault((r["speaker_id"], r["chapter_id"]), []).append(r)
|
|
chapters = sorted(by_ch)
|
|
random.Random(SEED).shuffle(chapters)
|
|
gap = np.zeros(int(0.25 * SR), dtype=np.float32)
|
|
targets = [20 + (58 - 20) * k / 19 for k in range(20)]
|
|
ci = 0
|
|
for k, tgt in enumerate(targets):
|
|
while True:
|
|
utts = sorted(by_ch[chapters[ci % len(chapters)]], key=lambda r: r["id"])
|
|
ci += 1
|
|
parts, refs, ids, n = [], [], [], 0
|
|
for r in utts:
|
|
a = audio_of(r["audio"])
|
|
parts += [a, gap]
|
|
refs.append(r["text"])
|
|
ids.append(r["id"])
|
|
n += len(a) + len(gap)
|
|
if n / SR >= tgt:
|
|
break
|
|
if tgt <= n / SR <= 60:
|
|
break
|
|
a = np.concatenate(parts[:-1])
|
|
uid = f"b20_60_{k:02d}"
|
|
lat.append(dict(id=uid, wav=write("lat", uid, a), dur=round(len(a) / SR, 3), ref=" ".join(refs), bin="b20_60", src=ids))
|
|
dump("lat", lat)
|
|
for tag in ("b1_3", "b3_8", "b8_20", "b20_60"):
|
|
ds = [r["dur"] for r in lat if r["bin"] == tag]
|
|
print(" ", tag, "n", len(ds), "min", min(ds), "median", sorted(ds)[len(ds) // 2], "max", max(ds))
|
|
|
|
# positive control: 1.5 s of digital silence at 40 % of the utterance
|
|
pc = []
|
|
for r, a in samples["ls-clean"]:
|
|
if len(a) / SR >= 6.0 and len(pc) < 40:
|
|
b = a.copy()
|
|
s0 = int(0.40 * len(a))
|
|
s1 = s0 + int(1.5 * SR)
|
|
b[s0:s1] = 0.0
|
|
pc.append(dict(id=r["id"], wav=write("pc", r["id"], b), dur=round(len(b) / SR, 3), ref=r["text"],
|
|
silence=[round(s0 / SR, 3), round(s1 / SR, 3)]))
|
|
dump("pc", pc)
|
|
|
|
# null control: -0.5 dB gain
|
|
g = 10 ** (-0.5 / 20)
|
|
nul = []
|
|
for r, a in samples["ls-clean"]:
|
|
nul.append(dict(id=r["id"], wav=write("null", r["id"], a * g), dur=round(len(a) / SR, 3), ref=r["text"]))
|
|
dump("null", nul)
|