Files
esh-pfi-infrastructure/services/parakeet-ab-2026-09-30/code/prep_data.py
T
vh a6c1d3c454 docs(parakeet): seat A/B vs parakeet-unified-en-0.6b - latency is the int8-on-CPU runtime; unified wins WER
A/B of the live STT seat (fv-ml1 GPU 0, sherpa-onnx int8 v3) against
nvidia/parakeet-unified-en-0.6b, measured on GPU 3 with the seat's own image,
k2-fsa's published unified int8 export, fp32/fp16 exports made with k2-fsa's
recipe, v2 int8, and NeMo 3.0.0 (fp32, bf16 autocast, bf16 weights).

- Seat int8 graph runs on one CPU thread (cpu/wall 1.00, GPU 2-9%).
- unified-en under NeMo: -121/-234/-530 ms vs the seat at 1-3/3-8/8-20 s
  (paired, n=120/bin; floor <=6 ms; +50 ms positive control reads +52-54).
- unified-en WER lower in every runtime: -0.7 pp clean, -1.5 pp other,
  -3.2 to -4.4 pp AMI (paired CIs exclude 0).
- Seat defects found: hard 400 s input ceiling (HTTP 500), truncation after
  a quiet 1.5 s pause, and severe long-window dropouts (int8 v3 only).
- B-bf16w needs +0.8 to +1.5 GB over the seat's 1,690 MiB on GPU 0.

Raw requests, hypotheses, manifests and the full harness under
services/parakeet-ab-2026-09-30/. No deploy; live seat untouched apart
from 240 light test requests.
2026-09-30 18:51:44 -07:00

150 lines
5.5 KiB
Python

"""Build the A/B test sets from the pinned parquets. Seeded; writes 16 kHz mono PCM16 WAVs + manifests.
ls-clean, ls-other : 400 utterances each, uniform random sample (seed 20260930) of LibriSpeech test
ami : 400 AMI IHM test utterances >= 1.0 s, uniform random sample (same seed)
lat : latency clips, 20 per bin, from ALL of test-clean:
1-3 s, 3-8 s, 8-20 s: single utterances at evenly spaced duration quantiles of the bin
20-60 s: consecutive utterances of one chapter joined with 0.25 s silence, targets 20..58 s
pc : positive control, 40 ls-clean sample utterances >= 6 s with 1.5 s of digital silence
placed at 40 % of the utterance (the reference still holds the words)
null : the 400 ls-clean utterances at -0.5 dB gain (a should-not-matter perturbation)
Manifests: data/<set>.jsonl rows {id, wav, dur, ref, ...}. Prints counts only.
"""
import io
import json
import os
import random
import numpy as np
import pyarrow.parquet as pq
import soundfile as sf
DL = "/ab/dl"
OUT = "/ab/data"
SEED = 20260930
SR = 16000
def audio_of(cell):
a, sr = sf.read(io.BytesIO(cell["bytes"]), dtype="float32")
if a.ndim > 1:
a = a.mean(axis=1)
assert sr == SR, sr
return a.astype(np.float32)
def write(set_name, uid, a):
d = f"{OUT}/{set_name}"
os.makedirs(d, exist_ok=True)
p = f"{d}/{uid}.wav"
sf.write(p, np.clip(a, -1, 1), SR, subtype="PCM_16")
return p
def dump(set_name, rows):
with open(f"{OUT}/{set_name}.jsonl", "w") as f:
for r in rows:
f.write(json.dumps(r) + "\n")
print(set_name, len(rows), "utts", round(sum(r["dur"] for r in rows) / 60, 1), "min")
def libri(split):
t = pq.read_table(f"{DL}/openslr__librispeech_asr__all__test.{split}__0000.parquet").to_pylist()
return t
clean = libri("clean")
other = libri("other")
print("test-clean", len(clean), "test-other", len(other))
samples = {}
for name, tab in (("ls-clean", clean), ("ls-other", other)):
rng = random.Random(SEED)
pick = rng.sample(range(len(tab)), 400)
rows = []
for i in pick:
r = tab[i]
a = audio_of(r["audio"])
rows.append(dict(id=r["id"], wav=write(name, r["id"], a), dur=round(len(a) / SR, 3), ref=r["text"],
speaker=r["speaker_id"], chapter=r["chapter_id"]))
samples.setdefault(name, []).append((r, a))
dump(name, rows)
# AMI IHM test
ami = []
for k in range(4):
for r in pq.read_table(f"{DL}/edinburghcstr__ami__ihm__test-0000{k}-of-00004.parquet").to_pylist():
dur = float(r["end_time"]) - float(r["begin_time"])
if dur >= 1.0 and r["text"].strip():
ami.append(r)
rng = random.Random(SEED)
rows = []
for r in rng.sample(ami, 400):
a = audio_of(r["audio"])
rows.append(dict(id=r["audio_id"], wav=write("ami", r["audio_id"], a), dur=round(len(a) / SR, 3), ref=r["text"],
meeting=r["meeting_id"], speaker=r["speaker_id"]))
print("ami eligible (>=1 s)", len(ami), "meetings in sample", len({r["meeting"] for r in rows}))
dump("ami", rows)
# latency clips
durs = [(len(audio_of(r["audio"])) / SR, i) for i, r in enumerate(clean)]
lat = []
for lo, hi, tag in ((1, 3, "b1_3"), (3, 8, "b3_8"), (8, 20, "b8_20")):
inbin = sorted((d, i) for d, i in durs if lo <= d < hi)
for k in range(20):
d, i = inbin[int((k + 0.5) / 20 * len(inbin))]
r = clean[i]
a = audio_of(r["audio"])
uid = f"{tag}_{k:02d}"
lat.append(dict(id=uid, wav=write("lat", uid, a), dur=round(len(a) / SR, 3), ref=r["text"], bin=tag, src=[r["id"]]))
# 20-60 s: join consecutive utterances within a chapter (ordered by id)
by_ch = {}
for r in clean:
by_ch.setdefault((r["speaker_id"], r["chapter_id"]), []).append(r)
chapters = sorted(by_ch)
random.Random(SEED).shuffle(chapters)
gap = np.zeros(int(0.25 * SR), dtype=np.float32)
targets = [20 + (58 - 20) * k / 19 for k in range(20)]
ci = 0
for k, tgt in enumerate(targets):
while True:
utts = sorted(by_ch[chapters[ci % len(chapters)]], key=lambda r: r["id"])
ci += 1
parts, refs, ids, n = [], [], [], 0
for r in utts:
a = audio_of(r["audio"])
parts += [a, gap]
refs.append(r["text"])
ids.append(r["id"])
n += len(a) + len(gap)
if n / SR >= tgt:
break
if tgt <= n / SR <= 60:
break
a = np.concatenate(parts[:-1])
uid = f"b20_60_{k:02d}"
lat.append(dict(id=uid, wav=write("lat", uid, a), dur=round(len(a) / SR, 3), ref=" ".join(refs), bin="b20_60", src=ids))
dump("lat", lat)
for tag in ("b1_3", "b3_8", "b8_20", "b20_60"):
ds = [r["dur"] for r in lat if r["bin"] == tag]
print(" ", tag, "n", len(ds), "min", min(ds), "median", sorted(ds)[len(ds) // 2], "max", max(ds))
# positive control: 1.5 s of digital silence at 40 % of the utterance
pc = []
for r, a in samples["ls-clean"]:
if len(a) / SR >= 6.0 and len(pc) < 40:
b = a.copy()
s0 = int(0.40 * len(a))
s1 = s0 + int(1.5 * SR)
b[s0:s1] = 0.0
pc.append(dict(id=r["id"], wav=write("pc", r["id"], b), dur=round(len(b) / SR, 3), ref=r["text"],
silence=[round(s0 / SR, 3), round(s1 / SR, 3)]))
dump("pc", pc)
# null control: -0.5 dB gain
g = 10 ** (-0.5 / 20)
nul = []
for r, a in samples["ls-clean"]:
nul.append(dict(id=r["id"], wav=write("null", r["id"], a * g), dur=round(len(a) / SR, 3), ref=r["text"]))
dump("null", nul)