"""Build the A/B test sets from the pinned parquets. Seeded; writes 16 kHz mono PCM16 WAVs + manifests. ls-clean, ls-other : 400 utterances each, uniform random sample (seed 20260930) of LibriSpeech test ami : 400 AMI IHM test utterances >= 1.0 s, uniform random sample (same seed) lat : latency clips, 20 per bin, from ALL of test-clean: 1-3 s, 3-8 s, 8-20 s: single utterances at evenly spaced duration quantiles of the bin 20-60 s: consecutive utterances of one chapter joined with 0.25 s silence, targets 20..58 s pc : positive control, 40 ls-clean sample utterances >= 6 s with 1.5 s of digital silence placed at 40 % of the utterance (the reference still holds the words) null : the 400 ls-clean utterances at -0.5 dB gain (a should-not-matter perturbation) Manifests: data/.jsonl rows {id, wav, dur, ref, ...}. Prints counts only. """ import io import json import os import random import numpy as np import pyarrow.parquet as pq import soundfile as sf DL = "/ab/dl" OUT = "/ab/data" SEED = 20260930 SR = 16000 def audio_of(cell): a, sr = sf.read(io.BytesIO(cell["bytes"]), dtype="float32") if a.ndim > 1: a = a.mean(axis=1) assert sr == SR, sr return a.astype(np.float32) def write(set_name, uid, a): d = f"{OUT}/{set_name}" os.makedirs(d, exist_ok=True) p = f"{d}/{uid}.wav" sf.write(p, np.clip(a, -1, 1), SR, subtype="PCM_16") return p def dump(set_name, rows): with open(f"{OUT}/{set_name}.jsonl", "w") as f: for r in rows: f.write(json.dumps(r) + "\n") print(set_name, len(rows), "utts", round(sum(r["dur"] for r in rows) / 60, 1), "min") def libri(split): t = pq.read_table(f"{DL}/openslr__librispeech_asr__all__test.{split}__0000.parquet").to_pylist() return t clean = libri("clean") other = libri("other") print("test-clean", len(clean), "test-other", len(other)) samples = {} for name, tab in (("ls-clean", clean), ("ls-other", other)): rng = random.Random(SEED) pick = rng.sample(range(len(tab)), 400) rows = [] for i in pick: r = tab[i] a = audio_of(r["audio"]) rows.append(dict(id=r["id"], wav=write(name, r["id"], a), dur=round(len(a) / SR, 3), ref=r["text"], speaker=r["speaker_id"], chapter=r["chapter_id"])) samples.setdefault(name, []).append((r, a)) dump(name, rows) # AMI IHM test ami = [] for k in range(4): for r in pq.read_table(f"{DL}/edinburghcstr__ami__ihm__test-0000{k}-of-00004.parquet").to_pylist(): dur = float(r["end_time"]) - float(r["begin_time"]) if dur >= 1.0 and r["text"].strip(): ami.append(r) rng = random.Random(SEED) rows = [] for r in rng.sample(ami, 400): a = audio_of(r["audio"]) rows.append(dict(id=r["audio_id"], wav=write("ami", r["audio_id"], a), dur=round(len(a) / SR, 3), ref=r["text"], meeting=r["meeting_id"], speaker=r["speaker_id"])) print("ami eligible (>=1 s)", len(ami), "meetings in sample", len({r["meeting"] for r in rows})) dump("ami", rows) # latency clips durs = [(len(audio_of(r["audio"])) / SR, i) for i, r in enumerate(clean)] lat = [] for lo, hi, tag in ((1, 3, "b1_3"), (3, 8, "b3_8"), (8, 20, "b8_20")): inbin = sorted((d, i) for d, i in durs if lo <= d < hi) for k in range(20): d, i = inbin[int((k + 0.5) / 20 * len(inbin))] r = clean[i] a = audio_of(r["audio"]) uid = f"{tag}_{k:02d}" lat.append(dict(id=uid, wav=write("lat", uid, a), dur=round(len(a) / SR, 3), ref=r["text"], bin=tag, src=[r["id"]])) # 20-60 s: join consecutive utterances within a chapter (ordered by id) by_ch = {} for r in clean: by_ch.setdefault((r["speaker_id"], r["chapter_id"]), []).append(r) chapters = sorted(by_ch) random.Random(SEED).shuffle(chapters) gap = np.zeros(int(0.25 * SR), dtype=np.float32) targets = [20 + (58 - 20) * k / 19 for k in range(20)] ci = 0 for k, tgt in enumerate(targets): while True: utts = sorted(by_ch[chapters[ci % len(chapters)]], key=lambda r: r["id"]) ci += 1 parts, refs, ids, n = [], [], [], 0 for r in utts: a = audio_of(r["audio"]) parts += [a, gap] refs.append(r["text"]) ids.append(r["id"]) n += len(a) + len(gap) if n / SR >= tgt: break if tgt <= n / SR <= 60: break a = np.concatenate(parts[:-1]) uid = f"b20_60_{k:02d}" lat.append(dict(id=uid, wav=write("lat", uid, a), dur=round(len(a) / SR, 3), ref=" ".join(refs), bin="b20_60", src=ids)) dump("lat", lat) for tag in ("b1_3", "b3_8", "b8_20", "b20_60"): ds = [r["dur"] for r in lat if r["bin"] == tag] print(" ", tag, "n", len(ds), "min", min(ds), "median", sorted(ds)[len(ds) // 2], "max", max(ds)) # positive control: 1.5 s of digital silence at 40 % of the utterance pc = [] for r, a in samples["ls-clean"]: if len(a) / SR >= 6.0 and len(pc) < 40: b = a.copy() s0 = int(0.40 * len(a)) s1 = s0 + int(1.5 * SR) b[s0:s1] = 0.0 pc.append(dict(id=r["id"], wav=write("pc", r["id"], b), dur=round(len(b) / SR, 3), ref=r["text"], silence=[round(s0 / SR, 3), round(s1 / SR, 3)])) dump("pc", pc) # null control: -0.5 dB gain g = 10 ** (-0.5 / 20) nul = [] for r, a in samples["ls-clean"]: nul.append(dict(id=r["id"], wav=write("null", r["id"], a * g), dur=round(len(a) / SR, 3), ref=r["text"])) dump("null", nul)