"""Cut the two public long-form files into ~6-minute windows that the seat can actually take whole. The seat's ONNX graph has a hard ceiling of 5,000 encoder frames (pos_emb_max_len 5000 -> a 9,999-wide relative-position table) = 400 s; anything longer returns HTTP 500. Windows: nominal boundaries every 360 s, each moved to the widest gap between consecutive ground-truth word starts within +-10 s (a pause), cut at the middle of that gap; every window must be < 395 s. Ground truth per window = the timed GT tokens whose start time falls inside it (the investigation's gt/.timed.json, unchanged). Writes data/long/-w.wav and data/long.jsonl {id, file, k, t0, t1, dur, wav, ref_tokens}. """ import json import sys import numpy as np OFFSET = float(sys.argv[1]) if len(sys.argv) > 1 else 0.0 # first cut at OFFSET s (second placement) NAME = sys.argv[2] if len(sys.argv) > 2 else "long" import soundfile as sf ROOT = "/tank/spikes/scriberr-slicer" OUT = "/tank/spikes/parakeet-ab/data" rows = [] for f in ("wilde", "scotus"): tg = json.load(open(f"{ROOT}/gt/{f}.timed.json")) tok, tim = tg["tokens"], tg["times"] a, sr = sf.read(f"{ROOT}/public/{f}.wav", dtype="float32") total = len(a) / sr starts = np.asarray(tim, float) gaps = [(starts[i + 1] - starts[i], (starts[i] + starts[i + 1]) / 2) for i in range(len(starts) - 1)] cuts = [0.0] k = 1 while total - cuts[-1] > 395: nominal = cuts[-1] + (OFFSET if (OFFSET and len(cuts) == 1) else 360) cand = [(g, m) for g, m in gaps if abs(m - nominal) <= 10] cut = max(cand)[1] if cand else nominal cuts.append(cut) k += 1 cuts.append(total) for k, (t0, t1) in enumerate(zip(cuts, cuts[1:])): assert t1 - t0 < 395, (f, k, t1 - t0) seg = a[int(t0 * sr):int(t1 * sr)] wav = f"{OUT}/{NAME}/{f}-w{k}.wav" import os os.makedirs(f"{OUT}/{NAME}", exist_ok=True) sf.write(wav, seg, sr, subtype="PCM_16") ref = [t for t, s in zip(tok, tim) if t0 <= s < t1] rows.append(dict(id=f"{f}-w{k}", file=f, k=k, t0=round(t0, 2), t1=round(t1, 2), dur=round(t1 - t0, 2), wav=wav, ref_tokens=ref)) print(f, k, round(t0, 1), round(t1, 1), "dur", round(t1 - t0, 1), "ref words", len(ref)) with open(f"{OUT}/{NAME}.jsonl", "w") as fo: for r in rows: fo.write(json.dumps(r) + "\n")