#!/usr/bin/env python3 """Derive per-engine reference sets from the canonical voice corpus. Reads manifest.yaml + engines.yaml and, for a chosen engine, writes derived//.wav (plus .txt when the engine needs a transcript). Usage: python derive.py [voice ...] # default: every voice in manifest Deps: pyyaml, soundfile. faster-whisper is imported lazily, only when an engine sets ref_sentence_bounded (dots) — it picks a clean sentence-boundary trim and its exact transcript. Run under a venv that has these (on irv-ml1 the dots + whisper venvs already do). Known follow-up: `resample: true` engines (chatterbox, zonos) currently COPY the canonical clip at its source SR rather than resampling — a proper resample step (soundfile + a resampler) is a TODO. dots sets resample:false (it resamples internally at load), so the dots path is complete. """ import sys import wave import pathlib import shutil import yaml ROOT = pathlib.Path(__file__).parent def load(): manifest = yaml.safe_load((ROOT / "manifest.yaml").read_text())["voices"] engines = yaml.safe_load((ROOT / "engines.yaml").read_text())["engines"] return manifest, engines DANGLING = {"and", "but", "so", "or", "the", "a", "an", "that", "to", "my", "because", "with", "of", "for", "as", "i", "we", "it", "is"} def sentence_bounded_trim(src, target_s, model, min_s=6.0): """Return (end_seconds, transcript) for a clip ending on a real sentence boundary. Accumulates whisper segments and takes the FIRST point past `min_s` where the running transcript ends in . ! ? — searching up to target_s+4 so a run-on conversational source (no boundary early) still lands on a real sentence end rather than a dangling clause. Only if the source has no boundary at all in that window does it fall back to a best-effort trim with the trailing dangling conjunction/article stripped — a partial-clause tail is exactly what dots.tts regurgitates into its output. """ target_s = float(target_s) if target_s else 10.0 hard_max = target_s + 4.0 segs = list(model.transcribe(src, beam_size=5)[0]) acc, best_end, best_txt = [], None, None for s in segs: if s.end > hard_max: break acc.append(s) txt = " ".join(x.text.strip() for x in acc).strip() if txt.endswith((".", "!", "?")): best_end, best_txt = s.end, txt if s.end >= min_s: break if best_end is not None: return best_end, best_txt or "" # no sentence boundary in-window — best effort, strip the dangling tail end = acc[-1].end if acc else 0.0 words = " ".join(x.text.strip() for x in acc).strip().rstrip(",").split() while words and words[-1].lower().strip(",.") in DANGLING: words.pop() return end, " ".join(words) def trim_wav(src, dst, end_s): w = wave.open(str(src)) sr = w.getframerate() frames = w.readframes(int(end_s * sr)) w.close() o = wave.open(str(dst), "w") o.setnchannels(1) o.setsampwidth(2) o.setframerate(sr) o.writeframes(frames) o.close() def main(): if len(sys.argv) < 2: sys.exit("usage: derive.py [voice ...]") engine = sys.argv[1] manifest, engines = load() if engine not in engines: sys.exit(f"unknown engine '{engine}'; have {list(engines)}") prof = engines[engine] names = sys.argv[2:] or list(manifest) outdir = ROOT / "derived" / engine outdir.mkdir(parents=True, exist_ok=True) model = None if prof.get("ref_sentence_bounded"): from faster_whisper import WhisperModel model = WhisperModel("base.en", device="cpu", compute_type="int8") for v in names: vc = manifest[v] src = ROOT / vc["canonical"] dst_wav = outdir / f"{v}.wav" if prof.get("ref_sentence_bounded"): end, txt = sentence_bounded_trim(str(src), prof.get("ref_max_seconds") or 10, model) trim_wav(src, dst_wav, end) if prof.get("needs_transcript"): (outdir / f"{v}.txt").write_text(txt + "\n") print(f"{engine}/{v}: {end:.1f}s sentence-bounded | {txt}") else: # TODO: resample to prof['sample_rate'] when resample:true shutil.copy(src, dst_wav) if prof.get("needs_transcript"): (outdir / f"{v}.txt").write_text((ROOT / vc["transcript"]).read_text()) print(f"{engine}/{v}: copied canonical ({vc.get('source_sr')}Hz) -> {dst_wav.name}") if __name__ == "__main__": main()