Engine-agnostic voice corpus: canonical source clip + transcript per voice, per-engine reference sets derived by derive.py from engines.yaml profiles. First residents donut/glados/emmie/miranda optimized + verified clean for dots.tts (sentence-bounded ref + accurate transcript — dots leaks reference audio into output otherwise). canonical/ + transcripts/ tracked; derived/ gitignored (regenerable). Records the dots.tts burn-in in persistent-memory.
126 lines
4.5 KiB
Python
126 lines
4.5 KiB
Python
#!/usr/bin/env python3
|
|
"""Derive per-engine reference sets from the canonical voice corpus.
|
|
|
|
Reads manifest.yaml + engines.yaml and, for a chosen engine, writes
|
|
derived/<engine>/<voice>.wav (plus <voice>.txt when the engine needs a
|
|
transcript).
|
|
|
|
Usage:
|
|
python derive.py <engine> [voice ...] # default: every voice in manifest
|
|
|
|
Deps: pyyaml, soundfile. faster-whisper is imported lazily, only when an engine
|
|
sets ref_sentence_bounded (dots) — it picks a clean sentence-boundary trim and
|
|
its exact transcript. Run under a venv that has these (on irv-ml1 the dots +
|
|
whisper venvs already do).
|
|
|
|
Known follow-up: `resample: true` engines (chatterbox, zonos) currently COPY the
|
|
canonical clip at its source SR rather than resampling — a proper resample step
|
|
(soundfile + a resampler) is a TODO. dots sets resample:false (it resamples
|
|
internally at load), so the dots path is complete.
|
|
"""
|
|
import sys
|
|
import wave
|
|
import pathlib
|
|
import shutil
|
|
import yaml
|
|
|
|
ROOT = pathlib.Path(__file__).parent
|
|
|
|
|
|
def load():
|
|
manifest = yaml.safe_load((ROOT / "manifest.yaml").read_text())["voices"]
|
|
engines = yaml.safe_load((ROOT / "engines.yaml").read_text())["engines"]
|
|
return manifest, engines
|
|
|
|
|
|
DANGLING = {"and", "but", "so", "or", "the", "a", "an", "that", "to", "my",
|
|
"because", "with", "of", "for", "as", "i", "we", "it", "is"}
|
|
|
|
|
|
def sentence_bounded_trim(src, target_s, model, min_s=6.0):
|
|
"""Return (end_seconds, transcript) for a clip ending on a real sentence
|
|
boundary.
|
|
|
|
Accumulates whisper segments and takes the FIRST point past `min_s` where the
|
|
running transcript ends in . ! ? — searching up to target_s+4 so a run-on
|
|
conversational source (no boundary early) still lands on a real sentence end
|
|
rather than a dangling clause. Only if the source has no boundary at all in
|
|
that window does it fall back to a best-effort trim with the trailing dangling
|
|
conjunction/article stripped — a partial-clause tail is exactly what dots.tts
|
|
regurgitates into its output.
|
|
"""
|
|
target_s = float(target_s) if target_s else 10.0
|
|
hard_max = target_s + 4.0
|
|
segs = list(model.transcribe(src, beam_size=5)[0])
|
|
acc, best_end, best_txt = [], None, None
|
|
for s in segs:
|
|
if s.end > hard_max:
|
|
break
|
|
acc.append(s)
|
|
txt = " ".join(x.text.strip() for x in acc).strip()
|
|
if txt.endswith((".", "!", "?")):
|
|
best_end, best_txt = s.end, txt
|
|
if s.end >= min_s:
|
|
break
|
|
if best_end is not None:
|
|
return best_end, best_txt or ""
|
|
# no sentence boundary in-window — best effort, strip the dangling tail
|
|
end = acc[-1].end if acc else 0.0
|
|
words = " ".join(x.text.strip() for x in acc).strip().rstrip(",").split()
|
|
while words and words[-1].lower().strip(",.") in DANGLING:
|
|
words.pop()
|
|
return end, " ".join(words)
|
|
|
|
|
|
def trim_wav(src, dst, end_s):
|
|
w = wave.open(str(src))
|
|
sr = w.getframerate()
|
|
frames = w.readframes(int(end_s * sr))
|
|
w.close()
|
|
o = wave.open(str(dst), "w")
|
|
o.setnchannels(1)
|
|
o.setsampwidth(2)
|
|
o.setframerate(sr)
|
|
o.writeframes(frames)
|
|
o.close()
|
|
|
|
|
|
def main():
|
|
if len(sys.argv) < 2:
|
|
sys.exit("usage: derive.py <engine> [voice ...]")
|
|
engine = sys.argv[1]
|
|
manifest, engines = load()
|
|
if engine not in engines:
|
|
sys.exit(f"unknown engine '{engine}'; have {list(engines)}")
|
|
prof = engines[engine]
|
|
names = sys.argv[2:] or list(manifest)
|
|
|
|
outdir = ROOT / "derived" / engine
|
|
outdir.mkdir(parents=True, exist_ok=True)
|
|
|
|
model = None
|
|
if prof.get("ref_sentence_bounded"):
|
|
from faster_whisper import WhisperModel
|
|
model = WhisperModel("base.en", device="cpu", compute_type="int8")
|
|
|
|
for v in names:
|
|
vc = manifest[v]
|
|
src = ROOT / vc["canonical"]
|
|
dst_wav = outdir / f"{v}.wav"
|
|
if prof.get("ref_sentence_bounded"):
|
|
end, txt = sentence_bounded_trim(str(src), prof.get("ref_max_seconds") or 10, model)
|
|
trim_wav(src, dst_wav, end)
|
|
if prof.get("needs_transcript"):
|
|
(outdir / f"{v}.txt").write_text(txt + "\n")
|
|
print(f"{engine}/{v}: {end:.1f}s sentence-bounded | {txt}")
|
|
else:
|
|
# TODO: resample to prof['sample_rate'] when resample:true
|
|
shutil.copy(src, dst_wav)
|
|
if prof.get("needs_transcript"):
|
|
(outdir / f"{v}.txt").write_text((ROOT / vc["transcript"]).read_text())
|
|
print(f"{engine}/{v}: copied canonical ({vc.get('source_sr')}Hz) -> {dst_wav.name}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|