revert(chatterbox-fast): drop context-priming (§1.6) — discard-cut leaks context

Revert the priming feature from d707439. Live A/B caught an audible artifact: the
context-priming discard-cut left part of the throwaway prefix in the output, so a
clause ("...without a trace of sarcasm,") was spoken an extra time.

Root cause is structural: generate() returns one finished waveform with no marker
for where the prefix ends, and the model renders the same prefix with different
timing when followed by content than when generated solo — so the duration-estimate
+ energy-minimum cut is a guess and can leave a sliver (or a whole clause) of prefix
in. A reliable cut would need token-level access (the abandoned native-streaming
arc) or a per-chunk ASR/alignment pass (heavy, still imperfect, eats the latency
budget). Fails the agreed bar: "keep only if it closes the gap without a seam."

Kept from d707439: the .gitignore (build artifacts). NOT re-applied: the bundled
margin_first fix — wiring it would shrink chunk 1 (more joins = worse coherence),
against the operator's priority, and margin=0.8 there is already starvation-safe.

Coherence loss at joins stays an accepted limitation; cold streaming was judged
"really good". Phase 1 + Phase 2 parity/perf untouched. Next: Phase 3 deploy.
This commit is contained in:
vh
2026-06-01 23:26:56 -07:00
parent d707439041
commit 090e70aed5
4 changed files with 22 additions and 190 deletions
+5 -48
View File
@@ -45,19 +45,13 @@ class FakeClock:
def make_generator(clock: FakeClock, *, true_rtf: float, sec_per_char: float):
"""A fake generate() that costs realistic wall-clock and returns audio_sec.
Audio duration is proportional to content length; generation costs
``audio_sec / true_rtf`` of (simulated) wall-clock, advancing the clock. A
primed chunk (context given) costs extra: a context-solo pass plus the
context portion of the joint pass — modelling the ~2× cost the scheduler
must budget for.
Audio duration is proportional to text length; generation costs
``audio_sec / true_rtf`` of (simulated) wall-clock, advancing the clock.
"""
def generate(text: str, context: str | None):
audio_sec = len(text) * sec_per_char # content only (context discarded)
gen_audio = audio_sec
if context:
gen_audio += 2 * len(context) * sec_per_char # ctx-solo + ctx in joint
clock.t += gen_audio / true_rtf
def generate(text: str):
audio_sec = len(text) * sec_per_char
clock.t += audio_sec / true_rtf
return None, audio_sec
return generate
@@ -191,43 +185,6 @@ def test_full_text_reconstructed():
assert joined == " ".join(split_sentences(PARAGRAPH))
def test_priming_fires_but_never_on_chunk_zero():
"""With prime_first_n=2, priming is best-effort (affordability-gated): it fires
on at least one early join at fleet RTF, and NEVER on chunk 0 (latency-critical)."""
cfg = ChunkConfig(prime_first_n=2)
results = run(true_rtf=3.8, cfg=cfg)
assert results[0].primed is False
assert any(r.primed for r in results[1:]), "expected at least one primed join"
# Priming only ever lands on chunks 1..N.
assert all(not r.primed for r in results if r.index > cfg.prime_first_n)
def test_priming_never_starves_at_fleet_rtf():
"""Priming must keep the no-starvation guarantee for any GPU at/above the
rtf_prior floor (3090 ~3.4, A6000 ~3.8–4.0)."""
cfg = ChunkConfig(prime_first_n=2)
for rtf in (4.0, 3.8, 3.4):
results = run(true_rtf=rtf, cfg=cfg)
assert all(not r.starved for r in results), (
rtf, [(r.index, r.drained) for r in results if r.starved]
)
def test_priming_self_skips_on_slow_gpu_no_starvation():
"""On a slower-than-fleet GPU (RTF below the prior), priming gracefully
self-skips rather than starving — the guarantee holds, priming just stops."""
cfg = ChunkConfig(prime_first_n=2)
results = run(true_rtf=3.0, cfg=cfg)
assert all(not r.starved for r in results)
assert not any(r.primed for r in results) # degraded to cold
def test_priming_off_by_default():
"""Default config primes nothing (so context is never requested)."""
results = run(true_rtf=3.8)
assert all(not r.primed for r in results)
def _main():
results = run(true_rtf=3.8)
print(f"{'idx':>3} {'chars':>5} {'audio_s':>8} {'gen_s':>7} "