revert(chatterbox-fast): drop context-priming (§1.6) — discard-cut leaks context
Revert the priming feature fromd707439. Live A/B caught an audible artifact: the context-priming discard-cut left part of the throwaway prefix in the output, so a clause ("...without a trace of sarcasm,") was spoken an extra time. Root cause is structural: generate() returns one finished waveform with no marker for where the prefix ends, and the model renders the same prefix with different timing when followed by content than when generated solo — so the duration-estimate + energy-minimum cut is a guess and can leave a sliver (or a whole clause) of prefix in. A reliable cut would need token-level access (the abandoned native-streaming arc) or a per-chunk ASR/alignment pass (heavy, still imperfect, eats the latency budget). Fails the agreed bar: "keep only if it closes the gap without a seam." Kept fromd707439: the .gitignore (build artifacts). NOT re-applied: the bundled margin_first fix — wiring it would shrink chunk 1 (more joins = worse coherence), against the operator's priority, and margin=0.8 there is already starvation-safe. Coherence loss at joins stays an accepted limitation; cold streaming was judged "really good". Phase 1 + Phase 2 parity/perf untouched. Next: Phase 3 deploy.
This commit is contained in:
@@ -63,24 +63,6 @@ class ChunkConfig:
|
||||
# granularity — plan §1.1).
|
||||
max_first_sec: float = 2.0
|
||||
|
||||
# Context-priming at joins (plan §1.6). Prime the first N joins (chunks
|
||||
# 1..N) by prepending the prior sentence as backward prosodic context, then
|
||||
# discarding its audio. 0 ⇒ off. Priming runs a 2nd "context-solo" generate,
|
||||
# so a primed chunk costs ~(2·context + content)/rtf. Priming is AFFORDABILITY-
|
||||
# GATED: a chunk is only primed when that cost fits the buffer; otherwise it
|
||||
# falls back to a cold (unprimed) generate, so priming can never starve the
|
||||
# stream. The earliest joins (smallest buffer) thus self-skip until the buffer
|
||||
# has ratcheted up enough to pay for the extra pass.
|
||||
prime_first_n: int = 0
|
||||
|
||||
# Priming headroom: only prime when the buffer is at least this multiple of
|
||||
# the primed cost, so the 2nd pass doesn't flatten the buffer below the slack
|
||||
# the NEXT chunk needs to absorb RTF-estimate error. At 1.5, priming fires on
|
||||
# the early joins for any GPU at/above the rtf_prior floor (3.4 = 3090; A6000
|
||||
# ~3.8–4.0), and on a slower-than-fleet GPU it self-skips entirely (degrades
|
||||
# to cold/unprimed) rather than starving.
|
||||
prime_buffer_factor: float = 1.5
|
||||
|
||||
|
||||
# ── result record ─────────────────────────────────────────────────────────
|
||||
|
||||
@@ -100,7 +82,6 @@ class ChunkResult:
|
||||
drained: float # seconds the buffer ran dry during gen (>0 ⇒ starvation)
|
||||
rtf: float # measured RTF after this chunk
|
||||
sec_per_char: float # measured sec/char after this chunk
|
||||
primed: bool = False # context-priming was applied to this chunk
|
||||
|
||||
@property
|
||||
def starved(self) -> bool:
|
||||
@@ -156,11 +137,6 @@ def _est_gen_time(text: str, *, rtf: float, sec_per_char: float) -> float:
|
||||
return (len(text) * sec_per_char) / rtf
|
||||
|
||||
|
||||
def _est_primed_gen_time(content: str, context: str, *, rtf: float, sec_per_char: float) -> float:
|
||||
"""Primed cost = context-solo pass + joint(context+content) pass."""
|
||||
return ((2 * len(context) + len(content)) * sec_per_char) / rtf
|
||||
|
||||
|
||||
def plan_chunk(
|
||||
remaining: Sequence[str],
|
||||
buffer_remaining: float,
|
||||
@@ -168,27 +144,19 @@ def plan_chunk(
|
||||
margin: float,
|
||||
rtf: float,
|
||||
sec_per_char: float,
|
||||
prime_context: str | None = None,
|
||||
) -> tuple[str, list[str]]:
|
||||
"""Greedily accumulate whole units until the next would blow the budget.
|
||||
|
||||
Always returns at least one unit (never empty, never splits a unit). With
|
||||
``buffer_remaining == 0`` (the first chunk) the budget is 0, so exactly the
|
||||
first unit is taken — which is the latency-critical chunk-1 rule.
|
||||
|
||||
When ``prime_context`` is set the chunk will be context-primed, so packing
|
||||
uses the (larger) primed cost estimate to leave room for the 2nd pass.
|
||||
"""
|
||||
budget = margin * buffer_remaining
|
||||
chunk = [remaining[0]]
|
||||
i = 1
|
||||
while i < len(remaining):
|
||||
candidate = " ".join(chunk + [remaining[i]])
|
||||
if prime_context is not None:
|
||||
est = _est_primed_gen_time(candidate, prime_context, rtf=rtf, sec_per_char=sec_per_char)
|
||||
else:
|
||||
est = _est_gen_time(candidate, rtf=rtf, sec_per_char=sec_per_char)
|
||||
if est > budget:
|
||||
if _est_gen_time(candidate, rtf=rtf, sec_per_char=sec_per_char) > budget:
|
||||
break
|
||||
chunk.append(remaining[i])
|
||||
i += 1
|
||||
@@ -226,10 +194,8 @@ def _ema(old: float, new: float, alpha: float) -> float:
|
||||
|
||||
# ── the online loop ───────────────────────────────────────────────────────
|
||||
|
||||
# generate(text, context) -> (audio_payload, audio_seconds)
|
||||
# context is the prior sentence to prime backward prosody (discarded by the
|
||||
# generator), or None for an unprimed chunk.
|
||||
GenerateFn = Callable[[str, "str | None"], "tuple[object, float]"]
|
||||
# generate(text) -> (audio_payload, audio_seconds)
|
||||
GenerateFn = Callable[[str], "tuple[object, float]"]
|
||||
ClockFn = Callable[[], float]
|
||||
|
||||
|
||||
@@ -261,7 +227,6 @@ def stream_chunks(
|
||||
buffer_remaining = 0.0
|
||||
remaining: list[str] = units
|
||||
index = 0
|
||||
prev_text: str | None = None
|
||||
|
||||
while remaining:
|
||||
first = index == 0
|
||||
@@ -271,34 +236,14 @@ def stream_chunks(
|
||||
remaining = relieve_leader(
|
||||
remaining, buffer_remaining, rtf=rtf, sec_per_char=sec_per_char
|
||||
)
|
||||
# margin_first tightens the FIRST transition (planning chunk 1 off chunk
|
||||
# 0's small buffer — highest starvation risk, plan §1.5).
|
||||
margin = cfg.margin_first if index == 1 else cfg.margin
|
||||
|
||||
# Context-priming (plan §1.6): eligible on chunks 1..N, affordability-gated
|
||||
# so the 2nd pass can never starve the buffer — fall back to cold otherwise.
|
||||
context: str | None = None
|
||||
if 0 < index <= cfg.prime_first_n and prev_text:
|
||||
prev_units = split_sentences(prev_text)
|
||||
candidate_ctx = prev_units[-1] if prev_units else prev_text
|
||||
min_primed = _est_primed_gen_time(
|
||||
remaining[0], candidate_ctx, rtf=rtf, sec_per_char=sec_per_char
|
||||
)
|
||||
if min_primed * cfg.prime_buffer_factor <= buffer_remaining:
|
||||
context = candidate_ctx
|
||||
|
||||
margin = cfg.margin_first if first else cfg.margin
|
||||
chunk_text, remaining = plan_chunk(
|
||||
remaining, buffer_remaining, margin=margin, rtf=rtf,
|
||||
sec_per_char=sec_per_char, prime_context=context,
|
||||
remaining, buffer_remaining, margin=margin, rtf=rtf, sec_per_char=sec_per_char
|
||||
)
|
||||
if context is not None:
|
||||
est_gen = _est_primed_gen_time(chunk_text, context, rtf=rtf, sec_per_char=sec_per_char)
|
||||
else:
|
||||
est_gen = _est_gen_time(chunk_text, rtf=rtf, sec_per_char=sec_per_char)
|
||||
primed = context is not None
|
||||
est_gen = _est_gen_time(chunk_text, rtf=rtf, sec_per_char=sec_per_char)
|
||||
|
||||
t0 = clock()
|
||||
audio, audio_sec = generate(chunk_text, context)
|
||||
audio, audio_sec = generate(chunk_text)
|
||||
gen_time = clock() - t0
|
||||
|
||||
# Starvation: did the buffer run dry while we generated this chunk?
|
||||
@@ -328,7 +273,5 @@ def stream_chunks(
|
||||
drained=drained,
|
||||
rtf=rtf,
|
||||
sec_per_char=sec_per_char,
|
||||
primed=primed,
|
||||
)
|
||||
prev_text = chunk_text
|
||||
index += 1
|
||||
|
||||
Reference in New Issue
Block a user