feat(tts): migrate off Zonos to chatterbox-fast; drop affect, hold English

Repoint the TTS client from the Zonos gateway (:8890 /v1/audio/speech) to
chatterbox-fast (:8197 /tts — bespoke non-OpenAI {text,voice,format,stream}
schema, no auth, 24kHz, infra-ops-verified). tts.py stays the single swap seam.

Dropped, no backward-compat (pre-v1):
- Affect (DEC-7): the Turbo checkpoint has no emotion knob, so PadState,
  EmotionDials, pad_to_dials, the /api/tts p/a fields, and the browser pad
  argument are deleted. Voice is now flat.
- Client-side chunking (DEC-10): chatterbox has no per-synth cap and chunks
  internally, so chunk_text/tts_stream_long/_pcm_after_header are deleted; a
  single tts_stream call voices a whole turn, the mid-stream yielded_any degrade
  folded into it.
- Language pin (DEC-9): no language field; re-purposed to sampling curbs (below).

Fixed / added:
- Browser Web Audio sample rate 44100 -> 24000 (the chatterbox rate).
- Default voice Cora -> glados_25s; donut registered lowercase at /refs/donut.wav.
- English-drift curb: Turbo is multilingual-leaky and wanders off English on a
  long generation (the gateway scheduler ratchets chunk size unbounded). Tighten
  sampling in gateway_body: top_k 1000->80, top_p 0.95->0.85, temperature
  0.8->0.5. These reduce drift probability; the guaranteed fix is a server-side
  max-chunk cap (infra-ops, greenlit).
- OOM guard (DEC-9a): a long generation can OOM the shared 3090, returning 200
  with a 0-byte body; /api/tts surfaces an empty 200 as 503 rather than
  committing silent audio.

Contract donut_voiced_interview.contract.md amended: migration banner, DEC-1/3/8
amended, DEC-7/9/10 retired with historical notes, DEC-9a added.

Tests rewritten to the new wire; 520 green. Live-smoked against the gateway
(24kHz synth + endpoint proxy + web console). persistent-memory.md committed
alongside (commit-along).
This commit is contained in:
vh
2026-08-07 10:23:13 -07:00
parent 2cc670e4a1
commit 19b499ab50
8 changed files with 393 additions and 760 deletions
+56 -285
View File
@@ -1,29 +1,27 @@
"""Tests for ratatoskr.tts — the STREAMING Zonos-gateway client + PAD→dial mapping.
"""Tests for ratatoskr.tts — the STREAMING chatterbox-fast gateway client.
tts_stream proxies the gateway's chunked response verbatim (no buffering, no header
rewrite — the placeholder-size streaming WAV is meant to be played progressively).
pad_to_dials / PadState / EmotionDials are pure + total.
rewrite — the placeholder-size streaming WAV is meant to be played progressively). It
is the sole synthesis primitive: chatterbox chunks arbitrary-length text internally, so
there is no client-side chunk-and-concatenate (retired with the Zonos migration), and no
affect dials (Turbo has no emotion knob). The mid-stream degrade policy is folded in.
"""
import math
import httpx
import pytest
import respx
from ratatoskr.tts import (
_TTS_CHUNK_CHAR_BUDGET,
EmotionDials,
PadState,
_TTS_TEMPERATURE,
_TTS_TOP_K,
_TTS_TOP_P,
CHATTERBOX_TTS_URL,
TtsUnavailable,
chunk_text,
gateway_body,
pad_to_dials,
tts_stream,
tts_stream_long,
)
_URL = "http://tts.example/v1/audio/speech"
_URL = "http://tts.example/tts"
# The gateway's streaming WAV bytes (placeholder 0xFFFFFFFF sizes). We pass them through
# untouched, so the content only has to round-trip.
_WAV = (
@@ -54,307 +52,80 @@ class _RaisingByteStream(httpx.AsyncByteStream):
pass
class TestPadState:
def test_from_obj_valid_mapping(self) -> None:
pad = PadState.from_obj({"pleasure": 0.5, "arousal": -0.2, "dominance": 0.1})
assert pad == PadState(pleasure=0.5, arousal=-0.2, dominance=0.1)
def test_from_obj_dominance_optional(self) -> None:
pad = PadState.from_obj({"pleasure": 0.5, "arousal": -0.2})
assert pad is not None and pad.dominance == 0.0
def test_from_obj_none_is_none(self) -> None:
assert PadState.from_obj(None) is None
def test_from_obj_non_mapping_is_none(self) -> None:
assert PadState.from_obj("not a mapping") is None
assert PadState.from_obj([0.1, 0.2]) is None
def test_from_obj_missing_key_is_none(self) -> None:
assert PadState.from_obj({"pleasure": 0.5}) is None # no arousal
def test_from_obj_non_numeric_is_none(self) -> None:
assert PadState.from_obj({"pleasure": "hot", "arousal": 0.1}) is None
def test_from_obj_huge_int_overflow_is_none(self) -> None:
huge = int("9" * 400)
assert PadState.from_obj({"pleasure": huge, "arousal": 0}) is None
class TestPadToDials:
def test_none_pad_is_neutral_disabled(self) -> None:
d = pad_to_dials(None)
assert d.emotion_enabled is False
assert d.to_body() == {}
def test_maps_pleasure_and_arousal(self) -> None:
d = pad_to_dials(PadState(pleasure=0.4, arousal=0.6))
assert d.emotion_enabled is True
assert d.emotion_valence == pytest.approx(0.4)
assert d.emotion_arousal == pytest.approx(0.6)
def test_clamps_out_of_range(self) -> None:
d = pad_to_dials(PadState(pleasure=5.0, arousal=-9.0))
assert d.emotion_valence == 1.0
assert d.emotion_arousal == -1.0
def test_nan_degrades_to_zero_never_raises(self) -> None:
d = pad_to_dials(PadState(pleasure=math.nan, arousal=math.inf))
assert d.emotion_valence == 0.0
assert d.emotion_arousal == 1.0
def test_non_padstate_input_degrades_to_neutral(self) -> None:
for bad in ({}, "bad", [0.1, 0.2], object(), 42):
d = pad_to_dials(bad)
assert d.emotion_enabled is False and d.to_body() == {}
class TestEmotionDialsToBody:
def test_disabled_emits_no_params(self) -> None:
assert EmotionDials(emotion_enabled=False).to_body() == {}
def test_enabled_emits_valence_arousal_strength(self) -> None:
body = EmotionDials(
emotion_enabled=True, emotion_valence=0.3, emotion_arousal=-0.1
).to_body()
assert body["emotion_enabled"] is True
assert body["emotion_valence"] == 0.3
assert body["emotion_arousal"] == -0.1
assert "emotion_strength" in body
class TestGatewayBody:
def test_always_wav_with_dials(self) -> None:
b = gateway_body("hi", "donut", pad_to_dials(PadState(pleasure=0.5, arousal=0.2)))
assert b["input"] == "hi" and b["voice"] == "donut"
assert b["response_format"] == "wav" # DEC-3 — ALWAYS wav
assert b["language"] == "en-us" # DEC-9 — pin English conditioning
assert b["emotion_valence"] == pytest.approx(0.5)
def test_bespoke_chatterbox_schema(self) -> None:
b = gateway_body("hi", "donut")
assert b["text"] == "hi" # "text", not "input"
assert b["voice"] == "donut"
assert b["format"] == "wav" # DEC-3 — "format", not "response_format"
assert b["stream"] is True # DEC-2 — play-as-it-arrives
def test_neutral_omits_emotion(self) -> None:
b = gateway_body("hi", "Cora", pad_to_dials(None))
assert "emotion_valence" not in b and b["response_format"] == "wav"
def test_sampling_curbs_below_gateway_defaults_hold_english(self) -> None:
# DEC-9: the model has no `language` pin and Turbo's multilingual capacity leaks under
# high-entropy sampling on long turns. gateway_body tightens temperature/top_p/top_k
# below the gateway defaults (0.8 / 0.95 / 1000) to hold English — top_k the highest-
# leverage. Pin presence + the below-default relationship (infra-ops-authoritative).
b = gateway_body("a long turn", "donut")
assert b["temperature"] == _TTS_TEMPERATURE and _TTS_TEMPERATURE < 0.8
assert b["top_p"] == _TTS_TOP_P and _TTS_TOP_P < 0.95
assert b["top_k"] == _TTS_TOP_K and _TTS_TOP_K < 1000
def test_no_zonos_era_fields(self) -> None:
# The Zonos body fields are gone: no OpenAI `input`/`response_format`, no
# `language` pin (DEC-9 retired), no affect dials (DEC-7 retired).
b = gateway_body("hi", "Cora")
for dead in ("input", "response_format", "language", "emotion_valence",
"emotion_arousal", "emotion_enabled", "emotion_strength"):
assert dead not in b
class TestTtsStream:
@respx.mock
async def test_streams_chunks_and_posts_wav_body(self) -> None:
async def test_streams_chunks_and_posts_bespoke_body(self) -> None:
route = respx.post(_URL).mock(return_value=httpx.Response(200, content=_WAV))
async with httpx.AsyncClient() as client:
out = await _drain(tts_stream(
"hello there", voice="donut",
dials=pad_to_dials(PadState(pleasure=0.5, arousal=0.2)),
client=client, url=_URL,
))
out = await _drain(tts_stream("hello there", voice="donut", client=client, url=_URL))
assert out == _WAV # passed through verbatim — no header rewrite
import json as _json
body = _json.loads(route.calls.last.request.content)
assert body["text"] == "hello there"
assert body["voice"] == "donut"
assert body["response_format"] == "wav"
assert body["language"] == "en-us" # DEC-9 — pin English conditioning
assert body["emotion_valence"] == pytest.approx(0.5)
assert body["format"] == "wav"
assert body["stream"] is True
@respx.mock
async def test_default_url_is_chatterbox(self) -> None:
route = respx.post(CHATTERBOX_TTS_URL).mock(
return_value=httpx.Response(200, content=_WAV)
)
async with httpx.AsyncClient() as client:
await _drain(tts_stream("hi", voice="donut", client=client))
assert route.called # the module default points at the chatterbox gateway
@respx.mock
async def test_non_200_open_raises_before_any_chunk(self) -> None:
respx.post(_URL).mock(return_value=httpx.Response(500, content=b"boom"))
async with httpx.AsyncClient() as client:
with pytest.raises(TtsUnavailable) as exc:
await _drain(tts_stream(
"hi", voice="Cora", dials=pad_to_dials(None), client=client, url=_URL
))
await _drain(tts_stream("hi", voice="Cora", client=client, url=_URL))
assert exc.value.status == 500
@respx.mock
async def test_transport_error_raises(self) -> None:
async def test_transport_error_on_open_raises(self) -> None:
respx.post(_URL).mock(side_effect=httpx.ConnectError("refused"))
async with httpx.AsyncClient() as client:
with pytest.raises(TtsUnavailable):
await _drain(tts_stream(
"hi", voice="Cora", dials=pad_to_dials(None), client=client, url=_URL
))
class TestChunkText:
"""chunk_text (DEC-10): paragraph-first greedy pack, sentence fallback for oversized
paragraphs, clause/word sub-split for oversized sentences; every chunk <= budget."""
def test_empty_and_whitespace_yield_no_chunks(self) -> None:
assert chunk_text("") == []
assert chunk_text(" \n\n \t ") == []
def test_short_text_is_one_chunk(self) -> None:
assert chunk_text("Hello, darling.", budget=100) == ["Hello, darling."]
def test_two_short_paragraphs_greedily_merge(self) -> None:
# Both fit in one budget -> one chunk, joined on the blank-line boundary.
out = chunk_text("First para.\n\nSecond para.", budget=100)
assert out == ["First para.\n\nSecond para."]
def test_paragraphs_split_on_blank_line_when_over_budget(self) -> None:
# Each paragraph fits alone but not together -> a seam on the paragraph boundary.
a, b = "A" * 30, "B" * 30
out = chunk_text(f"{a}\n\n{b}", budget=40)
assert out == [a, b]
def test_oversized_paragraph_falls_back_to_sentences(self) -> None:
para = "One sentence here. Two sentence here. Three sentence here."
out = chunk_text(para, budget=25)
assert all(len(c) <= 25 for c in out)
assert len(out) >= 2
# every word is preserved whole and in order (no split mid-word)
assert [w for c in out for w in c.split()] == para.split()
def test_oversized_sentence_sub_splits_never_mid_word(self) -> None:
sent = "alpha beta gamma delta epsilon zeta eta theta iota kappa lambda"
out = chunk_text(sent, budget=20)
assert all(len(c) <= 20 for c in out)
for c in out:
for word in c.split():
assert word in sent.split() # every emitted token is a whole source word
def test_every_chunk_within_budget_default(self) -> None:
para = ("Princess Donut does not wait. " * 200).strip()
out = chunk_text(para) # default budget
assert out and all(len(c) <= _TTS_CHUNK_CHAR_BUDGET for c in out)
def test_spaceless_over_budget_hard_cuts_as_last_resort(self) -> None:
out = chunk_text("x" * 50, budget=20)
assert all(len(c) <= 20 for c in out)
assert "".join(out) == "x" * 50
def test_non_positive_budget_does_not_hang(self) -> None:
# budget <= 0 would infinite-loop _hard_wrap; it's clamped to 1 so this terminates.
out = chunk_text("alpha beta", budget=0)
assert out and all(len(c) <= 1 for c in out)
assert "".join(out) == "alphabeta" # every char preserved, forward progress made
def test_oversized_sentence_prefers_clause_boundary_over_space(self) -> None:
# A comma-bearing over-budget sentence sub-splits at the CLAUSE boundary (", "),
# not merely at the last space — pins the _CLAUSE_BOUNDARIES preference (else dead).
out = chunk_text("alpha, beta gamma delta", budget=12)
assert all(len(c) <= 12 for c in out)
assert out[0] == "alpha," # clause cut, not "alpha, beta" (a space-only cut)
def test_default_budget_is_the_dec10_value(self) -> None:
# Pin the concrete 747 that FN chunk_text's POST commits to (75% of 71.2s @ 14 c/s).
# The suite's other budget checks compare against the imported constant and so move
# with it; this one anchors the value itself so a retune is a deliberate edit here.
assert _TTS_CHUNK_CHAR_BUDGET == 747
class TestTtsStreamLong:
"""tts_stream_long (DEC-10): concatenate per-chunk synthesis into ONE int16-PCM stream
— chunk 1 verbatim (header + PCM), chunks 2..N header-stripped."""
_PCM = b"\x11\x22" * 64
_WAV_CHUNK = (
b"RIFF\xff\xff\xff\xffWAVEfmt \x10\x00\x00\x00" + b"\x00" * 20
+ b"data\xff\xff\xff\xff" + _PCM
)
@respx.mock
async def test_single_chunk_passes_through_verbatim(self) -> None:
respx.post(_URL).mock(return_value=httpx.Response(200, content=self._WAV_CHUNK))
async with httpx.AsyncClient() as client:
out = await _drain(tts_stream_long(
"Short line.", voice="donut", dials=pad_to_dials(None),
client=client, url=_URL, budget=100,
))
assert out == self._WAV_CHUNK # one chunk => untouched
@respx.mock
async def test_multi_chunk_emits_one_header_then_concatenated_pcm(self) -> None:
respx.post(_URL).mock(return_value=httpx.Response(200, content=self._WAV_CHUNK))
text = "First part here. Second part here. Third part here." # budget 18 -> >=2 chunks
async with httpx.AsyncClient() as client:
out = await _drain(tts_stream_long(
text, voice="donut", dials=pad_to_dials(None), client=client, url=_URL, budget=18,
))
n = len(chunk_text(text, budget=18))
assert n >= 2
assert out.count(b"RIFF") == 1 and out.count(b"data") == 1 # exactly one header
# EXACT bytes: chunk 1 verbatim (header+PCM), chunks 2..N stripped to PCM. Asserting
# the exact stream catches a di+4-vs-di+8 strip off-by-one (2-byte sample alignment
# across seams) that a header-count check alone would miss.
assert out == self._WAV_CHUNK + self._PCM * (n - 1)
# DEC-10: identical voice+dials+language on EVERY chunk (uniform delivery across seams).
import json as _json
bodies = [_json.loads(c.request.content) for c in respx.calls]
assert len(bodies) == n
assert all(b["voice"] == "donut" and b["language"] == "en-us" for b in bodies)
await _drain(tts_stream("hi", voice="Cora", client=client, url=_URL))
@respx.mock
async def test_mid_stream_drop_after_first_byte_degrades_not_raises(self) -> None:
# A2: chunk 0 opens 200, yields bytes, then drops mid-stream. Because the 200 is
# committed (bytes already flowed), this must DEGRADE (return what streamed), never
# raise — the pivot is yielded_any, not the chunk index.
# The 200 is committed once bytes flow; a later transport drop must DEGRADE
# (return what streamed), never raise — the pivot is yielded_any, folded in from
# the retired tts_stream_long. Keeps a committed StreamingResponse from an ASGI trace.
respx.post(_URL).mock(
return_value=httpx.Response(200, stream=_RaisingByteStream(self._WAV_CHUNK))
return_value=httpx.Response(200, stream=_RaisingByteStream(_WAV))
)
async with httpx.AsyncClient() as client:
out = await _drain(tts_stream_long(
"hi", voice="donut", dials=pad_to_dials(None), client=client, url=_URL, budget=100,
))
assert out == self._WAV_CHUNK # head kept, no raise
@respx.mock
async def test_later_chunk_gateway_500_degrades_keeps_prior(self) -> None:
# chunk 1 = valid WAV; chunk 2 = a gateway 500 (OPEN failure on a later chunk).
respx.post(_URL).mock(side_effect=[
httpx.Response(200, content=self._WAV_CHUNK),
httpx.Response(500, content=b"boom"),
])
text = "First part here. Second part here." # budget 18 -> 2 chunks
async with httpx.AsyncClient() as client:
out = await _drain(tts_stream_long(
text, voice="donut", dials=pad_to_dials(None), client=client, url=_URL, budget=18,
))
assert out == self._WAV_CHUNK # INV-TTS-4 degrade: keep chunk 1, drop the tail, no raise
@respx.mock
async def test_later_chunk_missing_data_degrades_keeps_prior(self) -> None:
# chunk 1 = valid WAV; chunk 2 = a 200 non-WAV body (no `data` chunk) -> degrade.
respx.post(_URL).mock(side_effect=[
httpx.Response(200, content=self._WAV_CHUNK),
httpx.Response(200, content=b"xxxxx no marker present xxxxx"),
])
text = "First part here. Second part here." # budget 18 -> 2 chunks
async with httpx.AsyncClient() as client:
out = await _drain(tts_stream_long(
text, voice="donut", dials=pad_to_dials(None), client=client, url=_URL, budget=18,
))
# INV-TTS-4 degrade: chunk 1 audio retained verbatim, chunk 2 dropped (no raise, no
# garbage bytes emitted from the malformed body).
assert out == self._WAV_CHUNK
@respx.mock
async def test_first_chunk_gateway_failure_raises(self) -> None:
respx.post(_URL).mock(return_value=httpx.Response(500, content=b"boom"))
async with httpx.AsyncClient() as client:
with pytest.raises(TtsUnavailable):
await _drain(tts_stream_long(
"hi", voice="Cora", dials=pad_to_dials(None),
client=client, url=_URL, budget=100,
))
async def test_pcm_after_header_reassembles_data_marker_across_reads(self) -> None:
# The `data` marker can straddle two network reads; _pcm_after_header must accumulate
# until it lands, then yield only the PCM after it. Pins the docstring's straddle claim.
from ratatoskr.tts import _pcm_after_header
async def _split_stream():
yield b"RIFF\xff\xff\xff\xffWAVEfmt \x10\x00\x00\x00" + b"\x00" * 20 + b"da"
yield b"ta\xff\xff\xff\xff" + b"\x11\x22" * 4 # rest of 'data' + size + PCM
out = await _drain(_pcm_after_header(_split_stream()))
assert out == b"\x11\x22" * 4 # PCM only; marker reassembled across the read boundary
async def test_pcm_after_header_no_data_marker_raises(self) -> None:
from ratatoskr.tts import _pcm_after_header
async def _no_marker():
yield b"xxxxx no marker present xxxxx"
with pytest.raises(TtsUnavailable):
await _drain(_pcm_after_header(_no_marker()))
out = await _drain(tts_stream("hi", voice="donut", client=client, url=_URL))
assert out == _WAV # head kept, no raise