feat(tts): config-driven voices + two-voice dialogue/narration split (DEC-11)
Voice assignment moves from the hardcoded server map to ~/.config/ratatoskr/ voices.json (per-agent voice + optional narration_voice). An agent with a narration_voice gets a two-voice split: quoted speech in `voice`, narration in `narration_voice`, synthesized per-span and stitched under one WAV header. - new src/ratatoskr/voices.py: load_voice_config (degrade-not-crash), segment_dialogue (quote-based, straight + curly), resolve_voice_spans - tts.py: tts_stream_stitched replaces tts_stream — serial per-span synth, span 0 verbatim, spans 1..N header-stripped -> one gapless 48kHz stream; a single-span list is a byte-identical passthrough (no single-voice regression) - server.py: _tts_endpoint resolves spans from app.state.voice_config; the hardcoded _TTS_VOICE_MAP is retired; create_app gains a voice_config param - entrypoint.py: loads voices.json at startup - contract DEC-11 + INV-TTS-5/6/7; initial config donut->donut, sindra->miranda (dialogue) / emmie (narration) Live-verified on :8765: Sindra mixed turn -> 2 dots calls (emmie+miranda) stitched into one 48kHz WAV with a single RIFF header; Donut single-voice unchanged. 545 tests green (incl. new test_voices.py).
This commit is contained in:
+74
-21
@@ -1,10 +1,10 @@
|
||||
"""Tests for ratatoskr.tts — the STREAMING dots-tts gateway client.
|
||||
|
||||
tts_stream proxies the gateway's chunked response verbatim (no buffering, no header
|
||||
rewrite — the placeholder-size streaming WAV is meant to be played progressively). It
|
||||
is the sole synthesis primitive: dots streams a whole turn from one call, so there is
|
||||
no client-side chunk-and-concatenate, and no affect dials (dots has no emotion knob).
|
||||
The mid-stream degrade policy is folded in.
|
||||
tts_stream_stitched is the sole synthesis primitive: it synthesizes an ordered list of
|
||||
(voice, text) spans serially into one continuous stream — the first span verbatim, later
|
||||
spans header-stripped (DEC-11 two-voice split). A single-span list is a verbatim passthrough
|
||||
(no buffering, no header rewrite), so the single-voice / dialogue-only case is unchanged. No
|
||||
affect dials (dots has no emotion knob). The mid-stream degrade policy spans the sequence.
|
||||
"""
|
||||
|
||||
import httpx
|
||||
@@ -15,7 +15,7 @@ from ratatoskr.tts import (
|
||||
DOTS_TTS_URL,
|
||||
TtsUnavailable,
|
||||
gateway_body,
|
||||
tts_stream,
|
||||
tts_stream_stitched,
|
||||
)
|
||||
|
||||
_URL = "http://tts.example/v1/audio/speech"
|
||||
@@ -25,6 +25,12 @@ _WAV = (
|
||||
b"RIFF\xff\xff\xff\xffWAVEfmt \x10\x00\x00\x00" + b"\x00" * 20
|
||||
+ b"data\xff\xff\xff\xff" + b"\x11\x22" * 64
|
||||
)
|
||||
# A second span's WAV with distinct PCM — its header is stripped when stitched after span 0.
|
||||
_WAV2 = (
|
||||
b"RIFF\xff\xff\xff\xffWAVEfmt \x10\x00\x00\x00" + b"\x00" * 20
|
||||
+ b"data\xff\xff\xff\xff" + b"\x33\x44" * 32
|
||||
)
|
||||
_PCM2 = b"\x33\x44" * 32 # the part of _WAV2 after `data`+size (what stitching keeps)
|
||||
|
||||
|
||||
async def _drain(gen) -> bytes:
|
||||
@@ -34,6 +40,12 @@ async def _drain(gen) -> bytes:
|
||||
return out
|
||||
|
||||
|
||||
def _json_voice(route, i: int) -> str:
|
||||
import json as _json
|
||||
|
||||
return _json.loads(route.calls[i].request.content)["voice"]
|
||||
|
||||
|
||||
class _RaisingByteStream(httpx.AsyncByteStream):
|
||||
"""A 200-body stream that yields `head` then drops mid-stream (an httpx.ReadError, a
|
||||
RequestError subclass) — models a gateway connection drop AFTER the response committed."""
|
||||
@@ -73,12 +85,15 @@ class TestGatewayBody:
|
||||
assert dead not in b
|
||||
|
||||
|
||||
class TestTtsStream:
|
||||
class TestTtsStreamStitched:
|
||||
@respx.mock
|
||||
async def test_streams_chunks_and_posts_openai_body(self) -> None:
|
||||
async def test_single_span_verbatim_and_posts_openai_body(self) -> None:
|
||||
# A single-span list is a verbatim passthrough (INV-TTS-6) with the OpenAI body.
|
||||
route = respx.post(_URL).mock(return_value=httpx.Response(200, content=_WAV))
|
||||
async with httpx.AsyncClient() as client:
|
||||
out = await _drain(tts_stream("hello there", voice="donut", client=client, url=_URL))
|
||||
out = await _drain(
|
||||
tts_stream_stitched([("donut", "hello there")], client=client, url=_URL)
|
||||
)
|
||||
assert out == _WAV # passed through verbatim — no header rewrite
|
||||
import json as _json
|
||||
|
||||
@@ -89,12 +104,39 @@ class TestTtsStream:
|
||||
assert body["stream"] is True
|
||||
|
||||
@respx.mock
|
||||
async def test_default_url_is_dots(self) -> None:
|
||||
route = respx.post(DOTS_TTS_URL).mock(
|
||||
return_value=httpx.Response(200, content=_WAV)
|
||||
async def test_two_spans_stitched_one_header(self) -> None:
|
||||
# Span 0 verbatim (its WAV header + PCM), span 1 header-STRIPPED → one continuous
|
||||
# stream with exactly one leading header (INV-TTS-7). Distinct voices per span.
|
||||
route = respx.post(_URL).mock(
|
||||
side_effect=[httpx.Response(200, content=_WAV), httpx.Response(200, content=_WAV2)]
|
||||
)
|
||||
async with httpx.AsyncClient() as client:
|
||||
await _drain(tts_stream("hi", voice="donut", client=client))
|
||||
out = await _drain(
|
||||
tts_stream_stitched(
|
||||
[("miranda", "spoken bit"), ("emmie", "narrated bit")],
|
||||
client=client, url=_URL,
|
||||
)
|
||||
)
|
||||
assert out == _WAV + _PCM2 # span1's header dropped, PCM kept
|
||||
assert out.count(b"RIFF") == 1 # exactly one WAV header
|
||||
assert _json_voice(route, 0) == "miranda" and _json_voice(route, 1) == "emmie"
|
||||
|
||||
@respx.mock
|
||||
async def test_empty_span_skipped(self) -> None:
|
||||
route = respx.post(_URL).mock(return_value=httpx.Response(200, content=_WAV))
|
||||
async with httpx.AsyncClient() as client:
|
||||
out = await _drain(
|
||||
tts_stream_stitched(
|
||||
[("miranda", " "), ("donut", "real")], client=client, url=_URL
|
||||
)
|
||||
)
|
||||
assert out == _WAV and len(route.calls) == 1 # blank span never synthesized
|
||||
|
||||
@respx.mock
|
||||
async def test_default_url_is_dots(self) -> None:
|
||||
route = respx.post(DOTS_TTS_URL).mock(return_value=httpx.Response(200, content=_WAV))
|
||||
async with httpx.AsyncClient() as client:
|
||||
await _drain(tts_stream_stitched([("donut", "hi")], client=client))
|
||||
assert route.called # the module default points at the dots-tts gateway
|
||||
|
||||
@respx.mock
|
||||
@@ -102,7 +144,7 @@ class TestTtsStream:
|
||||
respx.post(_URL).mock(return_value=httpx.Response(500, content=b"boom"))
|
||||
async with httpx.AsyncClient() as client:
|
||||
with pytest.raises(TtsUnavailable) as exc:
|
||||
await _drain(tts_stream("hi", voice="Cora", client=client, url=_URL))
|
||||
await _drain(tts_stream_stitched([("glados", "hi")], client=client, url=_URL))
|
||||
assert exc.value.status == 500
|
||||
|
||||
@respx.mock
|
||||
@@ -110,16 +152,27 @@ class TestTtsStream:
|
||||
respx.post(_URL).mock(side_effect=httpx.ConnectError("refused"))
|
||||
async with httpx.AsyncClient() as client:
|
||||
with pytest.raises(TtsUnavailable):
|
||||
await _drain(tts_stream("hi", voice="Cora", client=client, url=_URL))
|
||||
await _drain(tts_stream_stitched([("glados", "hi")], client=client, url=_URL))
|
||||
|
||||
@respx.mock
|
||||
async def test_mid_stream_drop_after_first_byte_degrades_not_raises(self) -> None:
|
||||
# The 200 is committed once bytes flow; a later transport drop must DEGRADE
|
||||
# (return what streamed), never raise — the pivot is yielded_any, folded in from
|
||||
# the retired tts_stream_long. Keeps a committed StreamingResponse from an ASGI trace.
|
||||
respx.post(_URL).mock(
|
||||
return_value=httpx.Response(200, stream=_RaisingByteStream(_WAV))
|
||||
# (return what streamed), never raise — the pivot is yielded_any. Keeps a committed
|
||||
# StreamingResponse from an ASGI trace.
|
||||
respx.post(_URL).mock(return_value=httpx.Response(200, stream=_RaisingByteStream(_WAV)))
|
||||
async with httpx.AsyncClient() as client:
|
||||
out = await _drain(tts_stream_stitched([("donut", "hi")], client=client, url=_URL))
|
||||
assert out == _WAV # head kept, no raise
|
||||
|
||||
@respx.mock
|
||||
async def test_later_span_open_fail_after_commit_degrades(self) -> None:
|
||||
# Span 0 commits a 200 + bytes; span 1's OPEN then 500s. Because the stream is already
|
||||
# committed, this DEGRADES (keep span 0), never raises into the 200 (INV-TTS-4).
|
||||
route = respx.post(_URL).mock(
|
||||
side_effect=[httpx.Response(200, content=_WAV), httpx.Response(500, content=b"boom")]
|
||||
)
|
||||
async with httpx.AsyncClient() as client:
|
||||
out = await _drain(tts_stream("hi", voice="donut", client=client, url=_URL))
|
||||
assert out == _WAV # head kept, no raise
|
||||
out = await _drain(
|
||||
tts_stream_stitched([("miranda", "a"), ("emmie", "b")], client=client, url=_URL)
|
||||
)
|
||||
assert out == _WAV and len(route.calls) == 2 # span0 kept, span1 attempted then dropped
|
||||
|
||||
Reference in New Issue
Block a user