stacks/index-tts: own FastAPI wrapper for IndexTTS-2 + deploy playbook
Adds a third TTS to the irv-ml1 fleet. IndexTTS-2 is Bilibili's
emotion-controllable zero-shot TTS (paper 2506.21619). Distinguishing
capability vs the existing two: timbre and emotion are disentangled —
clone a voice's timbre from one reference and the emotion from a
different reference, OR set emotion via 8-vector, OR derive it from a
text description. Neither CosyVoice 3 nor Qwen3-TTS-1.7B-Base does
this cleanly in English.
Wrapper is owned end-to-end (~150 lines in app.py) — the only existing
FastAPI fork (csllpr/index-tts-fastapi) targets v1 and is a dormant
single-commit repo. Upstream IndexTTS-2 ships only a Gradio webui.
Layout follows the qwen3-tts pattern:
stacks/index-tts/
Dockerfile — CUDA 12.8 base, IndexTTS pinned to a SHA
app.py — FastAPI: POST /v1/audio/speech + /v1/voices
entrypoint.sh — one-time HF snapshot_download of the weights
compose.yaml — env-driven, GPU pinning support, bind mounts
.env.example — port 8192, fp16, paths
README.md — API examples + comparison vs the other TTS
playbooks/deploy-index-tts.yaml — elway playbook for irv-ml1
Voice and emotion libraries are flat host dirs of WAVs, bind-mounted.
Drop a new <name>.wav and /v1/voices picks it up immediately.
License caveat: IndexTTS-2 weights ship under a custom Bilibili
license (free at our scale, not OSI-open). README documents it.
This commit is contained in:
@@ -0,0 +1,154 @@
|
||||
"""IndexTTS-2 — minimal FastAPI wrapper.
|
||||
|
||||
Upstream (https://github.com/index-tts/index-tts) ships only a Gradio
|
||||
webui; the only existing FastAPI fork (csllpr/index-tts-fastapi) targets
|
||||
v1 and is dormant. We own this wrapper end-to-end.
|
||||
|
||||
API surface:
|
||||
POST /v1/audio/speech OpenAI-compat-ish, see SpeechRequest
|
||||
GET /v1/voices list speakers + emotion references
|
||||
GET /healthz liveness for compose healthcheck
|
||||
|
||||
Voice library is a flat directory of WAVs (one file per voice). Emotion
|
||||
references live in a parallel directory. Both are bind-mounted from the
|
||||
host so cloned voices survive container recreates.
|
||||
|
||||
Emotion control is opt-in and mutually exclusive (audio > vector > text):
|
||||
* emotion_voice — name of a WAV in the emotions dir, used as a SECOND
|
||||
reference whose timbre is ignored but emotion is
|
||||
transferred onto the speaker.
|
||||
* emotion_vector — 8-float [happy, angry, sad, afraid, disgusted,
|
||||
melancholic, surprised, calm].
|
||||
* emotion_text — free text; bundled QwenEmotion model derives the
|
||||
vector ("she said excitedly" → joy spike).
|
||||
|
||||
Without any of these the speaker WAV's natural emotion is reused.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import logging
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import List, Optional
|
||||
|
||||
# IndexTTS pins HF_HUB_CACHE at import time (./checkpoints/hf_cache by
|
||||
# default, see infer_v2.py:4). Override BEFORE the indextts import or
|
||||
# downloads land in the wrong tree.
|
||||
os.environ.setdefault(
|
||||
"HF_HUB_CACHE",
|
||||
os.environ.get("INDEX_TTS_HF_CACHE", "/app/checkpoints/hf_cache"),
|
||||
)
|
||||
|
||||
import soundfile as sf # noqa: E402
|
||||
from fastapi import FastAPI, HTTPException # noqa: E402
|
||||
from fastapi.responses import Response # noqa: E402
|
||||
from pydantic import BaseModel, Field # noqa: E402
|
||||
|
||||
from indextts.infer_v2 import IndexTTS2 # noqa: E402
|
||||
|
||||
# ── config from env ──────────────────────────────────────────────────
|
||||
MODEL_DIR = os.environ.get("INDEX_TTS_MODEL_DIR", "/app/checkpoints")
|
||||
CFG_PATH = os.environ.get("INDEX_TTS_CFG", f"{MODEL_DIR}/config.yaml")
|
||||
VOICES_DIR = Path(os.environ.get("INDEX_TTS_VOICES_DIR", "/app/voices"))
|
||||
EMOTIONS_DIR = Path(os.environ.get("INDEX_TTS_EMOTIONS_DIR", "/app/emotions"))
|
||||
USE_FP16 = os.environ.get("INDEX_TTS_FP16", "1") == "1"
|
||||
DEVICE = os.environ.get("INDEX_TTS_DEVICE") or None # "cuda:0", "cuda:1", or None=auto
|
||||
|
||||
VOICES_DIR.mkdir(parents=True, exist_ok=True)
|
||||
EMOTIONS_DIR.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
logging.basicConfig(level=os.environ.get("INDEX_TTS_LOG_LEVEL", "INFO"))
|
||||
log = logging.getLogger("index-tts")
|
||||
|
||||
log.info(
|
||||
"Loading IndexTTS2 from %s (device=%s, fp16=%s)",
|
||||
MODEL_DIR, DEVICE or "auto", USE_FP16,
|
||||
)
|
||||
tts = IndexTTS2(
|
||||
cfg_path=CFG_PATH,
|
||||
model_dir=MODEL_DIR,
|
||||
use_fp16=USE_FP16,
|
||||
device=DEVICE,
|
||||
)
|
||||
log.info("IndexTTS2 ready")
|
||||
|
||||
app = FastAPI(title="index-tts", version="0.1.0")
|
||||
|
||||
|
||||
class SpeechRequest(BaseModel):
|
||||
model: Optional[str] = "index-tts-2" # accepted but ignored
|
||||
input: str = Field(..., description="Text to synthesize")
|
||||
voice: str = Field(..., description="<name>.wav must exist in voices dir")
|
||||
response_format: str = Field("wav", description="wav (only)")
|
||||
# ── emotion (all optional, mutually exclusive) ──
|
||||
emotion_voice: Optional[str] = Field(
|
||||
None, description="<name>.wav in emotions dir, used as emotion ref"
|
||||
)
|
||||
emotion_vector: Optional[List[float]] = Field(
|
||||
None,
|
||||
description="8 floats: happy, angry, sad, afraid, disgusted, melancholic, surprised, calm",
|
||||
)
|
||||
emotion_text: Optional[str] = Field(
|
||||
None, description="Text describing emotion; QwenEmotion derives vector"
|
||||
)
|
||||
emotion_alpha: float = Field(1.0, ge=0.0, le=1.0)
|
||||
|
||||
|
||||
def _resolve(name: str, root: Path) -> Path:
|
||||
p = root / f"{name}.wav"
|
||||
if not p.is_file():
|
||||
raise HTTPException(status_code=404, detail=f"not found: {root.name}/{name}.wav")
|
||||
return p
|
||||
|
||||
|
||||
@app.get("/healthz")
|
||||
def healthz() -> dict:
|
||||
return {"status": "ok"}
|
||||
|
||||
|
||||
@app.get("/v1/voices")
|
||||
def list_voices() -> dict:
|
||||
return {
|
||||
"voices": sorted(p.stem for p in VOICES_DIR.glob("*.wav")),
|
||||
"emotions": sorted(p.stem for p in EMOTIONS_DIR.glob("*.wav")),
|
||||
}
|
||||
|
||||
|
||||
@app.post("/v1/audio/speech")
|
||||
def synthesize(req: SpeechRequest) -> Response:
|
||||
if req.response_format != "wav":
|
||||
raise HTTPException(status_code=400, detail="only response_format=wav is supported")
|
||||
|
||||
spk = str(_resolve(req.voice, VOICES_DIR))
|
||||
|
||||
# First emotion source set wins.
|
||||
emo_path = None
|
||||
emo_vector = None
|
||||
use_emo_text = False
|
||||
emo_text = None
|
||||
if req.emotion_voice:
|
||||
emo_path = str(_resolve(req.emotion_voice, EMOTIONS_DIR))
|
||||
elif req.emotion_vector is not None:
|
||||
if len(req.emotion_vector) != 8:
|
||||
raise HTTPException(status_code=400, detail="emotion_vector must have 8 elements")
|
||||
emo_vector = list(req.emotion_vector)
|
||||
elif req.emotion_text:
|
||||
use_emo_text = True
|
||||
emo_text = req.emotion_text
|
||||
|
||||
sr, audio = tts.infer(
|
||||
spk_audio_prompt=spk,
|
||||
text=req.input,
|
||||
output_path=None, # in-memory return: (sr, np_int16)
|
||||
emo_audio_prompt=emo_path,
|
||||
emo_alpha=req.emotion_alpha,
|
||||
emo_vector=emo_vector,
|
||||
use_emo_text=use_emo_text,
|
||||
emo_text=emo_text,
|
||||
verbose=False,
|
||||
)
|
||||
|
||||
buf = io.BytesIO()
|
||||
sf.write(buf, audio, sr, format="WAV", subtype="PCM_16")
|
||||
return Response(content=buf.getvalue(), media_type="audio/wav")
|
||||
Reference in New Issue
Block a user