From 7875382aed1d5d7d8aef50acf00814d917bbe538 Mon Sep 17 00:00:00 2001 From: Vuong Hoang Date: Fri, 24 Apr 2026 16:58:19 -0700 Subject: [PATCH] qwen3-tts: add stack + deploy playbook for irv-ml1 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Alibaba's open-weight TTS (Apache 2.0, Jan 2026), deployed via groxaxo/Qwen3-TTS-Openai-Fastapi wrapper. Built locally from a pinned git SHA via docker buildx's git context — no source vendored. 1.7B flagship model by default; 0.6B available via QWEN3_TTS_MODEL env override. Why we need a second TTS stack: cosyvoice 3 emits Chinese phonemes for English content per upstream FunAudioLLM/CosyVoice#1790 (unfixed). Qwen3-TTS is from the same Alibaba team but with English first-class in the checkpoint — 10 languages, 97 ms streaming TTFB, instruction-driven emotion. Coexists with cosyvoice on irv-ml1 (port 8191; cosyvoice keeps 8190). Voice cloning shape DIFFERS from cosyvoice: profile-based, not voice-id. Profiles live under voice_library/profiles// and are referenced as voice="clone:". Path layout: /worktank/qwen3-tts/{cache,voices}/, with cache excluded from restic (regenerable from HF Hub) and voices included (cloned profiles need original reference audio to recreate). playbooks/deploy-qwen3-tts.yaml: 10 steps + 5 verify, idempotent; the wait step polls /health for up to ~10 min to absorb first-run model download. Stack only — restic profile update for /worktank/qwen3-tts/voices/ to follow when this is empirically validated against the GLaDOS voice (the "did Qwen inherit the Chinese-bias bug?" question). --- playbooks/deploy-qwen3-tts.yaml | 103 ++++++++++++++++++++++++++++++++ stacks/qwen3-tts/.env.example | 52 ++++++++++++++++ stacks/qwen3-tts/README.md | 103 ++++++++++++++++++++++++++++++++ stacks/qwen3-tts/compose.yaml | 63 +++++++++++++++++++ 4 files changed, 321 insertions(+) create mode 100644 playbooks/deploy-qwen3-tts.yaml create mode 100644 stacks/qwen3-tts/.env.example create mode 100644 stacks/qwen3-tts/README.md create mode 100644 stacks/qwen3-tts/compose.yaml diff --git a/playbooks/deploy-qwen3-tts.yaml b/playbooks/deploy-qwen3-tts.yaml new file mode 100644 index 0000000..525a327 --- /dev/null +++ b/playbooks/deploy-qwen3-tts.yaml @@ -0,0 +1,103 @@ +# Deploy Qwen3-TTS to irv-ml1. +# +# Builds the image locally from groxaxo/Qwen3-TTS-Openai-Fastapi via +# docker buildx's git URL context (no source vendored on the host), +# stages compose + .env under /opt/docker/compose/qwen3-tts/, brings +# up, waits for /health, and verifies the web surface + MCP-irrelevant +# REST endpoints answer. +# +# First run is slow: ~3-5 min for the docker build (CUDA torch + +# transformers wheels) plus ~3-5 min for the 1.7B model download +# from HF on first inference / warmup. The healthz wait below has +# a generous deadline. +# +# Usage: +# scripts/elway irv-ml1 --playbook playbooks/deploy-qwen3-tts.yaml +# +# Idempotent — every step is creates-/when-gated; rerun is safe. + +vars: + compose_dir: /opt/docker/compose/qwen3-tts + cache_dir: /worktank/qwen3-tts/cache + voices_dir: /worktank/qwen3-tts/voices + host_port: "8191" + +steps: + # ── host-side dirs ────────────────────────────────────────────────── + + - name: Ensure /worktank/qwen3-tts root exists (one-time, sudo) + shell: mkdir -p /worktank/qwen3-tts + sudo: true + creates: /worktank/qwen3-tts + + - name: Chown /worktank/qwen3-tts to lkraven + shell: chown lkraven:lkraven /worktank/qwen3-tts + sudo: true + when: '[ "$(stat -c %U /worktank/qwen3-tts)" != lkraven ]' + + - name: Ensure cache dir exists + shell: mkdir -p {{ cache_dir }} + creates: "{{ cache_dir }}" + + - name: Ensure voices dir exists + shell: mkdir -p {{ voices_dir }} + creates: "{{ voices_dir }}" + + - name: Ensure compose dir exists + shell: mkdir -p {{ compose_dir }} + creates: "{{ compose_dir }}" + + # ── deploy compose files ──────────────────────────────────────────── + + - name: Upload compose.yaml + upload: + src: stacks/qwen3-tts/compose.yaml + dest: "{{ compose_dir }}/compose.yaml" + mode: "0644" + + - name: Seed .env from template (only if absent) + upload: + src: stacks/qwen3-tts/.env.example + dest: "{{ compose_dir }}/.env" + mode: "0644" + when: "[ ! -f {{ compose_dir }}/.env ]" + + # ── build + bring up ──────────────────────────────────────────────── + + - name: docker compose build (~3-5 min first time; cached after) + shell: cd {{ compose_dir }} && docker compose build + + - name: docker compose up -d + shell: cd {{ compose_dir }} && docker compose up -d + + - name: Wait for /health to respond + # 1.7B model download on first boot can take a few minutes; allow + # up to ~10 minutes for the wait, polling every 5s. + shell: | + for i in $(seq 1 120); do + curl -sf -o /dev/null --max-time 3 http://localhost:{{ host_port }}/health && exit 0 + sleep 5 + done + exit 1 + changed_when: "false" + +verify: + - name: /health returns 200 + shell: curl -sf -o /dev/null http://localhost:{{ host_port }}/health + changed_when: "false" + + - name: /v1/models lists at least one model + shell: curl -sf http://localhost:{{ host_port }}/v1/models | grep -q '"data"\|Qwen3-TTS\|model' + changed_when: "false" + + - name: /v1/voices endpoint reachable + shell: curl -sf -o /dev/null http://localhost:{{ host_port }}/v1/voices + changed_when: "false" + + - name: Web UI root serves HTML + shell: '[ "$(curl -s -o /dev/null -w %{http_code} http://localhost:{{ host_port }}/)" -eq 200 ]' + changed_when: "false" + + - name: Container running + healthy or starting + shell: docker inspect qwen3-tts --format '{{.State.Status}}' | grep -q running + changed_when: "false" diff --git a/stacks/qwen3-tts/.env.example b/stacks/qwen3-tts/.env.example new file mode 100644 index 0000000..f83b86a --- /dev/null +++ b/stacks/qwen3-tts/.env.example @@ -0,0 +1,52 @@ +# Qwen3-TTS stack tunables. Copy to `.env` on irv-ml1 before deploying. + +# ── build pin ──────────────────────────────────────────────────────── +# SHA of groxaxo/Qwen3-TTS-Openai-Fastapi to build from. Bump + rebuild +# when you want upstream wrapper updates. +QWEN3_TTS_SHA=10323ce778c48a75dbda93d0a4891983fb371f58 + +# Local image tag — bump when you change build context to force a +# fresh layer build. +QWEN3_TTS_TAG=v1 + +# ── network ────────────────────────────────────────────────────────── +# Host port (container listens on 8880 internally). +QWEN3_TTS_PORT=8191 + +# Bind address. 0.0.0.0 exposes on all interfaces (incl. WG tunnel +# interface 10.100.79.3); 127.0.0.1 restricts to local-only. +QWEN3_TTS_BIND=0.0.0.0 + +# ── runtime ────────────────────────────────────────────────────────── +# Inference backend. `official` = default upstream; `optimized` = +# faster but slightly less robust; `vllm_omni` = vLLM-backed (needs +# more VRAM); `pytorch` = bare pytorch path. +QWEN3_TTS_BACKEND=official + +# Model variant. 1.7B = flagship, 6–8 GB VRAM with bfloat16, best +# quality + control. 0.6B = lightweight, ~2–3 GB VRAM, faster, slightly +# less expressive. +QWEN3_TTS_MODEL=Qwen/Qwen3-TTS-12Hz-1.7B + +# Warm the model on container start so the first synthesis request +# doesn't pay the load latency. Adds ~30 s to startup. Recommended. +QWEN3_TTS_WARMUP=true + +# Concurrency cap on synthesis requests. Single GPU + 1.7B model → +# leave at 1 unless you're load-testing. +QWEN3_TTS_MAX_CONCURRENT=1 + +# Mount the gradio voice-studio UI at /voice-studio for browser-side +# voice cloning. Set "false" to disable for headless deployments. +QWEN3_TTS_VOICE_STUDIO=true + +# ── persistent storage on the host ─────────────────────────────────── +# HuggingFace cache (model weights, ~5 GB after first run). Bind-mounted +# so model state survives container recreate. Excluded from restic +# (regenerable from HF Hub). +QWEN3_TTS_CACHE_DIR=/worktank/qwen3-tts/cache + +# Cloned voice profiles (meta.json + reference.wav per voice). Precious +# — cloned voices need the original reference audio to recreate. +# Included in restic. +QWEN3_TTS_VOICES_DIR=/worktank/qwen3-tts/voices diff --git a/stacks/qwen3-tts/README.md b/stacks/qwen3-tts/README.md new file mode 100644 index 0000000..4bca156 --- /dev/null +++ b/stacks/qwen3-tts/README.md @@ -0,0 +1,103 @@ +# Qwen3-TTS + +Alibaba's open-weight TTS (Apache 2.0, released Jan 2026), deployed +via the [groxaxo/Qwen3-TTS-Openai-Fastapi](https://github.com/groxaxo/Qwen3-TTS-Openai-Fastapi) +OpenAI-compatible wrapper. + +**Server:** irv-ml1 (Irvine, WireGuard-only) +**Port:** 8191 (container 8880) +**GPUs:** both exposed (`NVIDIA_VISIBLE_DEVICES=all`); 1.7B model +fits on either the RTX 3090 (24 GB) or A6000 (48 GB) with headroom +**Image:** `local/qwen3-tts:v1` — built locally from a pinned git SHA +of the wrapper repo via docker buildx's git URL context + +## Why this stack alongside cosyvoice + +CosyVoice 3 (the other stack on this host) emits Chinese-flavored +phonemes when given English content. Confirmed against upstream +issue [FunAudioLLM/CosyVoice#1790](https://github.com/FunAudioLLM/CosyVoice/issues/1790) +— closed without a fix as of 2026-04-14. + +Qwen3-TTS is from the same Alibaba team but built English-first into +the checkpoint: 10 languages, 97 ms streaming TTFB, instruction-driven +emotional expression, voice cloning. It's the better choice for any +English-narration use; keep CosyVoice 3 around for Chinese / dialect +work where it shines. + +## API endpoints + +| Method + path | Purpose | +|---|---| +| `POST /v1/audio/speech` | OpenAI-compatible TTS (drop-in for OpenAI clients) | +| `GET /v1/voices` | List cloned voice profiles | +| `GET /v1/models` | List available model checkpoints | +| `GET /health` | Healthcheck | +| `GET /` | Web UI | +| `GET /voice-studio` | Gradio voice-cloning UI (when `ENABLE_VOICE_STUDIO=true`) | + +## Voice cloning — different shape from cosyvoice + +Profile-based, not voice-id-based: + +1. Clone a voice via the `/voice-studio` web UI (uploads reference + audio + transcript, names the profile, e.g. `glados`). +2. The wrapper writes + `/root/qwen3-tts/voice_library/profiles//{meta.json,reference.wav}` + inside the container (bind-mounted to + `/worktank/qwen3-tts/voices/profiles//` on the host). +3. Reference the profile in synthesis requests as + `voice="clone:"`. + +```bash +# OpenAI-shape, English with emotion via instruction +curl -X POST http://10.100.79.3:8191/v1/audio/speech \ + -H 'Content-Type: application/json' \ + -d '{ + "model": "Qwen/Qwen3-TTS-12Hz-1.7B", + "voice": "clone:glados", + "input": "You should leave now. Visitors are not welcome.", + "instructions": "speak with cold contempt", + "response_format": "wav" + }' \ + -o glados_en_angry.wav +``` + +(Note `instructions` field name; OpenAI standard. Whether English +instructions actually drive emotion correctly here is the open +question this stack exists to test — see the deploy playbook's verify +section.) + +## Path layout + +| Host path | Container path | Purpose | Restic? | +|---|---|---|---| +| `/worktank/qwen3-tts/cache/` | `/root/.cache/huggingface` | Model cache (~5 GB after first run) | excluded | +| `/worktank/qwen3-tts/voices/` | `/root/qwen3-tts/voice_library` | Cloned profiles | **included** (precious) | + +## Deploy + +Via elway — see `playbooks/deploy-qwen3-tts.yaml` in the +eshpfi-management root. The playbook builds the image, ensures the +host dirs, installs compose + .env, brings up. First boot downloads +the 1.7B model from HF (~6 GB), which is the slowest step. + +```bash +scripts/elway irv-ml1 --playbook playbooks/deploy-qwen3-tts.yaml +``` + +To pin a different upstream wrapper SHA, set `--var sha=` or +edit `.env` on the server and rebuild. + +## Switching to the smaller model + +If 1.7B is too heavy or you need to share GPUs more aggressively: + +```bash +ssh irv-ml1 ' + cd /opt/docker/compose/qwen3-tts && \ + sed -i "s|^QWEN3_TTS_MODEL=.*|QWEN3_TTS_MODEL=Qwen/Qwen3-TTS-12Hz-0.6B|" .env && \ + docker compose up -d +' +``` + +The new model auto-downloads on next start (~2–3 GB). diff --git a/stacks/qwen3-tts/compose.yaml b/stacks/qwen3-tts/compose.yaml new file mode 100644 index 0000000..1cd4592 --- /dev/null +++ b/stacks/qwen3-tts/compose.yaml @@ -0,0 +1,63 @@ +# Qwen3-TTS — Alibaba's open-weight TTS, deployed via the +# groxaxo/Qwen3-TTS-Openai-Fastapi wrapper. +# +# Why this stack exists alongside cosyvoice: CosyVoice 3 emits +# Chinese phonemes for non-Chinese inputs (upstream issue +# FunAudioLLM/CosyVoice#1790, no fix). Qwen3-TTS is from the same +# Alibaba team but with English first-class — 10 languages, 97 ms +# streaming TTFB, voice cloning, instruction-driven emotional +# expression. Released Jan 2026, Apache 2.0. +# +# Build: no prebuilt image; pinned to a SHA via docker buildx's git +# context URL so subsequent rebuilds are reproducible. ~5–10 min on +# first build (CUDA torch + transformers). +# +# Model: 1.7B flagship (~6–8 GB VRAM with bfloat16) by default; the +# host has plenty of VRAM. Switch to the 0.6B in .env if you ever +# need more headroom. +# +# Voice cloning shape DIFFERS from cosyvoice: profile-based, not +# voice-id. Profiles live under voice_library/profiles// with +# meta.json + reference.wav, and are referenced as +# `voice="clone:"` in /v1/audio/speech requests. +# +# All tunables live in .env — edit that, not this file. + +services: + qwen3-tts: + image: local/qwen3-tts:${QWEN3_TTS_TAG} + build: + context: https://github.com/groxaxo/Qwen3-TTS-Openai-Fastapi.git#${QWEN3_TTS_SHA} + dockerfile: Dockerfile + container_name: qwen3-tts + restart: unless-stopped + runtime: nvidia + ports: + - "${QWEN3_TTS_BIND:-0.0.0.0}:${QWEN3_TTS_PORT}:8880" + environment: + - NVIDIA_VISIBLE_DEVICES=all + - PORT=8880 + - TTS_BACKEND=${QWEN3_TTS_BACKEND:-official} + - TTS_MODEL_NAME=${QWEN3_TTS_MODEL:-Qwen/Qwen3-TTS-12Hz-1.7B} + - TTS_WARMUP_ON_START=${QWEN3_TTS_WARMUP:-true} + - TTS_MAX_CONCURRENT=${QWEN3_TTS_MAX_CONCURRENT:-1} + - ENABLE_VOICE_STUDIO=${QWEN3_TTS_VOICE_STUDIO:-true} + - VOICE_LIBRARY_DIR=/root/qwen3-tts/voice_library + - HF_HOME=/root/.cache/huggingface + volumes: + - ${QWEN3_TTS_CACHE_DIR}:/root/.cache/huggingface + - ${QWEN3_TTS_VOICES_DIR}:/root/qwen3-tts/voice_library + healthcheck: + test: ["CMD-SHELL", "curl -fsS http://localhost:8880/health >/dev/null || exit 1"] + interval: 30s + timeout: 10s + retries: 3 + # First boot pulls torch + Qwen3-TTS-12Hz-1.7B (~6 GB) and + # optionally warms the model — give it a generous budget. + start_period: 600s + labels: + - homepage.group=AI Systems + - homepage.name=Qwen3-TTS + - homepage.icon=mdi-account-voice + - homepage.description=Multilingual TTS with English-first emotion (irv-ml1) + - homepage.href=http://10.100.79.3:${QWEN3_TTS_PORT}