diff --git a/playbooks/deploy-ace-step.yaml b/playbooks/deploy-ace-step.yaml new file mode 100644 index 0000000..d6dd410 --- /dev/null +++ b/playbooks/deploy-ace-step.yaml @@ -0,0 +1,83 @@ +# Deploy ACE-Step 1.5 (Apache 2.0 music generation foundation model) +# to irv-ml1. +# +# Builds the image locally from ace-step/ACE-Step via docker buildx +# git URL context. ~10-15 min cold build (CUDA 12.6 runtime + torch + +# transformers + spacy + audio deps). First /generate triggers the +# model download (~5-10 GB) into the bind-mounted HF cache + warmup. +# +# Pre-req (user runs once): +# ssh -t irv-ml1 'sudo mkdir -p /worktank/ace-step/{checkpoints,outputs,logs,hf_cache} \ +# /opt/docker/compose/ace-step && \ +# sudo chown -R lkraven:lkraven /worktank/ace-step /opt/docker/compose/ace-step' +# +# Usage: +# scripts/elway irv-ml1 --playbook playbooks/deploy-ace-step.yaml +# +# Idempotent — every step is creates-/when-gated; rerun is safe. + +vars: + compose_dir: /opt/docker/compose/ace-step + worktank_root: /worktank/ace-step + host_port: "8210" + +steps: + # ── sanity checks (dirs were created by user-side sudo prep) ──────── + + - name: Verify /worktank/ace-step exists and is writable + shell: test -w {{ worktank_root }} + changed_when: "false" + + - name: Verify compose dir exists and is writable + shell: test -w {{ compose_dir }} + changed_when: "false" + + # ── deploy compose + env ──────────────────────────────────────────── + + - name: Upload compose.yaml + upload: + src: stacks/ace-step/compose.yaml + dest: "{{ compose_dir }}/compose.yaml" + mode: "0644" + + - name: Upload Dockerfile (patched for cu126 torch resolution) + upload: + src: stacks/ace-step/Dockerfile + dest: "{{ compose_dir }}/Dockerfile" + mode: "0644" + + - name: Seed .env from template (only if absent) + upload: + src: stacks/ace-step/.env.example + dest: "{{ compose_dir }}/.env" + mode: "0644" + when: "[ ! -f {{ compose_dir }}/.env ]" + + # ── build + bring up ──────────────────────────────────────────────── + + - name: docker compose build (~10-15 min first time; cached after) + shell: | + set -o pipefail + cd {{ compose_dir }} && docker compose build 2>&1 \ + | grep -vE '^#[0-9]+ |^ => |^=> |Collecting|Downloading|Requirement|Using cached|Installing collected|Successfully (installed|built)|━' + + - name: docker compose up -d + shell: cd {{ compose_dir }} && docker compose up -d + + - name: Wait for /health to respond (allow ~10 min for first model download + warmup) + shell: | + for i in $(seq 1 120); do + curl -sf -o /dev/null --max-time 3 http://localhost:{{ host_port }}/health && exit 0 + sleep 5 + done + exit 1 + changed_when: "false" + +verify: + - name: /health returns 200 + shell: curl -sf -o /dev/null http://localhost:{{ host_port }}/health + changed_when: "false" + + - name: Container is running + shell: docker inspect ace-step --format '{{.State.Status}}' | grep -q running + changed_when: "false" diff --git a/playbooks/deploy-stable-audio-open.yaml b/playbooks/deploy-stable-audio-open.yaml new file mode 100644 index 0000000..d55315f --- /dev/null +++ b/playbooks/deploy-stable-audio-open.yaml @@ -0,0 +1,126 @@ +# Deploy Stable Audio Open 1.0 (Stability AI diffusion SFX generator) +# to irv-ml1. +# +# Builds a small custom image from server.py + Dockerfile (no upstream +# Docker exists). ~5-10 min cold build (pytorch base + diffusers stack). +# First start pulls the model (~6 GB) from HF into the bind-mounted +# cache, then loads to VRAM (~30-60 s). +# +# Pre-req (user runs once): +# 1. Visit https://huggingface.co/stabilityai/stable-audio-open-1.0 +# and accept the Stability AI Community License (one click). +# 2. Generate a read token at https://huggingface.co/settings/tokens. +# 3. Put it in /opt/docker/compose/stable-audio-open/.env as +# SAO_HF_TOKEN=hf_xxx (the playbook seeds .env from .env.example +# with this field blank; the model gate fails closed without it). +# 4. ssh -t irv-ml1 'sudo mkdir -p \ +# /worktank/stable-audio-open/{hf_cache,outputs} \ +# /opt/docker/compose/stable-audio-open && \ +# sudo chown -R lkraven:lkraven \ +# /worktank/stable-audio-open /opt/docker/compose/stable-audio-open' +# +# Usage: +# scripts/elway irv-ml1 --playbook playbooks/deploy-stable-audio-open.yaml +# +# Idempotent — every step is creates-/when-gated; rerun is safe. + +vars: + compose_dir: /opt/docker/compose/stable-audio-open + worktank_root: /worktank/stable-audio-open + host_port: "8211" + +steps: + # ── sanity checks (dirs were created by user-side sudo prep) ──────── + + - name: Verify /worktank/stable-audio-open exists and is writable + shell: test -w {{ worktank_root }} + changed_when: "false" + + - name: Verify compose dir exists and is writable + shell: test -w {{ compose_dir }} + changed_when: "false" + + # ── upload build context (compose + Dockerfile + server.py) ───────── + + - name: Upload compose.yaml + upload: + src: stacks/stable-audio-open/compose.yaml + dest: "{{ compose_dir }}/compose.yaml" + mode: "0644" + + - name: Upload Dockerfile + upload: + src: stacks/stable-audio-open/Dockerfile + dest: "{{ compose_dir }}/Dockerfile" + mode: "0644" + + - name: Upload server.py + upload: + src: stacks/stable-audio-open/server.py + dest: "{{ compose_dir }}/server.py" + mode: "0644" + + - name: Seed .env from template (only if absent — REMEMBER TO SET SAO_HF_TOKEN) + upload: + src: stacks/stable-audio-open/.env.example + dest: "{{ compose_dir }}/.env" + mode: "0644" + when: "[ ! -f {{ compose_dir }}/.env ]" + + # Fail loud + early if the HF token is still empty — the model is + # gated and the container will crashloop on a 401 if we let it boot + # without one. Better to bail here than to wait for the healthcheck + # deadline to expire. + - name: Verify SAO_HF_TOKEN is set (model is gated, 401s without it) + shell: | + set -e + grep -q '^SAO_HF_TOKEN=hf_' {{ compose_dir }}/.env || { + echo "ERROR: SAO_HF_TOKEN is empty or invalid in {{ compose_dir }}/.env" >&2 + echo " 1. Accept license at https://huggingface.co/stabilityai/stable-audio-open-1.0" >&2 + echo " 2. Generate token at https://huggingface.co/settings/tokens" >&2 + echo " 3. Put hf_xxx token into {{ compose_dir }}/.env" >&2 + exit 1 + } + changed_when: "false" + + # ── build + bring up ──────────────────────────────────────────────── + + - name: docker compose build (~5-10 min first time; cached after) + shell: | + set -o pipefail + cd {{ compose_dir }} && docker compose build 2>&1 \ + | grep -vE '^#[0-9]+ |^ => |^=> |Collecting|Downloading|Requirement|Using cached|Installing collected|Successfully (installed|built)|━' + + - name: docker compose up -d + shell: cd {{ compose_dir }} && docker compose up -d + + - name: Wait for /health to respond (allow ~10 min for first model download + load) + shell: | + for i in $(seq 1 120); do + curl -sf -o /dev/null --max-time 3 http://localhost:{{ host_port }}/health && exit 0 + sleep 5 + done + exit 1 + changed_when: "false" + +verify: + - name: /health returns 200 and reports model loaded + shell: | + out=$(curl -sf --max-time 5 http://localhost:{{ host_port }}/health) + echo "$out" | grep -q '"loaded":true' || { echo "model not loaded: $out" >&2; exit 1; } + changed_when: "false" + + - name: /v1/audio/sfx returns a real WAV (cheap 1s clip, 10 steps) + shell: | + out=$(mktemp --suffix=.wav) + curl -sf -X POST http://localhost:{{ host_port }}/v1/audio/sfx \ + -H 'Content-Type: application/json' \ + -d '{"prompt":"a single soft bell chime","duration":1,"steps":10}' \ + -o "$out" --max-time 60 + file -b "$out" | grep -q '^RIFF.*WAVE' + rm -f "$out" + changed_when: "false" + + - name: Container is running + shell: docker inspect stable-audio-open --format '{{.State.Status}}' | grep -q running + changed_when: "false" diff --git a/stacks/ace-step/.env.example b/stacks/ace-step/.env.example new file mode 100644 index 0000000..e838291 --- /dev/null +++ b/stacks/ace-step/.env.example @@ -0,0 +1,47 @@ +# ACE-Step 1.5 stack tunables. Copy to `.env` on irv-ml1 before +# deploying. + +# ── build pin ──────────────────────────────────────────────────────── +# SHA of ace-step/ACE-Step to build from. Use the FULL 40-char SHA — +# docker buildx git source resolver rejects short hashes. `main` works +# at first deploy; pin to a real SHA before any production cutover so +# upstream commits don't surprise you on next rebuild. +ACE_STEP_SHA=main + +# Local image tag — bump when you change build context to force a +# fresh layer build. +ACE_STEP_TAG=v1 + +# ── network ────────────────────────────────────────────────────────── +# Host port (container listens on 8000 internally — infer-api.py +# hardcodes uvicorn.run(host=0.0.0.0, port=8000)). +# Reservations on irv-ml1: 8188 ComfyUI, 8190 CosyVoice, 8191 Qwen3-TTS, +# 8192 IndexTTS-2, 8193 Kokoro, 8194 VibeVoice, 8195 Fish, 8196 +# Chatterbox, 8197 Voxtral, 8765 Parakeet ASR. 8210 starts the +# audio-generation block (music + SFX) so future TTS adds can keep +# going from 8198+. +ACE_STEP_PORT=8210 +ACE_STEP_BIND=0.0.0.0 + +# ── runtime / GPU ──────────────────────────────────────────────────── +# GPU pinning. "0" = RTX 3090 (24 GB), "1" = RTX A6000 (48 GB). +# A6000 (1) recommended — Fish s2-pro lives there at ~17 GB, and +# ACE-Step adds ~10-12 GB during inference, leaving comfortable +# headroom on the 48 GB card. The 3090 is full with the TTS slate. +ACE_STEP_GPU_DEVICES=1 + +# ── persistent storage on the host ─────────────────────────────────── +# Model checkpoints — primary spot for any manually-staged checkpoints. +# ACE-Step's auto-download lands in HF_HOME (cache dir below). +ACE_STEP_CHECKPOINTS_DIR=/worktank/ace-step/checkpoints + +# Generated audio output — clients can pull from here via the +# returned file path in the /generate response. +ACE_STEP_OUTPUTS_DIR=/worktank/ace-step/outputs + +# Application logs. +ACE_STEP_LOGS_DIR=/worktank/ace-step/logs + +# HF cache — first start pulls the ACE-Step checkpoint (~5-10 GB) +# into this dir. Persistent across container recreates. +ACE_STEP_CACHE_DIR=/worktank/ace-step/hf_cache diff --git a/stacks/ace-step/Dockerfile b/stacks/ace-step/Dockerfile new file mode 100644 index 0000000..250ffc0 --- /dev/null +++ b/stacks/ace-step/Dockerfile @@ -0,0 +1,66 @@ +# Custom Dockerfile for ACE-Step 1.5. +# +# Mirrors upstream's Dockerfile structure, but fixes a CUDA-version +# mismatch that crashloops the upstream image as of April 2026: +# * upstream's requirements.txt lists `torch torchvision torchaudio` +# with no version pins; +# * upstream's pip install uses `--extra-index-url cu126`, which is +# a FALLBACK only — pypi default wins for resolution; +# * pypi-default torch is now cu13, so torch installs cu13 + the +# cu126 fallback only kicks in for torchvision/torchaudio → +# `RuntimeError: Detected that PyTorch and torchvision were compiled +# with different CUDA major versions`. +# +# Fix: install torch/torchvision/torchaudio FIRST from the cu126 index +# (forced via --index-url, not --extra-index-url). Then `pip install +# -r requirements.txt` sees they're already satisfied and leaves them +# alone. +# +# Also: command is `python3 infer-api.py` (REST), not `gui.py` (Gradio) +# — see compose.yaml command override; CMD here is the same default +# so the image works standalone too. +FROM nvidia/cuda:12.6.0-runtime-ubuntu22.04 AS base + +ENV PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 \ + HF_HUB_ENABLE_HF_TRANSFER=1 \ + DEBIAN_FRONTEND=noninteractive + +RUN apt-get update && apt-get install -y --no-install-recommends \ + python3.10 \ + python3-pip \ + python3-venv \ + python3-dev \ + build-essential \ + git \ + curl \ + ca-certificates \ + && apt-get clean \ + && rm -rf /var/lib/apt/lists/* \ + && ln -sf /usr/bin/python3 /usr/bin/python + +RUN python -m venv /opt/venv +ENV PATH="/opt/venv/bin:$PATH" + +WORKDIR /app + +# Clone upstream. Bake the SHA into a layer-cache key so a different +# SHA invalidates everything below. +ARG ACE_STEP_REF=main +RUN git clone https://github.com/ace-step/ACE-Step.git . \ + && git checkout ${ACE_STEP_REF} \ + && echo "ace-step ref: $(git rev-parse HEAD)" + +# Pre-install torch/torchvision/torchaudio from the cu126 index — this +# satisfies the unpinned entries in requirements.txt so the next pip +# install doesn't re-resolve them from pypi default (cu13). +RUN pip install --no-cache-dir --upgrade pip \ + && pip install --no-cache-dir \ + torch torchvision torchaudio \ + --index-url https://download.pytorch.org/whl/cu126 \ + && pip install --no-cache-dir hf_transfer peft \ + && pip install --no-cache-dir -r requirements.txt \ + && pip install --no-cache-dir . + +EXPOSE 8000 +CMD ["python3", "infer-api.py"] diff --git a/stacks/ace-step/README.md b/stacks/ace-step/README.md new file mode 100644 index 0000000..1770b68 --- /dev/null +++ b/stacks/ace-step/README.md @@ -0,0 +1,53 @@ +# ace-step + +ACE-Step 1.5 — Apache 2.0 open-source music generation foundation +model. Hybrid diffusion + LLM. Generates lyric-aware multi-minute +songs (vocals + instrumentation). + +| | | +|---|---| +| host | `irv-ml1` | +| port | `8210` | +| GPU | A6000 (`device_ids: ["1"]`) | +| VRAM | ~10-12 GB during inference | +| upstream | https://github.com/ace-step/ACE-Step | +| license | Apache 2.0 | + +## API surface + +`infer-api.py` (FastAPI) exposes: + +- `GET /health` — liveness, returns 200 once the process is up + (model is lazy-loaded on first /generate). +- `POST /generate` — body: `ACEStepInput` Pydantic model with + ~27 params (prompt, lyrics, audio_duration, guidance_scale, etc.). + Returns `{status, output_path, message}`. + +The container does NOT expose the Gradio UI — we override the upstream +default `python3 acestep/gui.py` with `python3 infer-api.py`. If you +want the Gradio UI for ad-hoc experimentation, run a one-off: + +```bash +ssh irv-ml1 'docker exec -it ace-step python3 acestep/gui.py --server_name 0.0.0.0 --port 7865' +``` + +…and port-forward 7865 to your laptop. + +## Deploy + +```bash +scripts/elway irv-ml1 --playbook playbooks/deploy-ace-step.yaml +``` + +Idempotent. Cold build is ~10-15 min (CUDA + torch + transformers + +spacy + audio deps). First `/generate` triggers the model download +(~5-10 GB) and warmup (~30-60 s). + +## Tunables + +See `.env.example` — copy to `.env` on the host (lives at +`/opt/docker/compose/ace-step/.env`, gitignored). Common knobs: + +- `ACE_STEP_SHA` — pin upstream commit +- `ACE_STEP_GPU_DEVICES` — GPU index +- `ACE_STEP_*_DIR` — bind-mount paths under `/worktank/ace-step/` diff --git a/stacks/ace-step/compose.yaml b/stacks/ace-step/compose.yaml new file mode 100644 index 0000000..8de4475 --- /dev/null +++ b/stacks/ace-step/compose.yaml @@ -0,0 +1,70 @@ +# ACE-Step 1.5 — open-source music generation foundation model +# (April 2026). Hybrid diffusion + LLM architecture, Apache 2.0. +# ~50-80 s for a 4-minute song on A6000; under 4 GB VRAM at idle, +# ~10-12 GB during inference. Beats YuE / DiffRhythm on the +# speed/coherence trade. +# +# We launch upstream's REST API (`infer-api.py`) instead of the +# default `gui.py` (Gradio). The REST surface is what we'll point +# clients + automation at; Gradio is dev-time eye candy. +# +# Image is built locally from upstream's repo via docker buildx git +# context, same pattern as fish-s2. +# +# All tunables live in .env — edit that, not this file. + +services: + ace-step: + image: local/ace-step:${ACE_STEP_TAG} + build: + # Build from local Dockerfile (not upstream's git context) — we + # ship a patched Dockerfile that fixes upstream's torch/cu126 + # resolution bug. Playbook uploads Dockerfile alongside this + # compose.yaml. + context: . + dockerfile: Dockerfile + args: + ACE_STEP_REF: ${ACE_STEP_SHA} + container_name: ace-step + restart: unless-stopped + runtime: nvidia + ports: + # Container default for infer-api.py is 8000 (hardcoded + # uvicorn.run(host=0.0.0.0, port=8000) — no flags). Map host + # ACE_STEP_PORT to it. + - "${ACE_STEP_BIND:-0.0.0.0}:${ACE_STEP_PORT}:8000" + environment: + - NVIDIA_VISIBLE_DEVICES=${ACE_STEP_GPU_DEVICES:-1} + # ACE_OUTPUT_DIR is read by acestep at generation time — keep + # in sync with the bind mount below. + - ACE_OUTPUT_DIR=/app/outputs + # HF_HOME points the HuggingFace cache at the bind mount so the + # ~5-10 GB checkpoint download survives container recreates. + - HF_HOME=/app/hf_cache + volumes: + - ${ACE_STEP_CHECKPOINTS_DIR}:/app/checkpoints + - ${ACE_STEP_OUTPUTS_DIR}:/app/outputs + - ${ACE_STEP_LOGS_DIR}:/app/logs + - ${ACE_STEP_CACHE_DIR}:/app/hf_cache + # Override upstream's default `python3 acestep/gui.py` with the + # REST API entry point. infer-api.py self-binds 0.0.0.0:8000 and + # exposes POST /generate + GET /health. + command: ["python3", "infer-api.py"] + healthcheck: + # /health is the cheapest signal infer-api.py exposes — returns + # 200 as soon as the FastAPI app is up. The pipeline lazy-loads + # on first /generate, so /health says "process alive" not + # "model warm". Good enough for a liveness signal; first + # /generate has the ~30-60 s warmup baked in. + test: ["CMD-SHELL", "python3 -c \"import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5).status==200 else 1)\""] + interval: 30s + timeout: 10s + retries: 3 + # First boot pulls ACE-Step checkpoint (~5-10 GB) into HF cache. + start_period: 600s + labels: + - homepage.group=AI Systems + - homepage.name=ACE-Step + - homepage.icon=mdi-music-note-eighth + - homepage.description=Open-source music generation — 4-min song in ~60s, lyrics + style prompts (irv-ml1) + - homepage.href=http://10.100.79.3:${ACE_STEP_PORT} diff --git a/stacks/stable-audio-open/.env.example b/stacks/stable-audio-open/.env.example new file mode 100644 index 0000000..5052fe5 --- /dev/null +++ b/stacks/stable-audio-open/.env.example @@ -0,0 +1,48 @@ +# Stable Audio Open 1.0 stack tunables. Copy to `.env` on irv-ml1 +# before deploying. + +# ── image ──────────────────────────────────────────────────────────── +# Local image tag — bump when you change Dockerfile or server.py to +# force a fresh build. +SAO_TAG=v1 + +# Which Stable Audio model to load. As of 2026-04 the only released +# checkpoint is 1.0; future revisions can swap here without touching +# compose.yaml or server.py. +SAO_MODEL=stabilityai/stable-audio-open-1.0 + +# ── network ────────────────────────────────────────────────────────── +# Host port (container listens on 8000 internally). +# Reservations on irv-ml1: 8188 ComfyUI, 8190 CosyVoice, 8191 Qwen3-TTS, +# 8192 IndexTTS-2, 8193 Kokoro, 8194 VibeVoice, 8195 Fish, 8196 +# Chatterbox, 8197 Voxtral, 8210 ACE-Step (music), 8765 Parakeet ASR. +SAO_PORT=8211 +SAO_BIND=0.0.0.0 + +# ── runtime / GPU ──────────────────────────────────────────────────── +# GPU pinning. "0" = RTX 3090 (24 GB), "1" = RTX A6000 (48 GB). +# A6000 (1) recommended — Fish s2-pro lives there at ~17 GB; SAO adds +# ~6 GB practical (model fp16 + small VAE working set), and ACE-Step +# adds another ~12 GB during inference. Total ~35 GB / 48 GB still +# leaves headroom. The 3090 is full with the TTS slate. +SAO_GPU_DEVICES=1 + +# ── HuggingFace auth ───────────────────────────────────────────────── +# HF token — REQUIRED. Stable Audio Open is gated; you must: +# 1. Visit https://huggingface.co/stabilityai/stable-audio-open-1.0 +# and accept the Stability AI Community License (one click). +# 2. Generate a read token at +# https://huggingface.co/settings/tokens. +# 3. Paste it here. +# Without this, the first model download 401s and the container +# crashloops. +SAO_HF_TOKEN= + +# ── persistent storage on the host ─────────────────────────────────── +# HF cache — first start pulls the model (~6 GB) into this dir. +# Persistent across container recreates so we don't re-pull. +SAO_CACHE_DIR=/worktank/stable-audio-open/hf_cache + +# Generated audio output — clients can pull from here for any flow +# that wants a file path instead of a streamed WAV body. +SAO_OUTPUTS_DIR=/worktank/stable-audio-open/outputs diff --git a/stacks/stable-audio-open/Dockerfile b/stacks/stable-audio-open/Dockerfile new file mode 100644 index 0000000..22ae6e8 --- /dev/null +++ b/stacks/stable-audio-open/Dockerfile @@ -0,0 +1,43 @@ +# Stable Audio Open 1.0 inference image. +# pytorch/pytorch base ships torch + cuda + cudnn already linked, so +# we only layer the diffusers stack + a libsndfile for soundfile + the +# fastapi shim. Smaller and faster to build than starting from +# nvidia/cuda and pip-installing torch ourselves. +FROM pytorch/pytorch:2.5.1-cuda12.4-cudnn9-runtime AS base + +ENV PYTHONUNBUFFERED=1 \ + PYTHONDONTWRITEBYTECODE=1 \ + PIP_NO_CACHE_DIR=1 \ + PIP_DISABLE_PIP_VERSION_CHECK=1 \ + HF_HOME=/app/hf_cache + +# libsndfile1 is the C lib soundfile binds to. Without it the pip +# install of soundfile succeeds but `import soundfile` fails at +# runtime with OSError: cannot find libsndfile. +RUN apt-get update && apt-get install -y --no-install-recommends \ + libsndfile1 \ + && rm -rf /var/lib/apt/lists/* + +# protobuf + sentencepiece are pulled in by the T5 text encoder +# (Stable Audio Open uses google/t5-base-cb under the hood). +# accelerate gates the .to(device) fast path for diffusers. +# torchsde is required by CosineDPMSolverMultistepScheduler — diffusers +# doesn't pull it as a hard dep; without it, pipeline init fails with +# "CosineDPMSolverMultistepScheduler requires the torchsde library". +RUN pip install \ + "diffusers>=0.27.0" \ + "transformers>=4.40.0" \ + accelerate \ + protobuf \ + sentencepiece \ + soundfile \ + torchsde \ + fastapi \ + "uvicorn[standard]" \ + pydantic + +WORKDIR /app +COPY server.py /app/server.py + +EXPOSE 8000 +CMD ["uvicorn", "server:app", "--host", "0.0.0.0", "--port", "8000"] diff --git a/stacks/stable-audio-open/README.md b/stacks/stable-audio-open/README.md new file mode 100644 index 0000000..5c963a5 --- /dev/null +++ b/stacks/stable-audio-open/README.md @@ -0,0 +1,55 @@ +# stable-audio-open + +Stability AI's Stable Audio Open 1.0 — text-to-audio latent diffusion. +Strong on SFX, foley, ambience, short loops. Not a music model — it +does not generate intelligible vocals or structured songs (use +`ace-step` for that). + +| | | +|---|---| +| host | `irv-ml1` | +| port | `8211` | +| GPU | A6000 (`device_ids: ["1"]`) | +| VRAM | ~6 GB in fp16 | +| max clip | 47 s at 44.1 kHz | +| upstream | https://github.com/Stability-AI/stable-audio-tools | +| model | `stabilityai/stable-audio-open-1.0` (gated) | +| license | Stability AI Community (non-commercial / personal / research) | + +## API surface + +`server.py` (custom FastAPI shim) exposes: + +- `GET /health` — returns 200 once the model is loaded. +- `POST /v1/audio/sfx` — returns a `audio/wav` blob. + +```jsonc +{ + "prompt": "a vintage typewriter clacking in a quiet room", + "negative_prompt": "Low quality.", // optional, default "Low quality." + "duration": 10.0, // seconds, 0.5 – 47 + "steps": 100, // 10 – 300, more = better quality + "seed": 42, // optional + "cfg_scale": 7.0 // 0 – 20 +} +``` + +Why a custom shim: there's no upstream Docker image and no upstream +HTTP server for Stable Audio Open. Diffusers exposes +`StableAudioPipeline` cleanly — the shim is ~70 lines. + +## Deploy + +```bash +scripts/elway irv-ml1 --playbook playbooks/deploy-stable-audio-open.yaml +``` + +Pre-deploy: visit https://huggingface.co/stabilityai/stable-audio-open-1.0 +once and accept the Community License (HF token alone is not enough — +the gate is per-model). Then put the token in `SAO_HF_TOKEN` in `.env` +on the host. + +## Tunables + +See `.env.example` — copy to `.env` on the host (lives at +`/opt/docker/compose/stable-audio-open/.env`, gitignored). diff --git a/stacks/stable-audio-open/compose.yaml b/stacks/stable-audio-open/compose.yaml new file mode 100644 index 0000000..c64179f --- /dev/null +++ b/stacks/stable-audio-open/compose.yaml @@ -0,0 +1,61 @@ +# Stable Audio Open 1.0 — Stability AI's open-weight latent-diffusion +# SFX/ambience generator. 1.21B params, ~4-6 GB VRAM in fp16, up to +# 47 s clips at 44.1 kHz. Strong on text-aligned sound effects, foley, +# field-recording-style ambience. NOT a music model — it does not +# generate intelligible vocals or structured songs (use ACE-Step for +# that). +# +# LICENSE: Stability AI Community License. Personal / research use is +# free; commercial use requires a separate license from Stability +# (https://stability.ai/license). Same posture we already accepted +# for Voxtral. +# +# No upstream Docker image — we ship a custom Dockerfile + a small +# FastAPI shim (server.py) that wraps diffusers' StableAudioPipeline +# and exposes POST /v1/audio/sfx. +# +# All tunables live in .env — edit that, not this file. + +services: + stable-audio-open: + image: local/stable-audio-open:${SAO_TAG} + build: + # Build context is the compose dir on the host — the playbook + # uploads server.py + Dockerfile alongside this compose.yaml. + context: . + dockerfile: Dockerfile + container_name: stable-audio-open + restart: unless-stopped + runtime: nvidia + ports: + - "${SAO_BIND:-0.0.0.0}:${SAO_PORT}:8000" + environment: + - NVIDIA_VISIBLE_DEVICES=${SAO_GPU_DEVICES:-1} + - SAO_MODEL=${SAO_MODEL:-stabilityai/stable-audio-open-1.0} + - HF_HOME=/app/hf_cache + # Model is gated on HuggingFace (you must accept the Stability + # Community License once on the model page before the token can + # download it). Set SAO_HF_TOKEN in .env. Without this, the + # first model download 401s and the container crashloops. + - HF_TOKEN=${SAO_HF_TOKEN} + volumes: + - ${SAO_CACHE_DIR}:/app/hf_cache + - ${SAO_OUTPUTS_DIR}:/app/outputs + healthcheck: + # /health is set by server.py — returns 200 once FastAPI is up + # AND the pipeline finished loading (lifespan blocks startup + # until the model is in VRAM). + test: ["CMD-SHELL", "python -c \"import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5).status==200 else 1)\""] + interval: 30s + timeout: 10s + retries: 3 + # First boot pulls the model (~6 GB) into HF cache + loads to + # VRAM. Cold start ~3-5 min on a fast pipe; subsequent starts + # are ~30 s. + start_period: 600s + labels: + - homepage.group=AI Systems + - homepage.name=Stable Audio Open + - homepage.icon=mdi-waveform + - homepage.description=Diffusion SFX/ambience generator — up to 47s at 44.1 kHz (irv-ml1) + - homepage.href=http://10.100.79.3:${SAO_PORT} diff --git a/stacks/stable-audio-open/server.py b/stacks/stable-audio-open/server.py new file mode 100644 index 0000000..3c24955 --- /dev/null +++ b/stacks/stable-audio-open/server.py @@ -0,0 +1,80 @@ +# FastAPI shim around diffusers' StableAudioPipeline. +# Single endpoint POST /v1/audio/sfx returns a WAV blob. +# Model is loaded once on startup and held in process memory. +import io +import os +import time +from contextlib import asynccontextmanager +from typing import Optional + +import soundfile as sf +import torch +from diffusers import StableAudioPipeline +from fastapi import FastAPI, HTTPException, Response +from pydantic import BaseModel, Field + +MODEL_ID = os.environ.get("SAO_MODEL", "stabilityai/stable-audio-open-1.0") +DEVICE = "cuda" if torch.cuda.is_available() else "cpu" +DTYPE = torch.float16 if DEVICE == "cuda" else torch.float32 + +state: dict = {} + + +@asynccontextmanager +async def lifespan(app: FastAPI): + print(f"[sao] loading {MODEL_ID} on {DEVICE} ({DTYPE})", flush=True) + t0 = time.time() + pipe = StableAudioPipeline.from_pretrained(MODEL_ID, torch_dtype=DTYPE) + pipe = pipe.to(DEVICE) + state["pipe"] = pipe + print(f"[sao] loaded in {time.time() - t0:.1f}s", flush=True) + yield + state.clear() + + +app = FastAPI(lifespan=lifespan) + + +class SfxRequest(BaseModel): + prompt: str = Field(..., min_length=1) + negative_prompt: Optional[str] = "Low quality." + duration: float = Field(10.0, gt=0.5, le=47.0) + steps: int = Field(100, ge=10, le=300) + seed: Optional[int] = None + cfg_scale: float = Field(7.0, gt=0.0, le=20.0) + + +@app.get("/health") +def health(): + return { + "status": "ok", + "model": MODEL_ID, + "device": DEVICE, + "loaded": "pipe" in state, + } + + +@app.post("/v1/audio/sfx") +def sfx(req: SfxRequest): + pipe = state.get("pipe") + if pipe is None: + raise HTTPException(503, "model not loaded yet") + + generator = None + if req.seed is not None: + generator = torch.Generator(DEVICE).manual_seed(req.seed) + + audio = pipe( + req.prompt, + negative_prompt=req.negative_prompt, + num_inference_steps=req.steps, + audio_end_in_s=req.duration, + num_waveforms_per_prompt=1, + generator=generator, + ).audios + + waveform = audio[0].T.float().cpu().numpy() + buf = io.BytesIO() + sf.write(buf, waveform, pipe.vae.sampling_rate, format="WAV") + buf.seek(0) + return Response(content=buf.read(), media_type="audio/wav")