ace-step + stable-audio-open: deploy music + SFX generation to irv-ml1
Two new audio-generation stacks alongside the TTS slate: ace-step :8210 — Apache 2.0 music generation foundation model (hybrid diffusion + LLM). Lyric-aware multi-minute songs. ~10-12 GB VRAM during inference, A6000-pinned. Custom Dockerfile patches upstream's torch/cu126 resolution bug (--extra-index-url cu126 was falling back to pypi-default cu13 wheels, mismatching torchvision). stable-audio-open :8211 — Stability AI 1.21B latent-diffusion SFX + ambience. Up to 47s clips at 44.1 kHz. ~6 GB VRAM in fp16, A6000-pinned. Custom FastAPI shim around diffusers' StableAudioPipeline (no upstream HTTP server). Dockerfile pins torchsde explicitly — diffusers doesn't pull it as a hard dep but CosineDPMSolverMultistepScheduler needs it.
This commit is contained in:
@@ -0,0 +1,83 @@
|
||||
# Deploy ACE-Step 1.5 (Apache 2.0 music generation foundation model)
|
||||
# to irv-ml1.
|
||||
#
|
||||
# Builds the image locally from ace-step/ACE-Step via docker buildx
|
||||
# git URL context. ~10-15 min cold build (CUDA 12.6 runtime + torch +
|
||||
# transformers + spacy + audio deps). First /generate triggers the
|
||||
# model download (~5-10 GB) into the bind-mounted HF cache + warmup.
|
||||
#
|
||||
# Pre-req (user runs once):
|
||||
# ssh -t irv-ml1 'sudo mkdir -p /worktank/ace-step/{checkpoints,outputs,logs,hf_cache} \
|
||||
# /opt/docker/compose/ace-step && \
|
||||
# sudo chown -R lkraven:lkraven /worktank/ace-step /opt/docker/compose/ace-step'
|
||||
#
|
||||
# Usage:
|
||||
# scripts/elway irv-ml1 --playbook playbooks/deploy-ace-step.yaml
|
||||
#
|
||||
# Idempotent — every step is creates-/when-gated; rerun is safe.
|
||||
|
||||
vars:
|
||||
compose_dir: /opt/docker/compose/ace-step
|
||||
worktank_root: /worktank/ace-step
|
||||
host_port: "8210"
|
||||
|
||||
steps:
|
||||
# ── sanity checks (dirs were created by user-side sudo prep) ────────
|
||||
|
||||
- name: Verify /worktank/ace-step exists and is writable
|
||||
shell: test -w {{ worktank_root }}
|
||||
changed_when: "false"
|
||||
|
||||
- name: Verify compose dir exists and is writable
|
||||
shell: test -w {{ compose_dir }}
|
||||
changed_when: "false"
|
||||
|
||||
# ── deploy compose + env ────────────────────────────────────────────
|
||||
|
||||
- name: Upload compose.yaml
|
||||
upload:
|
||||
src: stacks/ace-step/compose.yaml
|
||||
dest: "{{ compose_dir }}/compose.yaml"
|
||||
mode: "0644"
|
||||
|
||||
- name: Upload Dockerfile (patched for cu126 torch resolution)
|
||||
upload:
|
||||
src: stacks/ace-step/Dockerfile
|
||||
dest: "{{ compose_dir }}/Dockerfile"
|
||||
mode: "0644"
|
||||
|
||||
- name: Seed .env from template (only if absent)
|
||||
upload:
|
||||
src: stacks/ace-step/.env.example
|
||||
dest: "{{ compose_dir }}/.env"
|
||||
mode: "0644"
|
||||
when: "[ ! -f {{ compose_dir }}/.env ]"
|
||||
|
||||
# ── build + bring up ────────────────────────────────────────────────
|
||||
|
||||
- name: docker compose build (~10-15 min first time; cached after)
|
||||
shell: |
|
||||
set -o pipefail
|
||||
cd {{ compose_dir }} && docker compose build 2>&1 \
|
||||
| grep -vE '^#[0-9]+ |^ => |^=> |Collecting|Downloading|Requirement|Using cached|Installing collected|Successfully (installed|built)|━'
|
||||
|
||||
- name: docker compose up -d
|
||||
shell: cd {{ compose_dir }} && docker compose up -d
|
||||
|
||||
- name: Wait for /health to respond (allow ~10 min for first model download + warmup)
|
||||
shell: |
|
||||
for i in $(seq 1 120); do
|
||||
curl -sf -o /dev/null --max-time 3 http://localhost:{{ host_port }}/health && exit 0
|
||||
sleep 5
|
||||
done
|
||||
exit 1
|
||||
changed_when: "false"
|
||||
|
||||
verify:
|
||||
- name: /health returns 200
|
||||
shell: curl -sf -o /dev/null http://localhost:{{ host_port }}/health
|
||||
changed_when: "false"
|
||||
|
||||
- name: Container is running
|
||||
shell: docker inspect ace-step --format '{{.State.Status}}' | grep -q running
|
||||
changed_when: "false"
|
||||
@@ -0,0 +1,126 @@
|
||||
# Deploy Stable Audio Open 1.0 (Stability AI diffusion SFX generator)
|
||||
# to irv-ml1.
|
||||
#
|
||||
# Builds a small custom image from server.py + Dockerfile (no upstream
|
||||
# Docker exists). ~5-10 min cold build (pytorch base + diffusers stack).
|
||||
# First start pulls the model (~6 GB) from HF into the bind-mounted
|
||||
# cache, then loads to VRAM (~30-60 s).
|
||||
#
|
||||
# Pre-req (user runs once):
|
||||
# 1. Visit https://huggingface.co/stabilityai/stable-audio-open-1.0
|
||||
# and accept the Stability AI Community License (one click).
|
||||
# 2. Generate a read token at https://huggingface.co/settings/tokens.
|
||||
# 3. Put it in /opt/docker/compose/stable-audio-open/.env as
|
||||
# SAO_HF_TOKEN=hf_xxx (the playbook seeds .env from .env.example
|
||||
# with this field blank; the model gate fails closed without it).
|
||||
# 4. ssh -t irv-ml1 'sudo mkdir -p \
|
||||
# /worktank/stable-audio-open/{hf_cache,outputs} \
|
||||
# /opt/docker/compose/stable-audio-open && \
|
||||
# sudo chown -R lkraven:lkraven \
|
||||
# /worktank/stable-audio-open /opt/docker/compose/stable-audio-open'
|
||||
#
|
||||
# Usage:
|
||||
# scripts/elway irv-ml1 --playbook playbooks/deploy-stable-audio-open.yaml
|
||||
#
|
||||
# Idempotent — every step is creates-/when-gated; rerun is safe.
|
||||
|
||||
vars:
|
||||
compose_dir: /opt/docker/compose/stable-audio-open
|
||||
worktank_root: /worktank/stable-audio-open
|
||||
host_port: "8211"
|
||||
|
||||
steps:
|
||||
# ── sanity checks (dirs were created by user-side sudo prep) ────────
|
||||
|
||||
- name: Verify /worktank/stable-audio-open exists and is writable
|
||||
shell: test -w {{ worktank_root }}
|
||||
changed_when: "false"
|
||||
|
||||
- name: Verify compose dir exists and is writable
|
||||
shell: test -w {{ compose_dir }}
|
||||
changed_when: "false"
|
||||
|
||||
# ── upload build context (compose + Dockerfile + server.py) ─────────
|
||||
|
||||
- name: Upload compose.yaml
|
||||
upload:
|
||||
src: stacks/stable-audio-open/compose.yaml
|
||||
dest: "{{ compose_dir }}/compose.yaml"
|
||||
mode: "0644"
|
||||
|
||||
- name: Upload Dockerfile
|
||||
upload:
|
||||
src: stacks/stable-audio-open/Dockerfile
|
||||
dest: "{{ compose_dir }}/Dockerfile"
|
||||
mode: "0644"
|
||||
|
||||
- name: Upload server.py
|
||||
upload:
|
||||
src: stacks/stable-audio-open/server.py
|
||||
dest: "{{ compose_dir }}/server.py"
|
||||
mode: "0644"
|
||||
|
||||
- name: Seed .env from template (only if absent — REMEMBER TO SET SAO_HF_TOKEN)
|
||||
upload:
|
||||
src: stacks/stable-audio-open/.env.example
|
||||
dest: "{{ compose_dir }}/.env"
|
||||
mode: "0644"
|
||||
when: "[ ! -f {{ compose_dir }}/.env ]"
|
||||
|
||||
# Fail loud + early if the HF token is still empty — the model is
|
||||
# gated and the container will crashloop on a 401 if we let it boot
|
||||
# without one. Better to bail here than to wait for the healthcheck
|
||||
# deadline to expire.
|
||||
- name: Verify SAO_HF_TOKEN is set (model is gated, 401s without it)
|
||||
shell: |
|
||||
set -e
|
||||
grep -q '^SAO_HF_TOKEN=hf_' {{ compose_dir }}/.env || {
|
||||
echo "ERROR: SAO_HF_TOKEN is empty or invalid in {{ compose_dir }}/.env" >&2
|
||||
echo " 1. Accept license at https://huggingface.co/stabilityai/stable-audio-open-1.0" >&2
|
||||
echo " 2. Generate token at https://huggingface.co/settings/tokens" >&2
|
||||
echo " 3. Put hf_xxx token into {{ compose_dir }}/.env" >&2
|
||||
exit 1
|
||||
}
|
||||
changed_when: "false"
|
||||
|
||||
# ── build + bring up ────────────────────────────────────────────────
|
||||
|
||||
- name: docker compose build (~5-10 min first time; cached after)
|
||||
shell: |
|
||||
set -o pipefail
|
||||
cd {{ compose_dir }} && docker compose build 2>&1 \
|
||||
| grep -vE '^#[0-9]+ |^ => |^=> |Collecting|Downloading|Requirement|Using cached|Installing collected|Successfully (installed|built)|━'
|
||||
|
||||
- name: docker compose up -d
|
||||
shell: cd {{ compose_dir }} && docker compose up -d
|
||||
|
||||
- name: Wait for /health to respond (allow ~10 min for first model download + load)
|
||||
shell: |
|
||||
for i in $(seq 1 120); do
|
||||
curl -sf -o /dev/null --max-time 3 http://localhost:{{ host_port }}/health && exit 0
|
||||
sleep 5
|
||||
done
|
||||
exit 1
|
||||
changed_when: "false"
|
||||
|
||||
verify:
|
||||
- name: /health returns 200 and reports model loaded
|
||||
shell: |
|
||||
out=$(curl -sf --max-time 5 http://localhost:{{ host_port }}/health)
|
||||
echo "$out" | grep -q '"loaded":true' || { echo "model not loaded: $out" >&2; exit 1; }
|
||||
changed_when: "false"
|
||||
|
||||
- name: /v1/audio/sfx returns a real WAV (cheap 1s clip, 10 steps)
|
||||
shell: |
|
||||
out=$(mktemp --suffix=.wav)
|
||||
curl -sf -X POST http://localhost:{{ host_port }}/v1/audio/sfx \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"prompt":"a single soft bell chime","duration":1,"steps":10}' \
|
||||
-o "$out" --max-time 60
|
||||
file -b "$out" | grep -q '^RIFF.*WAVE'
|
||||
rm -f "$out"
|
||||
changed_when: "false"
|
||||
|
||||
- name: Container is running
|
||||
shell: docker inspect stable-audio-open --format '{{.State.Status}}' | grep -q running
|
||||
changed_when: "false"
|
||||
Reference in New Issue
Block a user