From f8ecc6c0475d8b5be0982861ba3095b84cc17cfd Mon Sep 17 00:00:00 2001 From: Vuong Hoang Date: Mon, 11 May 2026 15:48:33 -0700 Subject: [PATCH] =?UTF-8?q?ace-step:=20stream=20audio=20bytes=20inline;=20?= =?UTF-8?q?catalog=20v3=20=E2=86=92=20v4?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The pre-fix wrapper at stacks/ace-step/infer-api.py returned a JSON {output_path: "..."} reference to a file written inside the container at /app/outputs/. That path was unreachable from outside the container — every consumer got 134 bytes of JSON-pretending-to- be-WAV instead of audio. Surfaced by the asset_engine consumer's end-to-end smoke (althing thread 01KRCJF7NGMXYE9F62Q1A6KFD4 msg 5); my own earlier smoke missed it because I checked HTTP=200 and stopped reading instead of inspecting the response body. Wrapper now reads back the file the pipeline writes and streams the bytes via fastapi.responses.Response with media_type set from the audio_format request field (audio/wav | audio/mpeg | audio/flac). The in-container path is exposed via X-Output-Path header for log correlation but is no longer load-bearing. Verified end-to-end against live ace-step on irv-ml1: POST /generate -> HTTP 200 in 80s content-type: audio/wav content-length: 945226 x-output-path: /app/outputs/output_cfe87d1d....wav $ file response.wav RIFF (little-endian) data, WAVE audio, Microsoft PCM, 16 bit, stereo 48000 Hz Catalog: ace-step bumped version 3 -> 4. Dropped response.output_field (no longer applicable). reproducibility.notes expanded to record both the v2 18-arg-tuple fix and this v4 inline-streaming change so the history is auditable from the catalog itself. Stale ACEStepOutput Pydantic model left in infer-api.py for now — unused but small; future cleanup. --- docs/asset-engine/services.yaml | 13 +++++++--- stacks/ace-step/infer-api.py | 42 ++++++++++++++++++++++++++++----- 2 files changed, 46 insertions(+), 9 deletions(-) diff --git a/docs/asset-engine/services.yaml b/docs/asset-engine/services.yaml index 8635e0e..a494c98 100644 --- a/docs/asset-engine/services.yaml +++ b/docs/asset-engine/services.yaml @@ -752,7 +752,7 @@ services: Apache-2.0 hybrid diffusion+LLM music generation. Multi-minute lyric-aware songs with vocals + instrumentation. category: music - version: 3 + version: 4 host: irv-ml1 endpoint: http://10.100.79.3:8210/generate method: POST @@ -948,16 +948,23 @@ services: Wrapper-side cleanup queued — once the upstream model defaults this, the catalog field will become optional or be dropped entirely. response: + # As of wrapper version that ships with image local/ace-step:v1 + # post 2026-05-11, /generate streams audio bytes inline with + # Content-Type set from the audio_format request field. The + # in-container output_path is exposed via X-Output-Path header + # for log correlation but is no longer load-bearing. type: audio mime_from_field: audio_format - output_field: output_path reproducibility: seedable: true deterministic: true notes: > actual_seeds parameter exposed; identical seeds + params = identical audio. Local infer-api.py patches upstream's broken 24-arg pipeline signature - (was 18 in upstream — caused crashes with audio_duration in `format` slot). + (was 18 in upstream — caused crashes with audio_duration in `format` slot) + AND inline-streams the generated audio bytes (was returning a JSON + path reference to a file inside the container, which was unreachable + from outside). estimated_latency: cold_start_s: 30 warm_per_unit: "~10–60s depending on audio_duration + infer_step" diff --git a/stacks/ace-step/infer-api.py b/stacks/ace-step/infer-api.py index d53624d..4c02bf1 100644 --- a/stacks/ace-step/infer-api.py +++ b/stacks/ace-step/infer-api.py @@ -16,6 +16,7 @@ # audio2audio_enable=False, ref_audio_strength=0.5, no ref audio, no # LoRA.) from fastapi import FastAPI, HTTPException +from fastapi.responses import Response from pydantic import BaseModel from typing import List, Optional import os @@ -24,6 +25,15 @@ import uuid app = FastAPI(title="ACEStep Pipeline API (patched)") +# audio_format -> Content-Type. The wrapper writes the file in the +# format the upstream pipeline supports for the requested extension; +# we just need to label the bytes correctly on the wire. +_AUDIO_MIME = { + "wav": "audio/wav", + "mp3": "audio/mpeg", + "flac": "audio/flac", +} + class ACEStepInput(BaseModel): checkpoint_path: str @@ -78,8 +88,17 @@ def initialize_pipeline( ) -@app.post("/generate", response_model=ACEStepOutput) +@app.post("/generate") async def generate_audio(input_data: ACEStepInput): + """Generate music; respond with the audio bytes inline. + + Pre-2026-05-11 this returned a JSON {output_path} reference to a + file inside the container — useless to any external consumer + since the path wasn't reachable from outside. Now streams the + bytes back directly with the right Content-Type, and the file in + /app/outputs/ is incidental bookkeeping (the pipeline writes + there as part of its normal flow; we read it back for the wire). + """ try: model_demo = initialize_pipeline( input_data.checkpoint_path, @@ -115,14 +134,25 @@ async def generate_audio(input_data: ACEStepInput): input_data.lora_weight, ) - output_path = input_data.output_path or f"/app/outputs/output_{uuid.uuid4().hex}.wav" + ext = input_data.audio_format.lower() + output_path = ( + input_data.output_path + or f"/app/outputs/output_{uuid.uuid4().hex}.{ext}" + ) model_demo(*params, save_path=output_path) - return ACEStepOutput( - status="success", - output_path=output_path, - message="Audio generated successfully", + with open(output_path, "rb") as f: + audio_bytes = f.read() + + return Response( + content=audio_bytes, + media_type=_AUDIO_MIME.get(ext, "application/octet-stream"), + headers={ + # Surfaces the in-container path for debugging/log correlation + # without putting it on the response body. + "X-Output-Path": output_path, + }, ) except Exception as e: