# Parakeet ASR via NeMo torch (unified-en-0.6b, bf16 weights) + our own thin FastAPI wrapper. # # Replacement seat for stacks/parakeet (sherpa-onnx int8). Same port (:8300), same endpoints, # same body — LiteLLM and `talk` need no change. Rollback: stop this container, start the old # `parakeet` one (kept; container and image both intact). # # HOST: fv-ml1, GPU 0. # # ⚠ GPU 0 room came from gen-small's util SLACK, not its KV: util went 0.48 -> 0.36 (now 0.33 in its .env), # and the KV is byte-pinned (unchanged at 670,142 tokens). Plan GPU 0 against this seat's STEADY STATE of # 3,582 MiB (the torch cache after a long, windowed request), not its at-rest figure; that leaves Free ~385 MiB # (the old seat held 1,690 MiB). Do not raise that util back without re-measuring nvidia-smi Free # on GPU 0 — util does not predict resident VRAM (see the 09-15 note in gen-small-seat/.env). # # ⚠ Weights: /tank/aimodels/huggingface mounted READ-ONLY. nvidia/parakeet-unified-en-0.6b @ # fe53cd885760c96b6a5f51a0bfd362cb4584a98b. HF_HUB_OFFLINE=1 in the image: the seat never phones home. # # API (identical to the replaced seat): # POST /transcribe — multipart file upload, returns {"text": "..."} # POST /v1/audio/transcriptions — same body, OpenAI-compatible path alias # GET /healthz # # All tunables live in .env — edit that, not this file. services: parakeet-nemo: image: local/parakeet-nemo:${PARAKEET_NEMO_TAG} container_name: parakeet-nemo restart: unless-stopped ports: - "${PARAKEET_NEMO_BIND:-0.0.0.0}:${PARAKEET_NEMO_PORT}:8000" environment: - MODEL_PATH=/hf/hub/models--nvidia--parakeet-unified-en-0.6b/snapshots/${PARAKEET_NEMO_REV}/parakeet-unified-en-0.6b.nemo - WARMUP_SECONDS=${PARAKEET_NEMO_WARMUP:-1,8,60} # CUDA graphs pin memory in torch's cache, which fights the empty_cache the windowed path # needs to return memory to GPU 0's neighbours (illegal-memory-access wedge, 2026-10-01). - CUDA_GRAPHS=${PARAKEET_NEMO_CUDA_GRAPHS:-0} - MEM_CAP_MIB=${PARAKEET_NEMO_MEM_CAP_MIB:-3840} - WINDOW_S=${PARAKEET_NEMO_WINDOW_S:-360} - LOG_LEVEL=${PARAKEET_NEMO_LOG_LEVEL:-INFO} volumes: - /tank/aimodels/huggingface:/hf:ro deploy: resources: reservations: devices: - driver: nvidia device_ids: ["${PARAKEET_NEMO_GPU:-0}"] capabilities: [gpu] networks: - tnet healthcheck: test: ["CMD-SHELL", "wget -q -O /dev/null http://localhost:8000/healthz || exit 1"] interval: 30s timeout: 10s retries: 3 # Import-time model load + three warm-up decodes; no download (weights are mounted). start_period: 240s labels: - homepage.group=AI - Audio Tools - homepage.name=Parakeet ASR (NeMo) - homepage.icon=mdi-microphone - homepage.description=Parakeet-unified-en speech-to-text via NeMo bf16 (fv-ml1 GPU 0) - homepage.href=http://10.251.50.54:${PARAKEET_NEMO_PORT} networks: tnet: name: traefik-net external: true