131 lines
7.4 KiB
YAML
131 lines
7.4 KiB
YAML
# Scriberr — self-hosted audio/video transcription with speaker diarization.
|
||
# Upstream: https://github.com/rishikanthc/Scriberr (Go + SvelteKit, SQLite).
|
||
#
|
||
# Transcription runs locally via WhisperX (Whisper + pyannote diarization);
|
||
# NVIDIA Parakeet / Canary models are also selectable in the UI. Optional
|
||
# summarisation / transcript chat talks to any OpenAI-compatible endpoint —
|
||
# point it at the LiteLLM gateway rather than a paid API (see README).
|
||
#
|
||
# ── IMAGE: BUILT LOCALLY, ON PURPOSE ──────────────────────────────────────
|
||
# fv-ml1's RTX PRO 6000 Blackwell cards are **sm_120**. Upstream publishes
|
||
# `scriberr-cuda` (built for sm_61…sm_89 — no sm_120 kernels) and documents a
|
||
# `scriberr-cuda-blackwell` image that **has never actually been published**
|
||
# (GHCR returns no tags for it, checked 2026-08-23). The sm_120 path upstream
|
||
# ships is `Dockerfile.cuda.12.9` (CUDA 12.9.1 + cu128 torch), built locally.
|
||
# Do NOT "simplify" this to the published `scriberr-cuda` image — it will
|
||
# fail on these cards or silently fall back to CPU.
|
||
# Rebuild: see README "Rebuilding" — checkout lives at
|
||
# /tank/scriberr/src/Scriberr on fv-ml1.
|
||
#
|
||
# ── GPU PINNING ───────────────────────────────────────────────────────────
|
||
# Pinned to **GPU1** via explicit device_ids, per the house convention and
|
||
# because GPU0 is fully committed to the `gen` seat. GPU1 shares space with
|
||
# the `sec` seat, so this stack is a guest there — keep an eye on VRAM.
|
||
# NOTE: do NOT add `NVIDIA_VISIBLE_DEVICES=all` (as upstream's compose does).
|
||
# It overrides the device_ids reservation and exposes both cards.
|
||
#
|
||
# All tunables live in .env — edit that, not this file.
|
||
|
||
services:
|
||
scriberr:
|
||
image: ${SCRIBERR_IMAGE:-scriberr:local-blackwell}
|
||
container_name: scriberr
|
||
restart: unless-stopped
|
||
ports:
|
||
- "${SCRIBERR_BIND:-0.0.0.0}:${SCRIBERR_PORT}:8080"
|
||
volumes:
|
||
# Bind mounts rather than named volumes: /var/lib/docker on fv-ml1
|
||
# lives on zroot with limited headroom, while /tank has terabytes.
|
||
# Model weights (Whisper, pyannote, NeMo) land in whisperx-env and are
|
||
# multi-GB — they must not go anywhere near the root pool.
|
||
- ${SCRIBERR_DATA_DIR}:/app/data
|
||
- ${SCRIBERR_ENV_DIR}:/app/whisperx-env
|
||
environment:
|
||
# ⚠ 10001, NOT the fleet-usual 1000 — this is load-bearing.
|
||
# Dockerfile.cuda.12.9 creates `appuser` at uid 10001 (Ubuntu 24.04's
|
||
# base image already owns uid 1000 as `ubuntu`, so upstream moved it) and
|
||
# chowns /app to 10001. The entrypoint's PUID remapping only chowns
|
||
# /app/data + /app/whisperx-env, not /app itself, so running as 1000
|
||
# leaves the app unable to open its SQLite DB and it crash-loops with
|
||
# `unable to open database file: out of memory (14)` — which is
|
||
# SQLITE_CANTOPEN wearing a misleading message, not a real OOM.
|
||
# The host bind-mount dirs are therefore chowned to 10001:10001 too.
|
||
# Verified 2026-08-23: PUID=1000 crash-loops, PUID=10001 starts clean.
|
||
- PUID=${SCRIBERR_PUID:-10001}
|
||
- PGID=${SCRIBERR_PGID:-10001}
|
||
- APP_ENV=production
|
||
# Served over plain HTTP on the LAN. Left at the production default of
|
||
# `true`, the session cookie is marked Secure and the browser silently
|
||
# drops it — you log in, get bounced back to the login page, and the
|
||
# logs show nothing wrong. This must stay false while access is HTTP.
|
||
- SECURE_COOKIES=${SCRIBERR_SECURE_COOKIES:-false}
|
||
# Upstream defaults to localhost origins only, which fails CORS when
|
||
# reached by host IP. Keep this in sync with how the app is reached.
|
||
- ALLOWED_ORIGINS=${SCRIBERR_ALLOWED_ORIGINS}
|
||
- NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||
# Scriberr builds each model backend's Python env with `uv` at runtime.
|
||
# uv's default link mode reflink/hardlinks out of its cache, which fails
|
||
# on this overlayfs+ZFS combination with a misleading
|
||
# "Failed to clone ... Resource temporarily unavailable (os error 11)"
|
||
# and takes out the Parakeet + Sortformer backends (WhisperX survives).
|
||
# `copy` trades a little disk and time for it actually working.
|
||
- UV_LINK_MODE=${SCRIBERR_UV_LINK_MODE:-copy}
|
||
# ── GPU 1 memory budget (2026-09-30, Prime: Scriberr shares GPU 1 with
|
||
# intern-decision, which holds ~9.7 GB resting / 10.3 GB peak). ──────────
|
||
# Parakeet's buffered path cuts audio into slices of this many seconds
|
||
# (Scriberr reads it in parakeet_adapter.go for BOTH the "is this long
|
||
# audio" threshold and --chunk-len; upstream default 300). Measured peak
|
||
# GPU memory on a 35-min file, n=3 each, deterministic:
|
||
# 300 s 9,384 MiB · 120 s 6,510 · 60 s 5,976 · 10 s 5,634 (fixed floor)
|
||
# 120 s + expandable_segments 5,496 · 60 s + expandable_segments 5,502
|
||
# The floor, not the slice, dominates below ~120 s; expandable_segments is
|
||
# what removes the fragmentation on top of it. 120 s + expandable fits
|
||
# beside intern-decision (9.0 GiB cap, 9,876 MiB card peak) even at both
|
||
# peaks: 9,876 + 5,496 = 15,372 of GPU 1's 15,442 MiB nvidia-smi Free
|
||
# (70 MiB spare; Free is NOT total−used, the driver reserves ~640 MiB). Transcripts
|
||
# change slightly: 95.8% word-sequence similarity vs 300 s (7,599 vs
|
||
# 7,645 words); the diffs are mostly casing/punctuation spread through
|
||
# the file, ~40 words at the 17 cuts. A-vs-A at 300 s: identical.
|
||
# Scriberr passes os.Environ() to the uv subprocess, so both reach NeMo.
|
||
- PARAKEET_CHUNK_THRESHOLD_SECS=${SCRIBERR_PARAKEET_CHUNK_SECS:-120}
|
||
- PYTORCH_CUDA_ALLOC_CONF=${SCRIBERR_PYTORCH_CUDA_ALLOC_CONF:-expandable_segments:True}
|
||
deploy:
|
||
resources:
|
||
reservations:
|
||
devices:
|
||
- driver: nvidia
|
||
device_ids: ["${SCRIBERR_GPU_ID:-1}"]
|
||
capabilities: [gpu]
|
||
healthcheck:
|
||
# 127.0.0.1 rather than localhost — the IPv6-first resolution trap has
|
||
# bitten news-digest and chatterbox in this fleet before.
|
||
# start_period is generous: first boot builds a Python env and pulls
|
||
# several GB of model weights before the port answers.
|
||
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8080/ >/dev/null || exit 1"]
|
||
interval: 30s
|
||
timeout: 5s
|
||
retries: 3
|
||
start_period: 600s
|
||
networks:
|
||
- tnet
|
||
labels:
|
||
# ⚠ The group name MUST match a key in the dashboard's settings.yaml
|
||
# `layout:` block. A group that appears nowhere in that block gets no
|
||
# `tab:` assignment, and Homepage renders an untabbed group on EVERY tab.
|
||
# This label read `AI Systems` — a group that existed nowhere — from
|
||
# 2026-08-23 until it was caught on 2026-08-24.
|
||
# `AI - Studios` and not one of the ASR groups because Scriberr is a
|
||
# transcription UI you open and work in, which is what Studios collects;
|
||
# the bare ASR endpoints (Parakeet, Speaches) live in the collapsed
|
||
# `AI - Audio Tools` group instead.
|
||
- homepage.group=AI - Studios
|
||
- homepage.name=Scriberr
|
||
- homepage.icon=mdi-microphone-message
|
||
- homepage.description=Audio/video transcription + diarization (fv-ml1, GPU1)
|
||
- homepage.href=http://10.251.50.54:${SCRIBERR_PORT}
|
||
|
||
networks:
|
||
tnet:
|
||
name: traefik-net
|
||
external: true
|