# Scriberr — self-hosted audio/video transcription with speaker diarization. # Upstream: https://github.com/rishikanthc/Scriberr (Go + SvelteKit, SQLite). # # Transcription runs locally via WhisperX (Whisper + pyannote diarization); # NVIDIA Parakeet / Canary models are also selectable in the UI. Optional # summarisation / transcript chat talks to any OpenAI-compatible endpoint — # point it at the LiteLLM gateway rather than a paid API (see README). # # ── IMAGE: BUILT LOCALLY, ON PURPOSE ────────────────────────────────────── # fv-ml1's RTX PRO 6000 Blackwell cards are **sm_120**. Upstream publishes # `scriberr-cuda` (built for sm_61…sm_89 — no sm_120 kernels) and documents a # `scriberr-cuda-blackwell` image that **has never actually been published** # (GHCR returns no tags for it, checked 2026-08-23). The sm_120 path upstream # ships is `Dockerfile.cuda.12.9` (CUDA 12.9.1 + cu128 torch), built locally. # Do NOT "simplify" this to the published `scriberr-cuda` image — it will # fail on these cards or silently fall back to CPU. # Rebuild: see README "Rebuilding" — checkout lives at # /tank/scriberr/src/Scriberr on fv-ml1. # # ── GPU PINNING ─────────────────────────────────────────────────────────── # Pinned to **GPU1** via explicit device_ids, per the house convention and # because GPU0 is fully committed to the `gen` seat. GPU1 shares space with # the `sec` seat, so this stack is a guest there — keep an eye on VRAM. # NOTE: do NOT add `NVIDIA_VISIBLE_DEVICES=all` (as upstream's compose does). # It overrides the device_ids reservation and exposes both cards. # # All tunables live in .env — edit that, not this file. services: scriberr: image: ${SCRIBERR_IMAGE:-scriberr:local-blackwell} container_name: scriberr restart: unless-stopped ports: - "${SCRIBERR_BIND:-0.0.0.0}:${SCRIBERR_PORT}:8080" volumes: # Bind mounts rather than named volumes: /var/lib/docker on fv-ml1 # lives on zroot with limited headroom, while /tank has terabytes. # Model weights (Whisper, pyannote, NeMo) land in whisperx-env and are # multi-GB — they must not go anywhere near the root pool. - ${SCRIBERR_DATA_DIR}:/app/data - ${SCRIBERR_ENV_DIR}:/app/whisperx-env environment: # ⚠ 10001, NOT the fleet-usual 1000 — this is load-bearing. # Dockerfile.cuda.12.9 creates `appuser` at uid 10001 (Ubuntu 24.04's # base image already owns uid 1000 as `ubuntu`, so upstream moved it) and # chowns /app to 10001. The entrypoint's PUID remapping only chowns # /app/data + /app/whisperx-env, not /app itself, so running as 1000 # leaves the app unable to open its SQLite DB and it crash-loops with # `unable to open database file: out of memory (14)` — which is # SQLITE_CANTOPEN wearing a misleading message, not a real OOM. # The host bind-mount dirs are therefore chowned to 10001:10001 too. # Verified 2026-08-23: PUID=1000 crash-loops, PUID=10001 starts clean. - PUID=${SCRIBERR_PUID:-10001} - PGID=${SCRIBERR_PGID:-10001} - APP_ENV=production # Served over plain HTTP on the LAN. Left at the production default of # `true`, the session cookie is marked Secure and the browser silently # drops it — you log in, get bounced back to the login page, and the # logs show nothing wrong. This must stay false while access is HTTP. - SECURE_COOKIES=${SCRIBERR_SECURE_COOKIES:-false} # Upstream defaults to localhost origins only, which fails CORS when # reached by host IP. Keep this in sync with how the app is reached. - ALLOWED_ORIGINS=${SCRIBERR_ALLOWED_ORIGINS} - NVIDIA_DRIVER_CAPABILITIES=compute,utility # Scriberr builds each model backend's Python env with `uv` at runtime. # uv's default link mode reflink/hardlinks out of its cache, which fails # on this overlayfs+ZFS combination with a misleading # "Failed to clone ... Resource temporarily unavailable (os error 11)" # and takes out the Parakeet + Sortformer backends (WhisperX survives). # `copy` trades a little disk and time for it actually working. - UV_LINK_MODE=${SCRIBERR_UV_LINK_MODE:-copy} deploy: resources: reservations: devices: - driver: nvidia device_ids: ["${SCRIBERR_GPU_ID:-1}"] capabilities: [gpu] healthcheck: # 127.0.0.1 rather than localhost — the IPv6-first resolution trap has # bitten news-digest and chatterbox in this fleet before. # start_period is generous: first boot builds a Python env and pulls # several GB of model weights before the port answers. test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8080/ >/dev/null || exit 1"] interval: 30s timeout: 5s retries: 3 start_period: 600s networks: - tnet labels: # ⚠ The group name MUST match a key in the dashboard's settings.yaml # `layout:` block. A group that appears nowhere in that block gets no # `tab:` assignment, and Homepage renders an untabbed group on EVERY tab. # This label read `AI Systems` — a group that existed nowhere — from # 2026-08-23 until it was caught on 2026-08-24. # `AI - Studios` and not one of the ASR groups because Scriberr is a # transcription UI you open and work in, which is what Studios collects; # the bare ASR endpoints (Parakeet, Speaches) live in the collapsed # `AI - Audio Tools` group instead. - homepage.group=AI - Studios - homepage.name=Scriberr - homepage.icon=mdi-microphone-message - homepage.description=Audio/video transcription + diarization (fv-ml1, GPU1) - homepage.href=http://10.251.50.54:${SCRIBERR_PORT} networks: tnet: name: traefik-net external: true