# Copy to .env next to compose.yaml on the host. PARAKEET_NEMO_TAG=nemo-0.1.1 # Port the seat listens on. 8300 is the seat port LiteLLM's ext-stt/whisper-1 point at; # run acceptance on a temporary port first, then cut over by changing this line. PARAKEET_NEMO_PORT=8300 # PARAKEET_NEMO_BIND=0.0.0.0 # PARAKEET_NEMO_GPU=0 # Pinned HF revision of nvidia/parakeet-unified-en-0.6b (sha256 ec23ed91... of the .nemo). PARAKEET_NEMO_REV=fe53cd885760c96b6a5f51a0bfd362cb4584a98b # Ascending silent warm-up clips in seconds (CUDA-graph capture + longest-shape kernel warm). # PARAKEET_NEMO_WARMUP=1,8,60 # Hard per-process VRAM ceiling, MiB (set_per_process_memory_fraction). Over-cap requests get # 503 and the seat stays alive; do not raise past ~3,840 — GPU 0's vLLM neighbours need the rest. # PARAKEET_NEMO_MEM_CAP_MIB=3840 # 0 (default): the CUDA-graph decoder pins torch cache and wedges when the windowed path frees it. # PARAKEET_NEMO_CUDA_GRAPHS=0 # Long-form window size in seconds (files over this are transcribed in windows of this size). # PARAKEET_NEMO_WINDOW_S=360