diff --git a/stacks/parakeet-nemo/.env.example b/stacks/parakeet-nemo/.env.example index 91d871c..f483d6b 100644 --- a/stacks/parakeet-nemo/.env.example +++ b/stacks/parakeet-nemo/.env.example @@ -1,5 +1,5 @@ # Copy to .env next to compose.yaml on the host. -PARAKEET_NEMO_TAG=nemo-0.1.0 +PARAKEET_NEMO_TAG=nemo-0.1.1 # Port the seat listens on. 8300 is the seat port LiteLLM's ext-stt/whisper-1 point at; # run acceptance on a temporary port first, then cut over by changing this line. PARAKEET_NEMO_PORT=8300 @@ -9,3 +9,10 @@ PARAKEET_NEMO_PORT=8300 PARAKEET_NEMO_REV=fe53cd885760c96b6a5f51a0bfd362cb4584a98b # Ascending silent warm-up clips in seconds (CUDA-graph capture + longest-shape kernel warm). # PARAKEET_NEMO_WARMUP=1,8,60 +# Hard per-process VRAM ceiling, MiB (set_per_process_memory_fraction). Over-cap requests get +# 503 and the seat stays alive; do not raise past ~3,840 — GPU 0's vLLM neighbours need the rest. +# PARAKEET_NEMO_MEM_CAP_MIB=3840 +# 0 (default): the CUDA-graph decoder pins torch cache and wedges when the windowed path frees it. +# PARAKEET_NEMO_CUDA_GRAPHS=0 +# Long-form window size in seconds (files over this are transcribed in windows of this size). +# PARAKEET_NEMO_WINDOW_S=360 diff --git a/stacks/parakeet-nemo/README.md b/stacks/parakeet-nemo/README.md index 405cf30..22b817a 100644 --- a/stacks/parakeet-nemo/README.md +++ b/stacks/parakeet-nemo/README.md @@ -43,19 +43,29 @@ required; keep this section as the license note. Weights pinned at HF revision ## GPU 0 room -The seat rests ~2.1 GB, serves to ~2.8 GB, loads under ~3.0 GB (see the ops log). But its -**steady state after ANY long (windowed) request is 3,582 MiB** — torch caches the window peak -and does not return it (measured flat across repeated 12-min requests, audit 2026-10-01). Plan -GPU 0 against 3,582 MiB, not the at-rest figure: in steady state the card sits at ~385 MiB Free. -Room was taken from `vllm-gen-small`: `--gpu-memory-utilization` 0.48 → 0.36 at cut-over → 0.33 -after the audit (config-only; applies at its NEXT restart, and gives that restart ~3 GiB of -boot-check margin against the steady-state figure). Its KV is byte-pinned -(`--kv-cache-memory`), so the util number costs it nothing — boot log identical at 670,142 -tokens / 2.56×. ⚠ Before restarting ANY vLLM seat on this card, do the boot-check arithmetic -against measured `nvidia-smi` Free: required = util × total, available = Free + that seat's own -resident memory. The cut-over iteration (0.46, 0.40 both refusing boot before 0.36 booted) took -gen-small down ~34 minutes for want of that one line of arithmetic. util does NOT predict -resident VRAM. GPU 1 is NOT an option: its free memory is intern-decision's 32k headroom. +The seat rests ~2.1 GB. **Since nemo-0.1.1 it returns to rest after long requests**: the +windowed path calls `torch.cuda.empty_cache()` around each window, so a 12-min file peaks at +~3,028 MiB during the request and falls back to ~2,108 after (measured 2026-10-01, restarts=0). +Before 0.1.1 the seat PARKED at the window peak (3,582 MiB steady), and that cached peak left +gen-small no room for its runtime workspace — an EngineCore CUDA-OOM incident at 04:21 PT. +Three seat-side controls, all load-bearing: +- **`MEM_CAP_MIB=3840`** (env, default): a hard `set_per_process_memory_fraction` ceiling. An + over-cap request answers **503** with the seat still alive (proved at cap=2000: two 503s, then + short requests fine) — the failure lands on us, never on a neighbour's allocation. +- **`CUDA_GRAPHS=0`** (default): the CUDA-graph greedy decoder pins memory in torch's cache and + died with an illegal-memory-access the first time `empty_cache` freed a graph-pool block. + Graphs off costs latency (12-min file 3.0 s vs 1.2 s; short bins 35-62 ms vs 33-42 ms — still + 4-15x faster than the sherpa seat) and buys a lower, honestly-returned footprint. +- **`WINDOW_S=360`** (default): see the windowing note above; mask is T×T even under local attention. +Room came from `vllm-gen-small`: `--gpu-memory-utilization` 0.48 → 0.33 (config-only, applies at +its next restart; KV byte-pinned, boot log identical at 670,142 tokens / 2.56×). gen-small's +runtime growth was measured nvidia-smi-per-process, not computed: 3 realistic requests +(1,351 prompt / ~180 completion tokens) moved it ZERO from its 36,116 MiB — the workspace +allocation lands at engine init, right after restart (infra-ops saw the growth window at +restart+3-requests). ⚠ Before restarting ANY vLLM seat on this card, do the boot-check +arithmetic against measured `nvidia-smi` Free: required = util × total, available = Free + that +seat's own resident memory. GPU 1 is NOT an option: its free memory is intern-decision's 32k +headroom. ## Rollback diff --git a/stacks/parakeet-nemo/app.py b/stacks/parakeet-nemo/app.py index 4709585..672deef 100644 --- a/stacks/parakeet-nemo/app.py +++ b/stacks/parakeet-nemo/app.py @@ -31,6 +31,11 @@ from fastapi.responses import JSONResponse MODEL_PATH = os.environ["MODEL_PATH"] WARMUP_SECONDS = [int(x) for x in os.environ.get("WARMUP_SECONDS", "1,8,60").split(",")] +# Hard ceiling for the whole process, MiB. The measured window peak is 3,582 (audit 2026-10-01); +# 3,840 gives a little headroom and NO more. GPU 0 is shared with vLLM seats that grow at RUNTIME +# (~0.8 GB for gen-small), and an unbounded window cache OOMed gen-small's EngineCore on its first +# request after the switch — a cached peak is an unpaid debt to the neighbours. +MEM_CAP_MIB = int(os.environ.get("MEM_CAP_MIB", "3840")) SR = 16000 logger = logging.getLogger("parakeet-nemo") @@ -50,7 +55,7 @@ def _load(): d = m.cfg.decoding with open_dict(d): d.strategy = "greedy_batch" - d.greedy["use_cuda_graph_decoder"] = True + d.greedy["use_cuda_graph_decoder"] = os.environ.get("CUDA_GRAPHS", "1") == "1" m.change_decoding_strategy(d, verbose=False) # transcribe() sets these on entry; the direct path must match, and must not dither (dither is # a training-time augmentation and makes the same file decode differently on each call). @@ -71,6 +76,10 @@ def _load(): model = _load() +# Hard per-process ceiling on torch allocations (see MEM_CAP_MIB): an over-size request must +# fail HERE, at us, instead of stealing runtime room from a neighbour's process. torch counts +# RESERVED bytes against this, which is exactly the ledger we want capped. +torch.cuda.set_per_process_memory_fraction(MEM_CAP_MIB / (torch.cuda.get_device_properties(0).total_memory / 2**20)) def _hyp_text(h) -> str: @@ -127,7 +136,18 @@ def _decode(raw: bytes) -> str: win = int(os.environ.get("WINDOW_S", "360")) * SR if len(samples) <= win: return _infer(samples) - parts = [_infer(samples[i:i + win]) for i in range(0, len(samples), win)] + # Windowed path: empty BEFORE each window too — torch's cap counts RESERVED bytes, and cached + # blocks from the previous window would otherwise count against it and bite spuriously. + torch.cuda.empty_cache() + try: + parts = [] + for i in range(0, len(samples), win): + parts.append(_infer(samples[i:i + win])) + torch.cuda.empty_cache() + except torch.OutOfMemoryError as exc: + # Our cap (or a neighbour's pressure) bit: answer 503, give the cache back either way. + torch.cuda.empty_cache() + raise HTTPException(503, f"GPU memory limit reached for this file: {exc}") from exc return " ".join(p for p in parts if p) diff --git a/stacks/parakeet-nemo/compose.yaml b/stacks/parakeet-nemo/compose.yaml index ec96ad2..2272c24 100644 --- a/stacks/parakeet-nemo/compose.yaml +++ b/stacks/parakeet-nemo/compose.yaml @@ -32,6 +32,11 @@ services: environment: - MODEL_PATH=/hf/hub/models--nvidia--parakeet-unified-en-0.6b/snapshots/${PARAKEET_NEMO_REV}/parakeet-unified-en-0.6b.nemo - WARMUP_SECONDS=${PARAKEET_NEMO_WARMUP:-1,8,60} + # CUDA graphs pin memory in torch's cache, which fights the empty_cache the windowed path + # needs to return memory to GPU 0's neighbours (illegal-memory-access wedge, 2026-10-01). + - CUDA_GRAPHS=${PARAKEET_NEMO_CUDA_GRAPHS:-0} + - MEM_CAP_MIB=${PARAKEET_NEMO_MEM_CAP_MIB:-3840} + - WINDOW_S=${PARAKEET_NEMO_WINDOW_S:-360} - LOG_LEVEL=${PARAKEET_NEMO_LOG_LEVEL:-INFO} volumes: - /tank/aimodels/huggingface:/hf:ro