From 750675e391aab546cc5861834e7992383ae16fd7 Mon Sep 17 00:00:00 2001 From: Vuong Hoang Date: Wed, 30 Sep 2026 09:40:08 -0700 Subject: [PATCH] feat(intern-decision): cap 9.0 GiB with MAX_TOKENS 7168, the largest call measured to fit Both are required in compose because they are coupled: MAX_TOKENS is checked before the forward pass, so an oversized call is a clear 422 instead of reaching the cap as a 503. Pre-deploy floor is nvidia-smi Free >= 15,400 MiB on GPU 1 (card peak 9,876 + scriberr 5,496). --- .../acceptance/checks.py | 20 +++++++++++++++++-- stacks/intern-decision/.env.example | 13 +++++++----- stacks/intern-decision/README.md | 9 +++++++-- stacks/intern-decision/compose.yaml | 9 +++++---- 4 files changed, 38 insertions(+), 13 deletions(-) diff --git a/services/intern-decision-serve/acceptance/checks.py b/services/intern-decision-serve/acceptance/checks.py index 0b7c0fd..f8908f9 100644 --- a/services/intern-decision-serve/acceptance/checks.py +++ b/services/intern-decision-serve/acceptance/checks.py @@ -10,6 +10,7 @@ Stdlib only, so it runs on fv-ml1's host python or on nh3-dev against a live ins maxreq the largest request the API accepts: MAX_DECISIONS decisions x 16 options, the state grown until each 16-question call sits just under MAX_TOKENS (found from the model's own 422 message). Sent --repeats times; status, tokens and /health's allocator peak recorded. + maxone ONE binary question with the state filling MAX_TOKENS: --repeats x 200, one token more is 422 fit under the cap, the largest 16-question call that answers 200 (binary search, in tokens) oom the over-cap request (maxreq's body when it ran) -> 503 out_of_memory, then reserved memory back to its resting value and an ordinary request answered 200. Earlier text: a long request against a deliberately tight cap -> 503 out_of_memory, then reserved @@ -122,11 +123,11 @@ def question_block(i): for j in range(16)]} -def find_max_state(url, h, base_state, max_tokens, n_decisions=16): +def find_max_state(url, h, base_state, max_tokens, n_decisions=16, decisions=None): """Grow the state (repeating base_state) until the request is refused with 422, using the token count the model's own 422 reports; return the longest state the API accepts and its tokens per call. Sized with the WHOLE n_decisions request, so the largest call in it is the one that decides.""" - decisions = [question_block(i) for i in range(n_decisions)] + decisions = decisions or [question_block(i) for i in range(n_decisions)] text = (base_state + "\n\n") * 4 def tokens_for(chars): @@ -191,6 +192,20 @@ def check_fit(url, h, base_state): return {"largest_ok_state_chars": lo, "largest_ok_tokens": best, "probes": probes} +def check_maxone(url, h, base_state, max_tokens, repeats): + """The other extreme of the largest call: ONE binary question, the state filling the rest of MAX_TOKENS.""" + one = [{"id": "only", "question": "Does the policy allow the exception described?", "options": YESNO}] + state, tokens = find_max_state(url, h, base_state, max_tokens, decisions=one) + over = post(url + "/decide/shared", {"state": state + " extra words", "decisions": one}, h) + runs = [] + for _ in range(repeats): + s, j, ms = post(url + "/decide", {"id": "only", "state": state, **{k: one[0][k] for k in ("question", "options")}}, h) + runs.append({"status": s, "e2e_ms": round(ms, 1), "input_tokens": j.get("input_tokens"), + "code": (j.get("error") or {}).get("code"), "t_end": time.time()}) + return {"state_chars": len(state), "tokens": tokens, "one_more_is": [over[0], (over[1].get("error") or {}).get("message")], + "runs": runs, "pass": all(r["status"] == 200 for r in runs) and over[0] == 422} + + def check_oom(url, h, long_state): """The over-cap request: the largest one (maxreq) if it ran, else 16 criteria over the long state.""" before = get(url + "/health")[1]["model"] @@ -232,6 +247,7 @@ def main(): "maxreq": lambda: check_maxreq(url, h, long_state, health["max_tokens"], health["max_decisions"], args.repeats), "fit": lambda: check_fit(url, h, long_state), + "maxone": lambda: check_maxone(url, h, long_state, health["max_tokens"], args.repeats), "oom": lambda: check_oom(url, h, long_state)}[c]() report[c]["t_start"], report[c]["t_end"] = t, time.time() print(c, json.dumps({k: v for k, v in report[c].items() if k not in ("rows", "runs")})[:600], flush=True) diff --git a/stacks/intern-decision/.env.example b/stacks/intern-decision/.env.example index 449690a..81ad0be 100644 --- a/stacks/intern-decision/.env.example +++ b/stacks/intern-decision/.env.example @@ -6,10 +6,13 @@ HOST_IP=10.251.50.54 # fv-ml1 GPU 1 = the utility card (vllm-coder, erp, meromero, scriberr). GPU_ID=1 # HARD torch-allocator cap: the single knob that holds the container's WHOLE nvidia-smi footprint -# (CUDA context included) <= 10,300 MiB, the GPU 1 budget next to scriberr (2026-09-30). -# footprint <= cap + non-allocator overhead = 9,472 MiB + 662 MiB = 10,134 MiB -# (overhead measured flat at 660-662 MiB with the single inference thread; README "VRAM"). -# 9.25 still answers the largest request the API accepts (4 calls x 8,191 tokens) with a 200. -VRAM_CAP_GIB=9.25 +# (CUDA context included) inside GPU 1's budget next to scriberr (infra-ops, 2026-09-30): +# GPU 1 nvidia-smi Free >= 15,400 MiB = our card peak 9,876 (cap 9.0 GiB + 660 MiB outside the +# allocator, measured) + scriberr's peak 5,496, rounded up. README "VRAM". +VRAM_CAP_GIB=9.0 +# Tokens per CALL (state + up to 16 questions), checked BEFORE the forward pass: a longer call is a +# clear 422. 7,168 is the largest call measured to fit under VRAM_CAP_GIB=9.0. Change the two TOGETHER, +# and re-measure (README "VRAM"): a larger value would let a call reach the cap and return 503. +MAX_TOKENS=7168 # >= 32 characters; source of truth: secret get intern-decision/api-token INTERN_DECISION_API_TOKEN= diff --git a/stacks/intern-decision/README.md b/stacks/intern-decision/README.md index 8ae53d8..d4e439f 100644 --- a/stacks/intern-decision/README.md +++ b/stacks/intern-decision/README.md @@ -190,9 +190,14 @@ cd /opt/docker/compose/intern-decision && new=$(sed 's/^IMAGE=.*/IMAGE=intern-de && printf '%s\n' "$new" > .env && docker compose config -q && docker compose up -d ``` -**Before any deploy onto GPU 1:** GPU 1 must have at least 15,800 MiB free -(`nvidia-smi -i 1 --query-gpu=memory.free --format=csv`). If it has less, stop. Do not squeeze +**Before any deploy onto GPU 1**, check two things. If either fails, stop; do not squeeze scriberr. +- nvidia-smi's own `Free` on GPU 1 must be at least **15,400 MiB**: + `nvidia-smi -i 1 --query-gpu=memory.free --format=csv`. That is our card peak of 9,876 MiB plus + scriberr's 5,496, rounded up. Do not use total − used, which misses the driver's 640 MiB + reserve. +- Scriberr must not be running a job. This command must print 0: + `docker logs --since 2m scriberr | grep -c "Processing single-track job"`. Startup fails closed. A container that never reaches healthy did not pass its own checks: the `inference.py` hash, the pinned snapshot, the warm-up, the text-only swap and the prompt hash. diff --git a/stacks/intern-decision/compose.yaml b/stacks/intern-decision/compose.yaml index 021705e..3490bfb 100644 --- a/stacks/intern-decision/compose.yaml +++ b/stacks/intern-decision/compose.yaml @@ -8,11 +8,11 @@ # fv-ml1 from that dir. # # ⚠ VRAM_CAP_GIB is a HARD cap on torch's allocator (per-process memory fraction), set so the -# container's WHOLE nvidia-smi footprint, CUDA context included, stays <= 10,300 MiB whatever -# the request (Prime/infra-ops budget with scriberr, 2026-09-30). A request that needs more +# container's WHOLE nvidia-smi footprint, CUDA context included, fits beside scriberr's peak +# (infra-ops budget, 2026-09-30); MAX_TOKENS keeps every accepted call under the cap. A request that needs more # gets 503 out_of_memory and the service stays up. See the README before changing it. # -# .env (tunables): IMAGE, PORT, GPU_ID, VRAM_CAP_GIB, HOST_IP, INTERN_DECISION_API_TOKEN +# .env (tunables): IMAGE, PORT, GPU_ID, VRAM_CAP_GIB, MAX_TOKENS, HOST_IP, INTERN_DECISION_API_TOKEN # (vault intern-decision/api-token, >= 32 chars). name: intern-decision @@ -28,7 +28,8 @@ services: INTERN_DECISION_API_TOKEN: ${INTERN_DECISION_API_TOKEN:?set INTERN_DECISION_API_TOKEN} INTERN_DECISION_DEVICE: cuda INTERN_DECISION_VRAM_CAP_GIB: ${VRAM_CAP_GIB:?set VRAM_CAP_GIB} - INTERN_DECISION_MAX_TOKENS: ${MAX_TOKENS:-8192} + # Coupled to VRAM_CAP_GIB: the largest call measured to fit under the cap (README "VRAM"). + INTERN_DECISION_MAX_TOKENS: ${MAX_TOKENS:?set MAX_TOKENS} INTERN_DECISION_MAX_DECISIONS: ${MAX_DECISIONS:-64} # POSTs in progress (queued + scoring) before new ones get 429 busy. INTERN_DECISION_MAX_QUEUE: ${MAX_QUEUE:-32}