feat(intern-decision): cap 9.0 GiB with MAX_TOKENS 7168, the largest call measured to fit

Both are required in compose because they are coupled: MAX_TOKENS is checked before the forward
pass, so an oversized call is a clear 422 instead of reaching the cap as a 503. Pre-deploy floor
is nvidia-smi Free >= 15,400 MiB on GPU 1 (card peak 9,876 + scriberr 5,496).
This commit is contained in:
vh
2026-09-30 09:40:08 -07:00
parent 66034cc69e
commit 750675e391
4 changed files with 38 additions and 13 deletions
@@ -10,6 +10,7 @@ Stdlib only, so it runs on fv-ml1's host python or on nh3-dev against a live ins
maxreq the largest request the API accepts: MAX_DECISIONS decisions x 16 options, the state
grown until each 16-question call sits just under MAX_TOKENS (found from the model's own
422 message). Sent --repeats times; status, tokens and /health's allocator peak recorded.
maxone ONE binary question with the state filling MAX_TOKENS: --repeats x 200, one token more is 422
fit under the cap, the largest 16-question call that answers 200 (binary search, in tokens)
oom the over-cap request (maxreq's body when it ran) -> 503 out_of_memory, then reserved memory
back to its resting value and an ordinary request answered 200. Earlier text: a long request against a deliberately tight cap -> 503 out_of_memory, then reserved
@@ -122,11 +123,11 @@ def question_block(i):
for j in range(16)]}
def find_max_state(url, h, base_state, max_tokens, n_decisions=16):
def find_max_state(url, h, base_state, max_tokens, n_decisions=16, decisions=None):
"""Grow the state (repeating base_state) until the request is refused with 422, using the token
count the model's own 422 reports; return the longest state the API accepts and its tokens per call.
Sized with the WHOLE n_decisions request, so the largest call in it is the one that decides."""
decisions = [question_block(i) for i in range(n_decisions)]
decisions = decisions or [question_block(i) for i in range(n_decisions)]
text = (base_state + "\n\n") * 4
def tokens_for(chars):
@@ -191,6 +192,20 @@ def check_fit(url, h, base_state):
return {"largest_ok_state_chars": lo, "largest_ok_tokens": best, "probes": probes}
def check_maxone(url, h, base_state, max_tokens, repeats):
"""The other extreme of the largest call: ONE binary question, the state filling the rest of MAX_TOKENS."""
one = [{"id": "only", "question": "Does the policy allow the exception described?", "options": YESNO}]
state, tokens = find_max_state(url, h, base_state, max_tokens, decisions=one)
over = post(url + "/decide/shared", {"state": state + " extra words", "decisions": one}, h)
runs = []
for _ in range(repeats):
s, j, ms = post(url + "/decide", {"id": "only", "state": state, **{k: one[0][k] for k in ("question", "options")}}, h)
runs.append({"status": s, "e2e_ms": round(ms, 1), "input_tokens": j.get("input_tokens"),
"code": (j.get("error") or {}).get("code"), "t_end": time.time()})
return {"state_chars": len(state), "tokens": tokens, "one_more_is": [over[0], (over[1].get("error") or {}).get("message")],
"runs": runs, "pass": all(r["status"] == 200 for r in runs) and over[0] == 422}
def check_oom(url, h, long_state):
"""The over-cap request: the largest one (maxreq) if it ran, else 16 criteria over the long state."""
before = get(url + "/health")[1]["model"]
@@ -232,6 +247,7 @@ def main():
"maxreq": lambda: check_maxreq(url, h, long_state, health["max_tokens"], health["max_decisions"],
args.repeats),
"fit": lambda: check_fit(url, h, long_state),
"maxone": lambda: check_maxone(url, h, long_state, health["max_tokens"], args.repeats),
"oom": lambda: check_oom(url, h, long_state)}[c]()
report[c]["t_start"], report[c]["t_end"] = t, time.time()
print(c, json.dumps({k: v for k, v in report[c].items() if k not in ("rows", "runs")})[:600], flush=True)