Files
esh-pfi-infrastructure/services/intern-decision-serve/acceptance/gpu3-2026-09-30/oom/checks.json
T
vh a262477a61 feat(intern-decision): stack, DNS and GPU 3 acceptance for the SemIf replacement
stacks/intern-decision: compose (GPU 1, :8033, hard VRAM cap as the single .env knob,
healthcheck, Homepage group 'AI - Eval & Retrieval'), .env.example and README.
dns: intern-decision.fv.internal -> fv-ml1 (synced to ana/esh/nh3).
acceptance on fv-ml1 GPU 3, 3 fresh processes: bit-identical to the Jev bench's native rows
(pooled 240/259, Wyrd 79/84, 0/560 flips, Δp 0), negative control 10/122/14, 0 flips across
restarts; largest accepted request 200 at a 10,134 MiB card peak under a 9.25 GiB cap; 503 and
recovery proven at a tight cap. GPU 1 deploy held: nvidia-smi Free on GPU 1 is 15,442 MiB.
2026-09-30 09:38:00 -07:00

72 lines
2.6 KiB
JSON

{
"url": "http://127.0.0.1:18033",
"started_utc": "2026-09-30T16:34:04Z",
"health": {
"status": "ok",
"model": {
"name": "Intern-Decision-4B",
"source": "internlm/Intern-Decision-4B",
"revision": "0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd",
"checkpoint": "/hf/hub/models--internlm--Intern-Decision-4B/snapshots/0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd",
"inference_py_sha256": "c904e2c67ca0775621a22375ee373d2ba30b52117cda870c6c9ef74143b29863",
"temperature": 1.99241824,
"dtype": "bfloat16",
"attn_implementation": "sdpa",
"device": "cuda",
"max_length": 8192,
"torch_version": "2.10.0+cu128",
"transformers_version": "5.17.0",
"vision_tower": "removed",
"device_name": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition",
"allocated_gib": 7.937,
"reserved_gib": 7.969,
"max_reserved_gib": 8.861
},
"vram_cap_gib": 8.9,
"max_tokens": 8192,
"max_decisions": 64,
"max_questions_per_call": 16,
"chunking": "/decide/shared questions are packed greedily, in request order, into calls of at most 16 (1-16, 17-32, ...); each call is one prompt, so the questions in a call are asked together. With orderings, ordering k of every decision forms wave k, packed the same way.",
"workloads": []
},
"maxreq": {
"state_chars": 9508,
"tokens_per_call": null,
"decisions": 64,
"options_each": 16,
"body_bytes": 101019,
"runs": [
{
"status": 503,
"e2e_ms": 193.4,
"code": "out_of_memory",
"calls": null,
"input_tokens": null,
"reserved_gib_after": 7.969,
"max_reserved_gib": 8.861,
"t_end": 1790786053.86457
}
],
"t_start": 1790786044.040999,
"t_end": 1790786053.8645778
},
"oom": {
"oom_status": 503,
"oom_error": {
"code": "out_of_memory",
"message": "CUDA out of memory. Tried to allocate 64.00 MiB. GPU 0 has a total capacity of 94.97 GiB of which 85.46 GiB is free. Including non-PyTorch memory, this process has 9.50 GiB memory in use. 8.90 GiB allowed; Of the allocated memory 8.80 GiB is allocated by PyTorch, and 57.34 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)"
},
"oom_ms": 185.5,
"reserved_gib": {
"before": 7.969,
"after_oom": 7.969,
"after_next_request": 7.975
},
"next_request_status": 200,
"next_top": "yes",
"pass": true,
"t_start": 1790786053.864641,
"t_end": 1790786054.1036348
},
"finished_utc": "2026-09-30T16:34:14Z"
}