{ "url": "http://127.0.0.1:18033", "started_utc": "2026-09-30T16:34:04Z", "health": { "status": "ok", "model": { "name": "Intern-Decision-4B", "source": "internlm/Intern-Decision-4B", "revision": "0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd", "checkpoint": "/hf/hub/models--internlm--Intern-Decision-4B/snapshots/0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd", "inference_py_sha256": "c904e2c67ca0775621a22375ee373d2ba30b52117cda870c6c9ef74143b29863", "temperature": 1.99241824, "dtype": "bfloat16", "attn_implementation": "sdpa", "device": "cuda", "max_length": 8192, "torch_version": "2.10.0+cu128", "transformers_version": "5.17.0", "vision_tower": "removed", "device_name": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition", "allocated_gib": 7.937, "reserved_gib": 7.969, "max_reserved_gib": 8.861 }, "vram_cap_gib": 8.9, "max_tokens": 8192, "max_decisions": 64, "max_questions_per_call": 16, "chunking": "/decide/shared questions are packed greedily, in request order, into calls of at most 16 (1-16, 17-32, ...); each call is one prompt, so the questions in a call are asked together. With orderings, ordering k of every decision forms wave k, packed the same way.", "workloads": [] }, "maxreq": { "state_chars": 9508, "tokens_per_call": null, "decisions": 64, "options_each": 16, "body_bytes": 101019, "runs": [ { "status": 503, "e2e_ms": 193.4, "code": "out_of_memory", "calls": null, "input_tokens": null, "reserved_gib_after": 7.969, "max_reserved_gib": 8.861, "t_end": 1790786053.86457 } ], "t_start": 1790786044.040999, "t_end": 1790786053.8645778 }, "oom": { "oom_status": 503, "oom_error": { "code": "out_of_memory", "message": "CUDA out of memory. Tried to allocate 64.00 MiB. GPU 0 has a total capacity of 94.97 GiB of which 85.46 GiB is free. Including non-PyTorch memory, this process has 9.50 GiB memory in use. 8.90 GiB allowed; Of the allocated memory 8.80 GiB is allocated by PyTorch, and 57.34 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)" }, "oom_ms": 185.5, "reserved_gib": { "before": 7.969, "after_oom": 7.969, "after_next_request": 7.975 }, "next_request_status": 200, "next_top": "yes", "pass": true, "t_start": 1790786053.864641, "t_end": 1790786054.1036348 }, "finished_utc": "2026-09-30T16:34:14Z" }