{ "url": "http://127.0.0.1:18033", "started_utc": "2026-09-30T16:14:47Z", "health": { "status": "ok", "model": { "name": "Intern-Decision-4B", "source": "internlm/Intern-Decision-4B", "revision": "0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd", "checkpoint": "/hf/hub/models--internlm--Intern-Decision-4B/snapshots/0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd", "inference_py_sha256": "c904e2c67ca0775621a22375ee373d2ba30b52117cda870c6c9ef74143b29863", "temperature": 1.99241824, "dtype": "bfloat16", "attn_implementation": "sdpa", "device": "cuda", "max_length": 8192, "torch_version": "2.10.0+cu128", "transformers_version": "5.17.0", "vision_tower": "removed", "device_name": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition", "allocated_gib": 7.944, "reserved_gib": 7.969, "max_reserved_gib": 8.861 }, "vram_cap_gib": 9.25, "max_tokens": 8192, "max_decisions": 64, "max_questions_per_call": 16, "chunking": "/decide/shared questions are packed greedily, in request order, into calls of at most 16 (1-16, 17-32, ...); each call is one prompt, so the questions in a call are asked together. With orderings, ordering k of every decision forms wave k, packed the same way.", "workloads": [] }, "auth": { "/decide": { "no_token": 401, "wrong_token": 401, "right_token": 200 }, "/decide/shared": { "no_token": 401, "wrong_token": 401, "right_token": 200 }, "health_no_token": 200, "pass": true, "t_start": 1790784887.4892507, "t_end": 1790784887.5748613 }, "chunking": { "status": [ 200, 200, 200 ], "timing": { "total_seconds": 0.1498984640929848, "batch_size": 20, "calls": 2, "questions_per_call": [ 16, 4 ], "input_tokens": [ 1884, 1328 ], "inference_seconds": 0.1373 }, "rows": [ { "id": "d0", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q1", "questions": 16 }, "sha_equal": true }, { "id": "d1", "top_whole": "no", "top_split": "no", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q2", "questions": 16 }, "sha_equal": true }, { "id": "d2", "top_whole": "no", "top_split": "no", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q3", "questions": 16 }, "sha_equal": true }, { "id": "d3", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q4", "questions": 16 }, "sha_equal": true }, { "id": "d4", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q5", "questions": 16 }, "sha_equal": true }, { "id": "d5", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q6", "questions": 16 }, "sha_equal": true }, { "id": "d6", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q7", "questions": 16 }, "sha_equal": true }, { "id": "d7", "top_whole": "no", "top_split": "no", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q8", "questions": 16 }, "sha_equal": true }, { "id": "d8", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q9", "questions": 16 }, "sha_equal": true }, { "id": "d9", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q10", "questions": 16 }, "sha_equal": true }, { "id": "d10", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q11", "questions": 16 }, "sha_equal": true }, { "id": "d11", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q12", "questions": 16 }, "sha_equal": true }, { "id": "d12", "top_whole": "no", "top_split": "no", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q13", "questions": 16 }, "sha_equal": true }, { "id": "d13", "top_whole": "no", "top_split": "no", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q14", "questions": 16 }, "sha_equal": true }, { "id": "d14", "top_whole": "no", "top_split": "no", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q15", "questions": 16 }, "sha_equal": true }, { "id": "d15", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 0, "field": "q16", "questions": 16 }, "sha_equal": true }, { "id": "d16", "top_whole": "unclear", "top_split": "unclear", "max_dp": 0.0, "call_whole": { "index": 1, "field": "q1", "questions": 4 }, "sha_equal": true }, { "id": "d17", "top_whole": "unclear", "top_split": "unclear", "max_dp": 0.0, "call_whole": { "index": 1, "field": "q2", "questions": 4 }, "sha_equal": true }, { "id": "d18", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 1, "field": "q3", "questions": 4 }, "sha_equal": true }, { "id": "d19", "top_whole": "yes", "top_split": "yes", "max_dp": 0.0, "call_whole": { "index": 1, "field": "q4", "questions": 4 }, "sha_equal": true } ], "e2e_ms": 157.0, "pass": true, "t_start": 1790784887.5748994, "t_end": 1790784887.894067 }, "queue": { "sent": 48, "max_queue": 32, "ok": 32, "busy_429": 16, "codes_429": [ "busy" ], "other": [], "pass": true, "t_start": 1790784887.8941243, "t_end": 1790784895.1853623 }, "fit": { "largest_ok_state_chars": 5514, "largest_ok_tokens": 6826, "probes": [ [ 31950, 422, null ], [ 16475, 422, null ], [ 8737, 503, null ], [ 4868, 200, 6692 ], [ 6802, 503, null ], [ 5835, 503, null ], [ 5351, 200, 6785 ], [ 5593, 503, null ], [ 5472, 200, 6818 ], [ 5532, 503, null ], [ 5502, 200, 6824 ], [ 5517, 503, null ], [ 5509, 200, 6825 ], [ 5513, 200, 6826 ], [ 5515, 503, null ], [ 5514, 200, 6826 ] ], "t_start": 1790784895.1854463, "t_end": 1790784905.1581368 }, "maxreq": { "state_chars": 11337, "tokens_per_call": null, "decisions": 64, "options_each": 16, "body_bytes": 100953, "runs": [ { "status": 503, "e2e_ms": 188.3, "code": "out_of_memory", "calls": null, "input_tokens": null, "reserved_gib_after": 8.262, "max_reserved_gib": 9.248, "t_end": 1790784906.9150267 }, { "status": 503, "e2e_ms": 186.1, "code": "out_of_memory", "calls": null, "input_tokens": null, "reserved_gib_after": 8.262, "max_reserved_gib": 9.248, "t_end": 1790784907.104969 }, { "status": 503, "e2e_ms": 198.2, "code": "out_of_memory", "calls": null, "input_tokens": null, "reserved_gib_after": 8.262, "max_reserved_gib": 9.248, "t_end": 1790784907.306796 } ], "t_start": 1790784905.1582026, "t_end": 1790784907.3068023 }, "oom": { "oom_status": 503, "oom_error": { "code": "out_of_memory", "message": "CUDA out of memory. Tried to allocate 64.00 MiB. GPU 0 has a total capacity of 94.97 GiB of which 84.85 GiB is free. Including non-PyTorch memory, this process has 10.11 GiB memory in use. 9.25 GiB allowed; Of the allocated memory 9.12 GiB is allocated by PyTorch, and 97.33 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)" }, "oom_ms": 197.6, "reserved_gib": { "before": 8.262, "after_oom": 8.262, "after_next_request": 8.287 }, "next_request_status": 200, "next_top": "yes", "pass": true, "t_start": 1790784907.3068488, "t_end": 1790784907.554095 }, "finished_utc": "2026-09-30T16:15:07Z" }