Files
esh-pfi-infrastructure/services/intern-decision-serve/acceptance/gpu1-live/checks-live.json
T
vh 1cf763a7b1 docs(intern-decision): live on fv-ml1 GPU 1; semif marked REPLACED
intern-decision deployed 0941 PT (cap 9.0 GiB, MAX_TOKENS 7168). Live acceptance: positive
control 240/259 and Wyrd 79/84, bit-identical to the bench (0/560 rows, Δp 0); negative control
10/122/14; largest accepted requests 200 with no 503; per-process 8,812 MiB at rest and 9,866 peak;
GPU 1 Free 15,442 before and 6,581 after (lowest 5,569 under load). Latency from nh3-dev:
21 criteria 114 ms, 16 over ~3,900 tokens 238 ms. semif README banner now REPLACED with the
rollback; fv-ml1 GPU 1 note updated.
2026-09-30 09:48:30 -07:00

139 lines
3.1 KiB
JSON

{
"url": "http://intern-decision.fv.internal:8033",
"started_utc": "2026-09-30T16:43:39Z",
"health": {
"status": "ok",
"model": {
"name": "Intern-Decision-4B",
"source": "internlm/Intern-Decision-4B",
"revision": "0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd",
"checkpoint": "/hf/hub/models--internlm--Intern-Decision-4B/snapshots/0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd",
"inference_py_sha256": "c904e2c67ca0775621a22375ee373d2ba30b52117cda870c6c9ef74143b29863",
"temperature": 1.99241824,
"dtype": "bfloat16",
"attn_implementation": "sdpa",
"device": "cuda",
"max_length": 7168,
"torch_version": "2.10.0+cu128",
"transformers_version": "5.17.0",
"vision_tower": "removed",
"device_name": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition",
"allocated_gib": 7.937,
"reserved_gib": 7.969,
"max_reserved_gib": 8.861
},
"vram_cap_gib": 9.0,
"max_tokens": 7168,
"max_decisions": 64,
"max_questions_per_call": 16,
"chunking": "/decide/shared questions are packed greedily, in request order, into calls of at most 16 (1-16, 17-32, ...); each call is one prompt, so the questions in a call are asked together. With orderings, ordering k of every decision forms wave k, packed the same way.",
"workloads": []
},
"auth": {
"/decide": {
"no_token": 401,
"wrong_token": 401,
"right_token": 200
},
"/decide/shared": {
"no_token": 401,
"wrong_token": 401,
"right_token": 200
},
"health_no_token": 200,
"pass": true,
"t_start": 1790786619.1466322,
"t_end": 1790786619.3572643
},
"maxreq": {
"state_chars": 4793,
"tokens_per_call": 7168,
"decisions": 64,
"options_each": 16,
"body_bytes": 96250,
"runs": [
{
"status": 200,
"e2e_ms": 1542.5,
"code": null,
"calls": 4,
"input_tokens": [
7168,
7168,
7168,
7168
],
"reserved_gib_after": 7.969,
"max_reserved_gib": 8.996,
"t_end": 1790786706.9185867
},
{
"status": 200,
"e2e_ms": 1518.7,
"code": null,
"calls": 4,
"input_tokens": [
7168,
7168,
7168,
7168
],
"reserved_gib_after": 7.969,
"max_reserved_gib": 8.996,
"t_end": 1790786708.4557755
},
{
"status": 200,
"e2e_ms": 1539.6,
"code": null,
"calls": 4,
"input_tokens": [
7168,
7168,
7168,
7168
],
"reserved_gib_after": 7.969,
"max_reserved_gib": 8.996,
"t_end": 1790786710.014339
}
],
"t_start": 1790786619.357301,
"t_end": 1790786710.014397
},
"maxone": {
"state_chars": 28727,
"tokens": 7168,
"one_more_is": [
422,
"Example has 7171 tokens, above 7168; truncation is forbidden"
],
"runs": [
{
"status": 200,
"e2e_ms": 394.2,
"input_tokens": 7168,
"code": null,
"t_end": 1790786715.324555
},
{
"status": 200,
"e2e_ms": 375.9,
"input_tokens": 7168,
"code": null,
"t_end": 1790786715.7005756
},
{
"status": 200,
"e2e_ms": 372.7,
"input_tokens": 7168,
"code": null,
"t_end": 1790786716.0734687
}
],
"pass": true,
"t_start": 1790786710.0144272,
"t_end": 1790786716.0735004
},
"finished_utc": "2026-09-30T16:45:16Z"
}