stacks/intern-decision: compose (GPU 1, :8033, hard VRAM cap as the single .env knob, healthcheck, Homepage group 'AI - Eval & Retrieval'), .env.example and README. dns: intern-decision.fv.internal -> fv-ml1 (synced to ana/esh/nh3). acceptance on fv-ml1 GPU 3, 3 fresh processes: bit-identical to the Jev bench's native rows (pooled 240/259, Wyrd 79/84, 0/560 flips, Δp 0), negative control 10/122/14, 0 flips across restarts; largest accepted request 200 at a 10,134 MiB card peak under a 9.25 GiB cap; 503 and recovery proven at a tight cap. GPU 1 deploy held: nvidia-smi Free on GPU 1 is 15,442 MiB.
357 lines
6.8 KiB
JSON
357 lines
6.8 KiB
JSON
{
|
|
"url": "http://127.0.0.1:18033",
|
|
"started_utc": "2026-09-30T16:09:34Z",
|
|
"health": {
|
|
"status": "ok",
|
|
"model": {
|
|
"name": "Intern-Decision-4B",
|
|
"source": "internlm/Intern-Decision-4B",
|
|
"revision": "0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd",
|
|
"checkpoint": "/hf/hub/models--internlm--Intern-Decision-4B/snapshots/0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd",
|
|
"inference_py_sha256": "c904e2c67ca0775621a22375ee373d2ba30b52117cda870c6c9ef74143b29863",
|
|
"temperature": 1.99241824,
|
|
"dtype": "bfloat16",
|
|
"attn_implementation": "sdpa",
|
|
"device": "cuda",
|
|
"max_length": 8192,
|
|
"torch_version": "2.10.0+cu128",
|
|
"transformers_version": "5.17.0",
|
|
"vision_tower": "removed",
|
|
"device_name": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition",
|
|
"allocated_gib": 7.937,
|
|
"reserved_gib": 7.969,
|
|
"max_reserved_gib": 8.861
|
|
},
|
|
"vram_cap_gib": null,
|
|
"max_tokens": 8192,
|
|
"max_decisions": 64,
|
|
"max_questions_per_call": 16,
|
|
"chunking": "/decide/shared questions are packed greedily, in request order, into calls of at most 16 (1-16, 17-32, ...); each call is one prompt, so the questions in a call are asked together. With orderings, ordering k of every decision forms wave k, packed the same way.",
|
|
"workloads": []
|
|
},
|
|
"auth": {
|
|
"/decide": {
|
|
"no_token": 401,
|
|
"wrong_token": 401,
|
|
"right_token": 200
|
|
},
|
|
"/decide/shared": {
|
|
"no_token": 401,
|
|
"wrong_token": 401,
|
|
"right_token": 200
|
|
},
|
|
"health_no_token": 200,
|
|
"pass": true,
|
|
"t_start": 1790784574.0391982,
|
|
"t_end": 1790784574.1316206
|
|
},
|
|
"chunking": {
|
|
"status": [
|
|
200,
|
|
200,
|
|
200
|
|
],
|
|
"timing": {
|
|
"total_seconds": 0.14785360009409487,
|
|
"batch_size": 20,
|
|
"calls": 2,
|
|
"questions_per_call": [
|
|
16,
|
|
4
|
|
],
|
|
"input_tokens": [
|
|
1884,
|
|
1328
|
|
],
|
|
"inference_seconds": 0.13485
|
|
},
|
|
"rows": [
|
|
{
|
|
"id": "d0",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q1",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d1",
|
|
"top_whole": "no",
|
|
"top_split": "no",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q2",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d2",
|
|
"top_whole": "no",
|
|
"top_split": "no",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q3",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d3",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q4",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d4",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q5",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d5",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q6",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d6",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q7",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d7",
|
|
"top_whole": "no",
|
|
"top_split": "no",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q8",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d8",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q9",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d9",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q10",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d10",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q11",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d11",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q12",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d12",
|
|
"top_whole": "no",
|
|
"top_split": "no",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q13",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d13",
|
|
"top_whole": "no",
|
|
"top_split": "no",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q14",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d14",
|
|
"top_whole": "no",
|
|
"top_split": "no",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q15",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d15",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 0,
|
|
"field": "q16",
|
|
"questions": 16
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d16",
|
|
"top_whole": "unclear",
|
|
"top_split": "unclear",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 1,
|
|
"field": "q1",
|
|
"questions": 4
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d17",
|
|
"top_whole": "unclear",
|
|
"top_split": "unclear",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 1,
|
|
"field": "q2",
|
|
"questions": 4
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d18",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 1,
|
|
"field": "q3",
|
|
"questions": 4
|
|
},
|
|
"sha_equal": true
|
|
},
|
|
{
|
|
"id": "d19",
|
|
"top_whole": "yes",
|
|
"top_split": "yes",
|
|
"max_dp": 0.0,
|
|
"call_whole": {
|
|
"index": 1,
|
|
"field": "q4",
|
|
"questions": 4
|
|
},
|
|
"sha_equal": true
|
|
}
|
|
],
|
|
"e2e_ms": 151.3,
|
|
"pass": true,
|
|
"t_start": 1790784574.1316838,
|
|
"t_end": 1790784574.4364655
|
|
},
|
|
"maxreq": {
|
|
"state_chars": 11337,
|
|
"tokens_per_call": 8192,
|
|
"decisions": 64,
|
|
"options_each": 16,
|
|
"body_bytes": 100953,
|
|
"runs": [
|
|
{
|
|
"status": 422,
|
|
"e2e_ms": 415.8,
|
|
"code": "invalid_request",
|
|
"calls": null,
|
|
"input_tokens": null,
|
|
"reserved_gib_after": 7.969,
|
|
"max_reserved_gib": 9.557,
|
|
"t_end": 1790784583.8459144
|
|
},
|
|
{
|
|
"status": 422,
|
|
"e2e_ms": 405.2,
|
|
"code": "invalid_request",
|
|
"calls": null,
|
|
"input_tokens": null,
|
|
"reserved_gib_after": 7.969,
|
|
"max_reserved_gib": 9.557,
|
|
"t_end": 1790784584.2551143
|
|
},
|
|
{
|
|
"status": 422,
|
|
"e2e_ms": 406.5,
|
|
"code": "invalid_request",
|
|
"calls": null,
|
|
"input_tokens": null,
|
|
"reserved_gib_after": 7.969,
|
|
"max_reserved_gib": 9.557,
|
|
"t_end": 1790784584.6652017
|
|
}
|
|
],
|
|
"t_start": 1790784574.4365265,
|
|
"t_end": 1790784584.6653428
|
|
},
|
|
"finished_utc": "2026-09-30T16:09:44Z"
|
|
} |