feat(intern-decision): stack, DNS and GPU 3 acceptance for the SemIf replacement
stacks/intern-decision: compose (GPU 1, :8033, hard VRAM cap as the single .env knob, healthcheck, Homepage group 'AI - Eval & Retrieval'), .env.example and README. dns: intern-decision.fv.internal -> fv-ml1 (synced to ana/esh/nh3). acceptance on fv-ml1 GPU 3, 3 fresh processes: bit-identical to the Jev bench's native rows (pooled 240/259, Wyrd 79/84, 0/560 flips, Δp 0), negative control 10/122/14, 0 flips across restarts; largest accepted request 200 at a 10,134 MiB card peak under a 9.25 GiB cap; 503 and recovery proven at a tight cap. GPU 1 deploy held: nvidia-smi Free on GPU 1 is 15,442 MiB.
This commit is contained in:
+476
@@ -0,0 +1,476 @@
|
||||
{
|
||||
"url": "http://127.0.0.1:18033",
|
||||
"started_utc": "2026-09-30T16:14:47Z",
|
||||
"health": {
|
||||
"status": "ok",
|
||||
"model": {
|
||||
"name": "Intern-Decision-4B",
|
||||
"source": "internlm/Intern-Decision-4B",
|
||||
"revision": "0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd",
|
||||
"checkpoint": "/hf/hub/models--internlm--Intern-Decision-4B/snapshots/0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd",
|
||||
"inference_py_sha256": "c904e2c67ca0775621a22375ee373d2ba30b52117cda870c6c9ef74143b29863",
|
||||
"temperature": 1.99241824,
|
||||
"dtype": "bfloat16",
|
||||
"attn_implementation": "sdpa",
|
||||
"device": "cuda",
|
||||
"max_length": 8192,
|
||||
"torch_version": "2.10.0+cu128",
|
||||
"transformers_version": "5.17.0",
|
||||
"vision_tower": "removed",
|
||||
"device_name": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition",
|
||||
"allocated_gib": 7.944,
|
||||
"reserved_gib": 7.969,
|
||||
"max_reserved_gib": 8.861
|
||||
},
|
||||
"vram_cap_gib": 9.25,
|
||||
"max_tokens": 8192,
|
||||
"max_decisions": 64,
|
||||
"max_questions_per_call": 16,
|
||||
"chunking": "/decide/shared questions are packed greedily, in request order, into calls of at most 16 (1-16, 17-32, ...); each call is one prompt, so the questions in a call are asked together. With orderings, ordering k of every decision forms wave k, packed the same way.",
|
||||
"workloads": []
|
||||
},
|
||||
"auth": {
|
||||
"/decide": {
|
||||
"no_token": 401,
|
||||
"wrong_token": 401,
|
||||
"right_token": 200
|
||||
},
|
||||
"/decide/shared": {
|
||||
"no_token": 401,
|
||||
"wrong_token": 401,
|
||||
"right_token": 200
|
||||
},
|
||||
"health_no_token": 200,
|
||||
"pass": true,
|
||||
"t_start": 1790784887.4892507,
|
||||
"t_end": 1790784887.5748613
|
||||
},
|
||||
"chunking": {
|
||||
"status": [
|
||||
200,
|
||||
200,
|
||||
200
|
||||
],
|
||||
"timing": {
|
||||
"total_seconds": 0.1498984640929848,
|
||||
"batch_size": 20,
|
||||
"calls": 2,
|
||||
"questions_per_call": [
|
||||
16,
|
||||
4
|
||||
],
|
||||
"input_tokens": [
|
||||
1884,
|
||||
1328
|
||||
],
|
||||
"inference_seconds": 0.1373
|
||||
},
|
||||
"rows": [
|
||||
{
|
||||
"id": "d0",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q1",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d1",
|
||||
"top_whole": "no",
|
||||
"top_split": "no",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q2",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d2",
|
||||
"top_whole": "no",
|
||||
"top_split": "no",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q3",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d3",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q4",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d4",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q5",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d5",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q6",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d6",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q7",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d7",
|
||||
"top_whole": "no",
|
||||
"top_split": "no",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q8",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d8",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q9",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d9",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q10",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d10",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q11",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d11",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q12",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d12",
|
||||
"top_whole": "no",
|
||||
"top_split": "no",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q13",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d13",
|
||||
"top_whole": "no",
|
||||
"top_split": "no",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q14",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d14",
|
||||
"top_whole": "no",
|
||||
"top_split": "no",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q15",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d15",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 0,
|
||||
"field": "q16",
|
||||
"questions": 16
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d16",
|
||||
"top_whole": "unclear",
|
||||
"top_split": "unclear",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 1,
|
||||
"field": "q1",
|
||||
"questions": 4
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d17",
|
||||
"top_whole": "unclear",
|
||||
"top_split": "unclear",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 1,
|
||||
"field": "q2",
|
||||
"questions": 4
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d18",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 1,
|
||||
"field": "q3",
|
||||
"questions": 4
|
||||
},
|
||||
"sha_equal": true
|
||||
},
|
||||
{
|
||||
"id": "d19",
|
||||
"top_whole": "yes",
|
||||
"top_split": "yes",
|
||||
"max_dp": 0.0,
|
||||
"call_whole": {
|
||||
"index": 1,
|
||||
"field": "q4",
|
||||
"questions": 4
|
||||
},
|
||||
"sha_equal": true
|
||||
}
|
||||
],
|
||||
"e2e_ms": 157.0,
|
||||
"pass": true,
|
||||
"t_start": 1790784887.5748994,
|
||||
"t_end": 1790784887.894067
|
||||
},
|
||||
"queue": {
|
||||
"sent": 48,
|
||||
"max_queue": 32,
|
||||
"ok": 32,
|
||||
"busy_429": 16,
|
||||
"codes_429": [
|
||||
"busy"
|
||||
],
|
||||
"other": [],
|
||||
"pass": true,
|
||||
"t_start": 1790784887.8941243,
|
||||
"t_end": 1790784895.1853623
|
||||
},
|
||||
"fit": {
|
||||
"largest_ok_state_chars": 5514,
|
||||
"largest_ok_tokens": 6826,
|
||||
"probes": [
|
||||
[
|
||||
31950,
|
||||
422,
|
||||
null
|
||||
],
|
||||
[
|
||||
16475,
|
||||
422,
|
||||
null
|
||||
],
|
||||
[
|
||||
8737,
|
||||
503,
|
||||
null
|
||||
],
|
||||
[
|
||||
4868,
|
||||
200,
|
||||
6692
|
||||
],
|
||||
[
|
||||
6802,
|
||||
503,
|
||||
null
|
||||
],
|
||||
[
|
||||
5835,
|
||||
503,
|
||||
null
|
||||
],
|
||||
[
|
||||
5351,
|
||||
200,
|
||||
6785
|
||||
],
|
||||
[
|
||||
5593,
|
||||
503,
|
||||
null
|
||||
],
|
||||
[
|
||||
5472,
|
||||
200,
|
||||
6818
|
||||
],
|
||||
[
|
||||
5532,
|
||||
503,
|
||||
null
|
||||
],
|
||||
[
|
||||
5502,
|
||||
200,
|
||||
6824
|
||||
],
|
||||
[
|
||||
5517,
|
||||
503,
|
||||
null
|
||||
],
|
||||
[
|
||||
5509,
|
||||
200,
|
||||
6825
|
||||
],
|
||||
[
|
||||
5513,
|
||||
200,
|
||||
6826
|
||||
],
|
||||
[
|
||||
5515,
|
||||
503,
|
||||
null
|
||||
],
|
||||
[
|
||||
5514,
|
||||
200,
|
||||
6826
|
||||
]
|
||||
],
|
||||
"t_start": 1790784895.1854463,
|
||||
"t_end": 1790784905.1581368
|
||||
},
|
||||
"maxreq": {
|
||||
"state_chars": 11337,
|
||||
"tokens_per_call": null,
|
||||
"decisions": 64,
|
||||
"options_each": 16,
|
||||
"body_bytes": 100953,
|
||||
"runs": [
|
||||
{
|
||||
"status": 503,
|
||||
"e2e_ms": 188.3,
|
||||
"code": "out_of_memory",
|
||||
"calls": null,
|
||||
"input_tokens": null,
|
||||
"reserved_gib_after": 8.262,
|
||||
"max_reserved_gib": 9.248,
|
||||
"t_end": 1790784906.9150267
|
||||
},
|
||||
{
|
||||
"status": 503,
|
||||
"e2e_ms": 186.1,
|
||||
"code": "out_of_memory",
|
||||
"calls": null,
|
||||
"input_tokens": null,
|
||||
"reserved_gib_after": 8.262,
|
||||
"max_reserved_gib": 9.248,
|
||||
"t_end": 1790784907.104969
|
||||
},
|
||||
{
|
||||
"status": 503,
|
||||
"e2e_ms": 198.2,
|
||||
"code": "out_of_memory",
|
||||
"calls": null,
|
||||
"input_tokens": null,
|
||||
"reserved_gib_after": 8.262,
|
||||
"max_reserved_gib": 9.248,
|
||||
"t_end": 1790784907.306796
|
||||
}
|
||||
],
|
||||
"t_start": 1790784905.1582026,
|
||||
"t_end": 1790784907.3068023
|
||||
},
|
||||
"oom": {
|
||||
"oom_status": 503,
|
||||
"oom_error": {
|
||||
"code": "out_of_memory",
|
||||
"message": "CUDA out of memory. Tried to allocate 64.00 MiB. GPU 0 has a total capacity of 94.97 GiB of which 84.85 GiB is free. Including non-PyTorch memory, this process has 10.11 GiB memory in use. 9.25 GiB allowed; Of the allocated memory 9.12 GiB is allocated by PyTorch, and 97.33 MiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)"
|
||||
},
|
||||
"oom_ms": 197.6,
|
||||
"reserved_gib": {
|
||||
"before": 8.262,
|
||||
"after_oom": 8.262,
|
||||
"after_next_request": 8.287
|
||||
},
|
||||
"next_request_status": 200,
|
||||
"next_top": "yes",
|
||||
"pass": true,
|
||||
"t_start": 1790784907.3068488,
|
||||
"t_end": 1790784907.554095
|
||||
},
|
||||
"finished_utc": "2026-09-30T16:15:07Z"
|
||||
}
|
||||
+1
@@ -0,0 +1 @@
|
||||
35bbfe41a4b1b3227c24caa36199780398e09966ceb273fbe9402fb1631d6782
|
||||
@@ -0,0 +1,5 @@
|
||||
1790784749.283255359 sets:start
|
||||
1790784842.311562084 sets:done
|
||||
1790784887.437615498 shape:done
|
||||
1790784907.572533137 checks:done
|
||||
gpu3 after: 2 MiB
|
||||
+1
@@ -0,0 +1 @@
|
||||
{"status":"ok","model":{"name":"Intern-Decision-4B","source":"internlm/Intern-Decision-4B","revision":"0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd","checkpoint":"/hf/hub/models--internlm--Intern-Decision-4B/snapshots/0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd","inference_py_sha256":"c904e2c67ca0775621a22375ee373d2ba30b52117cda870c6c9ef74143b29863","temperature":1.99241824,"dtype":"bfloat16","attn_implementation":"sdpa","device":"cuda","max_length":8192,"torch_version":"2.10.0+cu128","transformers_version":"5.17.0","vision_tower":"removed","device_name":"NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition","allocated_gib":8.19,"reserved_gib":8.287,"max_reserved_gib":9.248},"vram_cap_gib":9.25,"max_tokens":8192,"max_decisions":64,"max_questions_per_call":16,"chunking":"/decide/shared questions are packed greedily, in request order, into calls of at most 16 (1-16, 17-32, ...); each call is one prompt, so the questions in a call are asked together. With orderings, ordering k of every decision forms wave k, packed the same way.","workloads":[]}
|
||||
+1
@@ -0,0 +1 @@
|
||||
{"status":"ok","model":{"name":"Intern-Decision-4B","source":"internlm/Intern-Decision-4B","revision":"0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd","checkpoint":"/hf/hub/models--internlm--Intern-Decision-4B/snapshots/0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd","inference_py_sha256":"c904e2c67ca0775621a22375ee373d2ba30b52117cda870c6c9ef74143b29863","temperature":1.99241824,"dtype":"bfloat16","attn_implementation":"sdpa","device":"cuda","max_length":8192,"torch_version":"2.10.0+cu128","transformers_version":"5.17.0","vision_tower":"removed","device_name":"NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition","allocated_gib":7.937,"reserved_gib":7.969,"max_reserved_gib":8.861},"vram_cap_gib":9.25,"max_tokens":8192,"max_decisions":64,"max_questions_per_call":16,"chunking":"/decide/shared questions are packed greedily, in request order, into calls of at most 16 (1-16, 17-32, ...); each call is one prompt, so the questions in a call are asked together. With orderings, ordering k of every decision forms wave k, packed the same way.","workloads":[]}
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+1490
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user