Files
esh-pfi-infrastructure/services/intern-decision-serve/acceptance/gpu3-2026-09-30/r1/checks.json
T
vh a262477a61 feat(intern-decision): stack, DNS and GPU 3 acceptance for the SemIf replacement
stacks/intern-decision: compose (GPU 1, :8033, hard VRAM cap as the single .env knob,
healthcheck, Homepage group 'AI - Eval & Retrieval'), .env.example and README.
dns: intern-decision.fv.internal -> fv-ml1 (synced to ana/esh/nh3).
acceptance on fv-ml1 GPU 3, 3 fresh processes: bit-identical to the Jev bench's native rows
(pooled 240/259, Wyrd 79/84, 0/560 flips, Δp 0), negative control 10/122/14, 0 flips across
restarts; largest accepted request 200 at a 10,134 MiB card peak under a 9.25 GiB cap; 503 and
recovery proven at a tight cap. GPU 1 deploy held: nvidia-smi Free on GPU 1 is 15,442 MiB.
2026-09-30 09:38:00 -07:00

488 lines
8.1 KiB
JSON

{
"url": "http://127.0.0.1:18033",
"started_utc": "2026-09-30T16:24:52Z",
"health": {
"status": "ok",
"model": {
"name": "Intern-Decision-4B",
"source": "internlm/Intern-Decision-4B",
"revision": "0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd",
"checkpoint": "/hf/hub/models--internlm--Intern-Decision-4B/snapshots/0e5e6aa7d6d750e2b1504ba11a8136cb58aeb3cd",
"inference_py_sha256": "c904e2c67ca0775621a22375ee373d2ba30b52117cda870c6c9ef74143b29863",
"temperature": 1.99241824,
"dtype": "bfloat16",
"attn_implementation": "sdpa",
"device": "cuda",
"max_length": 8192,
"torch_version": "2.10.0+cu128",
"transformers_version": "5.17.0",
"vision_tower": "removed",
"device_name": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition",
"allocated_gib": 7.937,
"reserved_gib": 7.969,
"max_reserved_gib": 8.861
},
"vram_cap_gib": 9.25,
"max_tokens": 8192,
"max_decisions": 64,
"max_questions_per_call": 16,
"chunking": "/decide/shared questions are packed greedily, in request order, into calls of at most 16 (1-16, 17-32, ...); each call is one prompt, so the questions in a call are asked together. With orderings, ordering k of every decision forms wave k, packed the same way.",
"workloads": []
},
"auth": {
"/decide": {
"no_token": 401,
"wrong_token": 401,
"right_token": 200
},
"/decide/shared": {
"no_token": 401,
"wrong_token": 401,
"right_token": 200
},
"health_no_token": 200,
"pass": true,
"t_start": 1790785492.2528422,
"t_end": 1790785492.3375697
},
"chunking": {
"status": [
200,
200,
200
],
"timing": {
"total_seconds": 0.15066236606799066,
"batch_size": 20,
"calls": 2,
"questions_per_call": [
16,
4
],
"input_tokens": [
1884,
1328
],
"inference_seconds": 0.13768
},
"rows": [
{
"id": "d0",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q1",
"questions": 16
},
"sha_equal": true
},
{
"id": "d1",
"top_whole": "no",
"top_split": "no",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q2",
"questions": 16
},
"sha_equal": true
},
{
"id": "d2",
"top_whole": "no",
"top_split": "no",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q3",
"questions": 16
},
"sha_equal": true
},
{
"id": "d3",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q4",
"questions": 16
},
"sha_equal": true
},
{
"id": "d4",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q5",
"questions": 16
},
"sha_equal": true
},
{
"id": "d5",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q6",
"questions": 16
},
"sha_equal": true
},
{
"id": "d6",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q7",
"questions": 16
},
"sha_equal": true
},
{
"id": "d7",
"top_whole": "no",
"top_split": "no",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q8",
"questions": 16
},
"sha_equal": true
},
{
"id": "d8",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q9",
"questions": 16
},
"sha_equal": true
},
{
"id": "d9",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q10",
"questions": 16
},
"sha_equal": true
},
{
"id": "d10",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q11",
"questions": 16
},
"sha_equal": true
},
{
"id": "d11",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q12",
"questions": 16
},
"sha_equal": true
},
{
"id": "d12",
"top_whole": "no",
"top_split": "no",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q13",
"questions": 16
},
"sha_equal": true
},
{
"id": "d13",
"top_whole": "no",
"top_split": "no",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q14",
"questions": 16
},
"sha_equal": true
},
{
"id": "d14",
"top_whole": "no",
"top_split": "no",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q15",
"questions": 16
},
"sha_equal": true
},
{
"id": "d15",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 0,
"field": "q16",
"questions": 16
},
"sha_equal": true
},
{
"id": "d16",
"top_whole": "unclear",
"top_split": "unclear",
"max_dp": 0.0,
"call_whole": {
"index": 1,
"field": "q1",
"questions": 4
},
"sha_equal": true
},
{
"id": "d17",
"top_whole": "unclear",
"top_split": "unclear",
"max_dp": 0.0,
"call_whole": {
"index": 1,
"field": "q2",
"questions": 4
},
"sha_equal": true
},
{
"id": "d18",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 1,
"field": "q3",
"questions": 4
},
"sha_equal": true
},
{
"id": "d19",
"top_whole": "yes",
"top_split": "yes",
"max_dp": 0.0,
"call_whole": {
"index": 1,
"field": "q4",
"questions": 4
},
"sha_equal": true
}
],
"e2e_ms": 157.3,
"pass": true,
"t_start": 1790785492.3376284,
"t_end": 1790785492.6581485
},
"queue": {
"sent": 48,
"max_queue": 32,
"ok": 32,
"busy_429": 16,
"codes_429": [
"busy"
],
"other": [],
"pass": true,
"t_start": 1790785492.6582093,
"t_end": 1790785499.6040177
},
"fit": {
"largest_ok_state_chars": 9508,
"largest_ok_tokens": 8191,
"probes": [
[
31950,
422,
null
],
[
16475,
422,
null
],
[
8737,
200,
8005
],
[
12606,
422,
null
],
[
10671,
422,
null
],
[
9704,
422,
null
],
[
9220,
200,
8120
],
[
9462,
200,
8176
],
[
9583,
422,
null
],
[
9522,
422,
null
],
[
9492,
200,
8185
],
[
9507,
200,
8191
],
[
9514,
422,
null
],
[
9510,
422,
null
],
[
9508,
200,
8191
],
[
9509,
422,
null
]
],
"t_start": 1790785499.6040773,
"t_end": 1790785508.5767787
},
"maxreq": {
"state_chars": 9508,
"tokens_per_call": 8191,
"decisions": 64,
"options_each": 16,
"body_bytes": 101019,
"runs": [
{
"status": 200,
"e2e_ms": 1671.9,
"code": null,
"calls": 4,
"input_tokens": [
8191,
8191,
8191,
8191
],
"reserved_gib_after": 7.969,
"max_reserved_gib": 9.25,
"t_end": 1790785520.5513089
},
{
"status": 200,
"e2e_ms": 1672.4,
"code": null,
"calls": 4,
"input_tokens": [
8191,
8191,
8191,
8191
],
"reserved_gib_after": 7.969,
"max_reserved_gib": 9.25,
"t_end": 1790785522.2301269
},
{
"status": 200,
"e2e_ms": 1675.4,
"code": null,
"calls": 4,
"input_tokens": [
8191,
8191,
8191,
8191
],
"reserved_gib_after": 7.969,
"max_reserved_gib": 9.25,
"t_end": 1790785523.9114661
}
],
"t_start": 1790785508.5768383,
"t_end": 1790785523.9116056
},
"oom": {
"oom_status": 200,
"oom_error": null,
"oom_ms": 1665.4,
"reserved_gib": {
"before": 7.969,
"after_oom": 7.969,
"after_next_request": 7.975
},
"next_request_status": 200,
"next_top": "yes",
"pass": false,
"t_start": 1790785523.9116578,
"t_end": 1790785525.6274366
},
"finished_utc": "2026-09-30T16:25:25Z"
}