feat(intern-decision-serve): 0.1.1 adds POST /v1/systemone (Jev wire shape)

Straight passthrough to the checkpoint's own DecisionEngine.predict — never the
semif mapping, whose different prompt would change the answers. Reuses the one
inference thread, bearer auth, MAX_QUEUE, VRAM cap and error envelope; no new
concurrency. 1..16 questions in ONE call (never chunked: Jev questions share a
prompt); images 422; over MAX_TOKENS 422 before the forward. /health advertises
the surface. Response 'model' is a string name@revision (JevBench's runner
hashes it; a dict broke its manifest step).

Acceptance on the live service (see acceptance/systemone-2026-09-30/): JevBench
v1.2.16 typesafe adapter over the 231 public items scores all 202/231, hard
83/111, with 0 changed answers across all 924 rows of the bench's own r1..r4;
controls 401/422x3 (token boundary proven at 7168 pass / 7169 refuse); GPU 1
per-process peak 9,866 MiB under the largest accepted request (budget 9,876);
/decide/shared positive control unchanged. 120 tests green.
This commit is contained in:
vh
2026-09-30 12:58:19 -07:00
parent 579b1f5d42
commit ff552abf1f
13 changed files with 3449 additions and 13 deletions
@@ -68,6 +68,19 @@ class SharedBody(BaseModel):
workload: str | None = None
class SystemOneBody(BaseModel):
"""The Jev request. `model` is accepted with any value and ignored (Jev clients send
'jev-latest'); `images` is kept only to refuse it with a clear 422. Extra keys are dropped:
only state+questions reach predict, exactly like the bench's 60-line wrapper."""
model_config = ConfigDict(extra="allow")
state: State
questions: dict[str, dict[str, Any]]
@property
def images_present(self) -> bool:
return "images" in (self.model_extra or {})
class ApiError(Exception):
def __init__(self, status: int, code: str, message: str):
super().__init__(message)
@@ -322,10 +335,11 @@ def create_app(settings: Settings, engine: Any, executor: ThreadPoolExecutor | N
"inference_seconds": inference_ms / 1000}
return out, timing
async def score(state: State, decisions: list[Decision], waves: list[list[Slot]]):
"""Run one request's calls on the inference thread; map their failures to contract codes."""
async def on_inference(fn):
"""Run fn on the ONE inference thread (INV-2) and map failures to contract codes (INV-4):
the shared dispatch+mapping for every surface, semif and systemone alike."""
try:
return await asyncio.get_running_loop().run_in_executor(executor, run, state, decisions, waves)
return await asyncio.get_running_loop().run_in_executor(executor, fn)
except ValueError as exc: # the model's own validation, token limit, ...
raise ApiError(422, "invalid_request", str(exc)) from exc
except OutOfMemory as exc:
@@ -333,11 +347,17 @@ def create_app(settings: Settings, engine: Any, executor: ThreadPoolExecutor | N
except Exception as exc: # noqa: BLE001 — any other failure, building the response included
raise ApiError(500, "scoring_failed", f"{type(exc).__name__}: {exc}") from exc
async def score(state: State, decisions: list[Decision], waves: list[list[Slot]]):
return await on_inference(lambda: run(state, decisions, waves))
@app.get("/health")
async def health():
return {"status": "ok", "model": engine.health(), "vram_cap_gib": settings.vram_cap_gib,
"max_tokens": settings.max_tokens, "max_decisions": settings.max_decisions,
"max_questions_per_call": MAX_QUESTIONS_PER_CALL, "chunking": CHUNKING, "workloads": []}
"max_questions_per_call": MAX_QUESTIONS_PER_CALL, "chunking": CHUNKING, "workloads": [],
"endpoints": ["/decide", "/decide/shared", "/v1/systemone"],
"systemone": {"max_questions": MAX_QUESTIONS_PER_CALL, "chunking": "none",
"images": "not supported", "max_tokens": settings.max_tokens}}
@app.post("/decide")
async def decide(request: Request):
@@ -363,4 +383,43 @@ def create_app(settings: Settings, engine: Any, executor: ThreadPoolExecutor | N
finally:
leave()
@app.post("/v1/systemone")
async def systemone(request: Request):
"""Jev wire shape. Straight passthrough to DecisionEngine.predict — NEVER the semif
mapping (a different prompt would change the answers). Contract § POST /v1/systemone."""
admit() # before the body is read
try:
payload = await parse(request, SystemOneBody)
if payload.images_present:
raise ApiError(422, "invalid_request", "images not supported (the vision tower is "
"removed to fit the GPU 1 budget, INV-7)")
questions = payload.questions
if not 1 <= len(questions) <= MAX_QUESTIONS_PER_CALL:
raise ApiError(422, "invalid_request",
f"this request has {len(questions)} questions; the limit is "
f"1..{MAX_QUESTIONS_PER_CALL} in one call — questions of one call "
"share one prompt, so they are never chunked")
response = await run_systemone({"state": payload.state, "questions": questions})
meta = engine.metadata
return {"answers": response["answers"], "usage": response["usage"],
"model": f"{meta.get('name')}@{meta.get('revision')}", # string: see contract
**{k: v for k, v in response.items() if k not in ("answers", "usage", "model")}}
finally:
leave()
async def run_systemone(call: dict) -> dict:
"""One native call on the inference thread, under the same lock and mapping as the semif
surface (on_inference). The engine's own validate_request enforces the per-question Jev
rules before any GPU work; its ValueError is a 422 like the token-limit check."""
def one():
with inference:
response, _sha = engine.predict(call)
usage = response.get("usage")
response["usage"] = usage if isinstance(usage, dict) else {} # contract: {} not absent
return response
response = await on_inference(one)
if not isinstance(response.get("answers"), dict):
raise ApiError(500, "scoring_failed", "the engine returned no answers object")
return response
return app