feat(intern-decision-serve): 0.1.1 adds POST /v1/systemone (Jev wire shape)
Straight passthrough to the checkpoint's own DecisionEngine.predict — never the semif mapping, whose different prompt would change the answers. Reuses the one inference thread, bearer auth, MAX_QUEUE, VRAM cap and error envelope; no new concurrency. 1..16 questions in ONE call (never chunked: Jev questions share a prompt); images 422; over MAX_TOKENS 422 before the forward. /health advertises the surface. Response 'model' is a string name@revision (JevBench's runner hashes it; a dict broke its manifest step). Acceptance on the live service (see acceptance/systemone-2026-09-30/): JevBench v1.2.16 typesafe adapter over the 231 public items scores all 202/231, hard 83/111, with 0 changed answers across all 924 rows of the bench's own r1..r4; controls 401/422x3 (token boundary proven at 7168 pass / 7169 refuse); GPU 1 per-process peak 9,866 MiB under the largest accepted request (budget 9,876); /decide/shared positive control unchanged. 120 tests green.
This commit is contained in:
@@ -68,6 +68,19 @@ class SharedBody(BaseModel):
|
||||
workload: str | None = None
|
||||
|
||||
|
||||
class SystemOneBody(BaseModel):
|
||||
"""The Jev request. `model` is accepted with any value and ignored (Jev clients send
|
||||
'jev-latest'); `images` is kept only to refuse it with a clear 422. Extra keys are dropped:
|
||||
only state+questions reach predict, exactly like the bench's 60-line wrapper."""
|
||||
model_config = ConfigDict(extra="allow")
|
||||
state: State
|
||||
questions: dict[str, dict[str, Any]]
|
||||
|
||||
@property
|
||||
def images_present(self) -> bool:
|
||||
return "images" in (self.model_extra or {})
|
||||
|
||||
|
||||
class ApiError(Exception):
|
||||
def __init__(self, status: int, code: str, message: str):
|
||||
super().__init__(message)
|
||||
@@ -322,10 +335,11 @@ def create_app(settings: Settings, engine: Any, executor: ThreadPoolExecutor | N
|
||||
"inference_seconds": inference_ms / 1000}
|
||||
return out, timing
|
||||
|
||||
async def score(state: State, decisions: list[Decision], waves: list[list[Slot]]):
|
||||
"""Run one request's calls on the inference thread; map their failures to contract codes."""
|
||||
async def on_inference(fn):
|
||||
"""Run fn on the ONE inference thread (INV-2) and map failures to contract codes (INV-4):
|
||||
the shared dispatch+mapping for every surface, semif and systemone alike."""
|
||||
try:
|
||||
return await asyncio.get_running_loop().run_in_executor(executor, run, state, decisions, waves)
|
||||
return await asyncio.get_running_loop().run_in_executor(executor, fn)
|
||||
except ValueError as exc: # the model's own validation, token limit, ...
|
||||
raise ApiError(422, "invalid_request", str(exc)) from exc
|
||||
except OutOfMemory as exc:
|
||||
@@ -333,11 +347,17 @@ def create_app(settings: Settings, engine: Any, executor: ThreadPoolExecutor | N
|
||||
except Exception as exc: # noqa: BLE001 — any other failure, building the response included
|
||||
raise ApiError(500, "scoring_failed", f"{type(exc).__name__}: {exc}") from exc
|
||||
|
||||
async def score(state: State, decisions: list[Decision], waves: list[list[Slot]]):
|
||||
return await on_inference(lambda: run(state, decisions, waves))
|
||||
|
||||
@app.get("/health")
|
||||
async def health():
|
||||
return {"status": "ok", "model": engine.health(), "vram_cap_gib": settings.vram_cap_gib,
|
||||
"max_tokens": settings.max_tokens, "max_decisions": settings.max_decisions,
|
||||
"max_questions_per_call": MAX_QUESTIONS_PER_CALL, "chunking": CHUNKING, "workloads": []}
|
||||
"max_questions_per_call": MAX_QUESTIONS_PER_CALL, "chunking": CHUNKING, "workloads": [],
|
||||
"endpoints": ["/decide", "/decide/shared", "/v1/systemone"],
|
||||
"systemone": {"max_questions": MAX_QUESTIONS_PER_CALL, "chunking": "none",
|
||||
"images": "not supported", "max_tokens": settings.max_tokens}}
|
||||
|
||||
@app.post("/decide")
|
||||
async def decide(request: Request):
|
||||
@@ -363,4 +383,43 @@ def create_app(settings: Settings, engine: Any, executor: ThreadPoolExecutor | N
|
||||
finally:
|
||||
leave()
|
||||
|
||||
@app.post("/v1/systemone")
|
||||
async def systemone(request: Request):
|
||||
"""Jev wire shape. Straight passthrough to DecisionEngine.predict — NEVER the semif
|
||||
mapping (a different prompt would change the answers). Contract § POST /v1/systemone."""
|
||||
admit() # before the body is read
|
||||
try:
|
||||
payload = await parse(request, SystemOneBody)
|
||||
if payload.images_present:
|
||||
raise ApiError(422, "invalid_request", "images not supported (the vision tower is "
|
||||
"removed to fit the GPU 1 budget, INV-7)")
|
||||
questions = payload.questions
|
||||
if not 1 <= len(questions) <= MAX_QUESTIONS_PER_CALL:
|
||||
raise ApiError(422, "invalid_request",
|
||||
f"this request has {len(questions)} questions; the limit is "
|
||||
f"1..{MAX_QUESTIONS_PER_CALL} in one call — questions of one call "
|
||||
"share one prompt, so they are never chunked")
|
||||
response = await run_systemone({"state": payload.state, "questions": questions})
|
||||
meta = engine.metadata
|
||||
return {"answers": response["answers"], "usage": response["usage"],
|
||||
"model": f"{meta.get('name')}@{meta.get('revision')}", # string: see contract
|
||||
**{k: v for k, v in response.items() if k not in ("answers", "usage", "model")}}
|
||||
finally:
|
||||
leave()
|
||||
|
||||
async def run_systemone(call: dict) -> dict:
|
||||
"""One native call on the inference thread, under the same lock and mapping as the semif
|
||||
surface (on_inference). The engine's own validate_request enforces the per-question Jev
|
||||
rules before any GPU work; its ValueError is a 422 like the token-limit check."""
|
||||
def one():
|
||||
with inference:
|
||||
response, _sha = engine.predict(call)
|
||||
usage = response.get("usage")
|
||||
response["usage"] = usage if isinstance(usage, dict) else {} # contract: {} not absent
|
||||
return response
|
||||
response = await on_inference(one)
|
||||
if not isinstance(response.get("answers"), dict):
|
||||
raise ApiError(500, "scoring_failed", "the engine returned no answers object")
|
||||
return response
|
||||
|
||||
return app
|
||||
|
||||
Reference in New Issue
Block a user