Audit finding on 0.1.1: a Jev client that always sends an images array with no images was rejected for nothing. Only a non-empty value is 422 now. Also aligns the contract's response example with the wire (model is the name@revision string, not an object). Re-accepted live: images []/null -> 200, ["a.png"] -> 422; JevBench all 202/231, hard 83/111, 0 changed rows across the bench's r1..r4 (924).
235 lines
9.8 KiB
Python
235 lines
9.8 KiB
Python
"""POST /v1/systemone: Jev wire shape, straight passthrough to DecisionEngine.predict.
|
|
Contract: intern-decision-serve.contract.md § POST /v1/systemone"""
|
|
import threading
|
|
import time
|
|
from concurrent.futures import ThreadPoolExecutor
|
|
|
|
import pytest
|
|
from fastapi.testclient import TestClient
|
|
|
|
from fake_engine import FakeEngine
|
|
from intern_decision_serve.app import create_app
|
|
from intern_decision_serve.config import Settings
|
|
from intern_decision_serve.errors import OutOfMemory
|
|
|
|
TOKEN = "t" * 40
|
|
AUTH = {"Authorization": f"Bearer {TOKEN}"}
|
|
|
|
CHOICE = {"type": "choice", "instructions": "Did it pass?",
|
|
"criteria": {"yes": "It passed.", "no": "It did not pass at all."}}
|
|
NOUL = {"type": "noul", "instructions": "Is the deploy healthy?"}
|
|
SCORE = {"type": "score", "instructions": "How risky?", "criteria": ["low", "medium", "high"]}
|
|
|
|
|
|
def make_client(engine=None, **overrides):
|
|
return TestClient(create_app(Settings(api_token=TOKEN, **overrides), engine or FakeEngine()))
|
|
|
|
|
|
def body(**over):
|
|
b = {"state": "The deploy finished at 14:00.", "model": "jev-latest", "questions": {"decision": CHOICE}}
|
|
b.update(over)
|
|
return b
|
|
|
|
|
|
# ---------------------------------------------------------------- happy paths
|
|
|
|
def test_choice_request_is_passed_straight_through_and_the_engine_answer_is_returned():
|
|
engine = FakeEngine()
|
|
response = make_client(engine).post("/v1/systemone", json=body(), headers=AUTH)
|
|
assert response.status_code == 200
|
|
native, _ = FakeEngine().predict(engine.calls[0])
|
|
# straight through: only state+questions reach predict, model is not forwarded, the question
|
|
# dict is untouched (no semif mapping: no A=id:description conversion, no field rename)
|
|
assert engine.calls == [{"state": "The deploy finished at 14:00.", "questions": {"decision": CHOICE}}]
|
|
out = response.json()
|
|
assert out["answers"] == native["answers"]
|
|
assert out["answers"]["decision"]["type"] == "choice"
|
|
assert out["answers"]["decision"]["choice"] in ("yes", "no")
|
|
assert out["usage"] == native["usage"]
|
|
# a STRING: JevBench's runner hashes the value into its manifest (a dict broke it on the
|
|
# 0.1.1 first run) and real clients compare it; the string names the real model and revision
|
|
assert out["model"] == "Intern-Decision-4B@" + "0" * 40
|
|
|
|
|
|
def test_noul_question_needs_no_criteria_and_answers_with_a_probability():
|
|
engine = FakeEngine()
|
|
response = make_client(engine).post("/v1/systemone", json=body(questions={"q": NOUL}), headers=AUTH)
|
|
assert response.status_code == 200
|
|
assert engine.calls == [{"state": "The deploy finished at 14:00.", "questions": {"q": NOUL}}]
|
|
answer = response.json()["answers"]["q"]
|
|
assert answer["type"] == "noul" and answer["noul"] == 0.75
|
|
|
|
|
|
def test_score_question_with_a_list_of_levels_returns_probabilities_over_the_levels():
|
|
engine = FakeEngine()
|
|
response = make_client(engine).post("/v1/systemone", json=body(questions={"q": SCORE}), headers=AUTH)
|
|
assert response.status_code == 200
|
|
answer = response.json()["answers"]["q"]
|
|
assert answer["type"] == "score"
|
|
assert set(answer["probabilities"]) == {"0", "1", "2"}
|
|
assert abs(sum(answer["probabilities"].values()) - 1.0) < 1e-9
|
|
|
|
|
|
def test_any_number_of_questions_up_to_16_is_ONE_call_so_they_share_one_prompt():
|
|
engine = FakeEngine()
|
|
questions = {f"q{i}": NOUL for i in range(16)}
|
|
response = make_client(engine).post("/v1/systemone", json=body(questions=questions), headers=AUTH)
|
|
assert response.status_code == 200
|
|
assert len(engine.calls) == 1 and len(engine.calls[0]["questions"]) == 16
|
|
|
|
|
|
def test_the_model_field_is_accepted_with_any_value_and_ignored():
|
|
engine = FakeEngine()
|
|
for model in ("jev-latest", "anything-else", 7, None):
|
|
response = make_client(engine).post("/v1/systemone", json=body(model=model), headers=AUTH)
|
|
assert response.status_code == 200
|
|
|
|
|
|
class NoUsageEngine(FakeEngine):
|
|
def predict(self, request):
|
|
response, sha = super().predict(request)
|
|
response.pop("usage")
|
|
return response, sha
|
|
|
|
|
|
def test_an_engine_that_reports_no_usage_yields_an_empty_usage_object():
|
|
response = make_client(NoUsageEngine()).post("/v1/systemone", json=body(), headers=AUTH)
|
|
assert response.status_code == 200
|
|
assert response.json()["usage"] == {}
|
|
|
|
|
|
# ---------------------------------------------------------------- refusals
|
|
|
|
@pytest.mark.parametrize("headers", [{}, {"Authorization": "Bearer wrong"}, {"Authorization": TOKEN}])
|
|
def test_without_the_right_bearer_a_systemone_post_is_401_and_never_reaches_the_engine(headers):
|
|
engine = FakeEngine()
|
|
response = make_client(engine).post("/v1/systemone", json=body(), headers=headers)
|
|
assert response.status_code == 401
|
|
assert response.json()["error"]["code"] == "unauthorized"
|
|
assert engine.calls == []
|
|
|
|
|
|
@pytest.mark.parametrize("images", [["a.png"], ["a.png", "b.png"], "a.png", {"a": "b"}, 7])
|
|
def test_a_nonempty_images_value_is_422_before_the_engine_runs(images):
|
|
engine = FakeEngine()
|
|
response = make_client(engine).post("/v1/systemone", json=body(images=images), headers=AUTH)
|
|
assert response.status_code == 422
|
|
assert response.json()["error"]["code"] == "invalid_request"
|
|
assert "images not supported" in response.json()["error"]["message"]
|
|
assert engine.calls == []
|
|
|
|
|
|
@pytest.mark.parametrize("images", [[], None])
|
|
def test_an_empty_or_null_images_field_counts_as_absent(images):
|
|
"""Audit finding 2026-09-30: a Jev client that always sends the field with no images
|
|
must not be rejected for nothing."""
|
|
engine = FakeEngine()
|
|
response = make_client(engine).post("/v1/systemone", json=body(images=images), headers=AUTH)
|
|
assert response.status_code == 200
|
|
assert engine.calls == [{"state": "The deploy finished at 14:00.", "questions": {"decision": CHOICE}}]
|
|
|
|
|
|
def test_17_questions_is_422_not_a_chunked_2_call_request():
|
|
engine = FakeEngine()
|
|
questions = {f"q{i}": NOUL for i in range(17)}
|
|
response = make_client(engine).post("/v1/systemone", json=body(questions=questions), headers=AUTH)
|
|
assert response.status_code == 422
|
|
assert "16" in response.json()["error"]["message"]
|
|
assert engine.calls == []
|
|
|
|
|
|
def test_zero_questions_is_422():
|
|
response = make_client().post("/v1/systemone", json=body(questions={}), headers=AUTH)
|
|
assert response.status_code == 422
|
|
|
|
|
|
def test_a_body_that_is_not_json_or_has_no_questions_is_422():
|
|
client = make_client()
|
|
assert client.post("/v1/systemone", content=b"{nope", headers={**AUTH, "content-type": "application/json"}
|
|
).status_code == 422
|
|
assert client.post("/v1/systemone", json={"state": "s"}, headers=AUTH).status_code == 422
|
|
|
|
|
|
class TokenLimitEngine(FakeEngine):
|
|
"""Stands in for the model's own pre-forward check: predict raises ValueError, never a tensor."""
|
|
def predict(self, request):
|
|
self.calls.append(request)
|
|
raise ValueError("Example has 7169 tokens, above 7168; truncation is forbidden")
|
|
|
|
|
|
def test_over_max_tokens_is_422_from_the_models_own_before_the_forward_check():
|
|
response = make_client(TokenLimitEngine(), max_tokens=7168).post("/v1/systemone", json=body(), headers=AUTH)
|
|
assert response.status_code == 422
|
|
assert "7169" in response.json()["error"]["message"]
|
|
|
|
|
|
# ---------------------------------------------------------------- queue and memory
|
|
|
|
class SlowEngine(FakeEngine):
|
|
def __init__(self):
|
|
super().__init__()
|
|
self.entered, self.release = threading.Event(), threading.Event()
|
|
|
|
def predict(self, request):
|
|
self.entered.set()
|
|
assert self.release.wait(5)
|
|
return super().predict(request)
|
|
|
|
|
|
def test_more_than_max_queue_systemone_posts_get_429_busy():
|
|
engine = SlowEngine()
|
|
with make_client(engine, max_queue=1) as client, ThreadPoolExecutor(2) as pool:
|
|
held = pool.submit(client.post, "/v1/systemone", json=body(), headers=AUTH)
|
|
assert engine.entered.wait(5)
|
|
time.sleep(0.2)
|
|
extra = client.post("/v1/systemone", json=body(), headers=AUTH)
|
|
assert extra.status_code == 429 and extra.json()["error"]["code"] == "busy"
|
|
engine.release.set()
|
|
assert held.result().status_code == 200
|
|
|
|
|
|
class OomOnceEngine(FakeEngine):
|
|
def __init__(self):
|
|
super().__init__()
|
|
self.raised = False
|
|
|
|
def predict(self, request):
|
|
if not self.raised:
|
|
self.raised, _ = True, self.calls.append(request)
|
|
raise OutOfMemory("CUDA out of memory.")
|
|
return super().predict(request)
|
|
|
|
|
|
def test_an_oom_is_503_with_the_contract_envelope_and_the_next_request_succeeds():
|
|
engine = OomOnceEngine()
|
|
client = make_client(engine)
|
|
first = client.post("/v1/systemone", json=body(), headers=AUTH)
|
|
assert first.status_code == 503 and first.json()["error"]["code"] == "out_of_memory"
|
|
second = client.post("/v1/systemone", json=body(), headers=AUTH)
|
|
assert second.status_code == 200
|
|
|
|
|
|
def test_systemone_runs_on_the_same_single_inference_thread_as_decide():
|
|
seen = []
|
|
|
|
class ThreadSpy(FakeEngine):
|
|
def predict(self, request):
|
|
seen.append(threading.current_thread().name)
|
|
return super().predict(request)
|
|
|
|
client = make_client(ThreadSpy())
|
|
client.post("/v1/systemone", json=body(), headers=AUTH)
|
|
client.post("/decide", json={"id": "r", "state": "s", "question": "q?", "options":
|
|
[{"id": "a", "description": "x"}, {"id": "b", "description": "yy"}]},
|
|
headers=AUTH)
|
|
assert len(set(seen)) == 1 and seen[0].startswith("inference")
|
|
|
|
|
|
# ---------------------------------------------------------------- /health
|
|
|
|
def test_health_advertises_the_systemone_surface_and_its_limits():
|
|
out = make_client(max_tokens=7168).get("/health").json()
|
|
assert set(out["endpoints"]) == {"/decide", "/decide/shared", "/v1/systemone"}
|
|
assert out["systemone"] == {"max_questions": 16, "chunking": "none",
|
|
"images": "not supported", "max_tokens": 7168}
|