fix(intern-decision-serve): code-review fixes

- a failure while building the response (a non-finite number included) is a 500 inside the
  envelope, never a 422 or a render crash outside it
- an engine ValueError keeps its message but is released and raised unchained, like an OOM
- the prompt is built (and the model's own validation run) before the forward
- the row cap is counted before any ordering is built
- /health reads a device name cached at load, so it makes no driver call off the inference thread
This commit is contained in:
vh
2026-09-30 09:30:40 -07:00
parent 618390c5fa
commit 21d16d7ad8
5 changed files with 110 additions and 21 deletions
@@ -464,3 +464,28 @@ def test_the_app_uses_the_executor_it_is_given():
assert client.post("/decide", json=ROW, headers=AUTH).status_code == 200
assert engine.threads == {home}
executor.shutdown()
class InfiniteEngine(FakeEngine):
"""A server-side defect: a probability of +inf, which breaks the averaging arithmetic."""
def predict(self, request):
response, sha = super().predict(request)
for answer in response["answers"].values():
first = next(iter(answer["probabilities"]))
answer["probabilities"][first] = float("inf")
return response, sha
def test_a_failure_while_building_the_response_is_500_never_a_422():
response = make_client(InfiniteEngine()).post("/decide", json={**ROW, "orderings": "rotations"}, headers=AUTH)
assert response.status_code == 500
assert response.json()["error"]["code"] == "scoring_failed"
def test_a_huge_orderings_request_is_refused_with_its_true_row_count():
engine = FakeEngine()
body = {"state": "s", "decisions": [{**d, "orderings": "rotations"} for d in decisions(3000, opts(16))]}
response = make_client(engine, max_body_bytes=16 * 2**20).post("/decide/shared", json=body, headers=AUTH)
assert response.status_code == 422
assert "48000 rows" in response.json()["error"]["message"]
assert engine.calls == []