feat(semif): 0.1.3 — order averaging, fast kernels, bug-hunt hardening (Prime)

Order averaging (Prime, after the 739aa03 spike):
- A decision may set orderings: rotations|all (all only for <= 4 options). Every
  ordering goes to the engine in one shared batch.
- The reply keeps each native result and adds combined {probabilities (log-mean),
  top, agreement, spread}.
- Through the service on SemIf's labelled sets (252 rows): 78.6% -> 88.1%
  (group-bootstrap 95% CI +5.1..+14.3). Unanimous agreement is 94.5% accurate.

Fast kernels: flash-linear-attention 0.5.2 and causal-conv1d 1.7.0 are now the
default build. A/B on the empty GPU 3:
- parity with upstream went from 142/144 to 144/144;
- a ~2k-token /decide went from 169 to 92 ms server-side;
- short 3-rotation batches cost ~3-6 ms more.
triton builds a C shim at runtime, so the image carries gcc. Without it the
warm-up failed and startup failed closed.

Heid bug-hunt panel (4/4 arms, thread 01M3H3F4RR7XBP90KQ3A39H4SX), folded:
- Startup validation: VRAM cap 0 no longer means uncapped (C1); limits must be
  >= 1 (S1); the token must be visible ASCII (S2); the calibration file must
  exist and parse, with T in [0.05, 20] (S8, and C3's NaN leg).
- The body limit is checked before a chunk is kept, and a Unicode-digit
  Content-Length no longer crashes (C2, S3).
- Failures while building the response now get the 500 envelope (C3).
- 429 busy past SEMIF_MAX_QUEUE requests in progress (C6).
- The engine releases memory on every non-validation failure, unchained after
  gc; an empty OOM message is handled; 'out of memory' RuntimeErrors map to 503
  (C4, C5, S9).
- The entry point forces HF_HUB_OFFLINE (S10). README wording fixed (S5, S6).
- New guard tests close the gaps the arms' mutation grids exposed: early stop of
  the body read, a shared-route lock, calibration pass-through, the gc cycle,
  the exact caps, TorchEngine.load's arch and device checks, and the offline
  entry point.
86 tests.

Deployed on fv-ml1 GPU 1: parity 144/144, OOM and burst release verified, shared
capacity 63/51/26/16 rows at ~140/520/1960/3900 prefix tokens.
This commit is contained in:
vh
2026-09-27 03:27:15 -07:00
parent d7ad235365
commit 77b8cb449c
22 changed files with 1018 additions and 92 deletions
+103
View File
@@ -1,5 +1,7 @@
"""semif-serve HTTP behaviour against a fake engine (no torch, no model).
Contract: services/semif-serve/semif-serve.contract.md"""
import json
import pytest
from fastapi.testclient import TestClient
@@ -231,3 +233,104 @@ def test_health_reports_the_pins_limits_and_workloads():
body = client.get("/health").json()
assert body == {"status": "ok", "semif_commit": SEMIF_COMMIT, "model": FakeEngine().health(),
"vram_cap_gib": 12.0, "max_tokens": 4096, "max_decisions": 8, "workloads": ["alerts", "triage"]}
def test_a_body_of_exactly_the_limit_is_accepted():
body = json.dumps(ROW).encode()
client = make_client(max_body_bytes=len(body))
assert client.post("/decide", content=body, headers={**AUTH, "content-type": "application/json"}).status_code == 200
def test_a_non_ascii_digit_content_length_is_ignored_not_a_crash():
"""HTTP clients cannot send one (httpx refuses; h11 rejects it), so check the reader directly."""
import asyncio
from semif_serve.app import read_limited
async def body():
yield b"{}"
assert asyncio.run(read_limited(body(), "²", 100)) == b"{}" # int("²") would raise
def test_exactly_max_decisions_is_accepted():
decisions = [{"id": str(i), "question": "Q?", "options": OPTIONS} for i in range(3)]
client = make_client(max_decisions=3)
assert client.post("/decide/shared", json={"state": "s", "decisions": decisions}, headers=AUTH).status_code == 200
def test_calibration_leaves_every_native_field_alone():
engine = FakeEngine(logits=(3.0, 1.0))
body = make_client(engine, calibration={"triage": 2.0}).post(
"/decide", json={**ROW, "workload": "triage"}, headers=AUTH).json()
body.pop("calibrated")
assert body == engine.direct(ROW)
class MalformedEngine(FakeEngine):
def direct(self, row):
return {"id": row["id"], "option_ids": ["yes", "no"], "probabilities": [0.5, 0.5]} # no option_logits
def shared(self, rows):
return [self.direct(r) for r in rows], {}
@pytest.mark.parametrize("path, body", [
("/decide", {**ROW, "workload": "triage"}),
("/decide", {**ROW, "orderings": "rotations"}),
])
def test_a_malformed_scorer_result_is_an_envelope_500_not_a_bare_one(path, body):
client = make_client(MalformedEngine(), calibration={"triage": 2.0})
response = client.post(path, json=body, headers=AUTH)
assert response.status_code == 500
assert response.json()["error"]["code"] == "scoring_failed"
class SharedSlowEngine(SlowEngine):
def shared(self, rows):
return [self.direct(r) for r in rows], {}
def test_shared_requests_are_serialised_too():
from concurrent.futures import ThreadPoolExecutor
engine = SharedSlowEngine()
body = {"state": "s", "decisions": [{"id": "a", "question": "Q?", "options": OPTIONS}]}
with make_client(engine) as client, ThreadPoolExecutor(3) as pool:
futures = [pool.submit(client.post, "/decide/shared", json=body, headers=AUTH) for _ in range(3)]
assert engine.entered.wait(5)
engine.release.set()
assert [f.result().status_code for f in futures] == [200] * 3
assert engine.peak == 1
def test_more_than_max_queue_requests_in_progress_get_429_busy():
from concurrent.futures import ThreadPoolExecutor
engine = SlowEngine()
with make_client(engine, max_queue=2) as client, ThreadPoolExecutor(3) as pool:
held = [pool.submit(client.post, "/decide", json={**ROW, "id": f"r{i}"}, headers=AUTH) for i in range(2)]
assert engine.entered.wait(5)
import time
deadline = time.monotonic() + 5
while engine.inside + 0 < 1 and time.monotonic() < deadline:
time.sleep(0.01)
time.sleep(0.2) # let the second request reach the queue
extra = client.post("/decide", json={**ROW, "id": "extra"}, headers=AUTH)
assert extra.status_code == 429 and extra.json()["error"]["code"] == "busy"
engine.release.set()
assert [f.result().status_code for f in held] == [200, 200]
assert make_client(FakeEngine(), max_queue=2).post("/decide", json=ROW, headers=AUTH).status_code == 200
def test_read_limited_stops_reading_at_the_crossing_chunk():
import asyncio
from semif_serve.app import ApiError, read_limited
consumed = []
async def chunks():
for i in range(10):
consumed.append(i)
yield b"x" * 100
with pytest.raises(ApiError) as info:
asyncio.run(read_limited(chunks(), None, 250))
assert info.value.status == 413
assert consumed == [0, 1, 2] # the third chunk crosses 250 and is never kept; nothing after is read