Files
esh-pfi-infrastructure/services/semif-serve/tests/test_orderings.py
T
vh 77b8cb449c feat(semif): 0.1.3 — order averaging, fast kernels, bug-hunt hardening (Prime)
Order averaging (Prime, after the 739aa03 spike):
- A decision may set orderings: rotations|all (all only for <= 4 options). Every
  ordering goes to the engine in one shared batch.
- The reply keeps each native result and adds combined {probabilities (log-mean),
  top, agreement, spread}.
- Through the service on SemIf's labelled sets (252 rows): 78.6% -> 88.1%
  (group-bootstrap 95% CI +5.1..+14.3). Unanimous agreement is 94.5% accurate.

Fast kernels: flash-linear-attention 0.5.2 and causal-conv1d 1.7.0 are now the
default build. A/B on the empty GPU 3:
- parity with upstream went from 142/144 to 144/144;
- a ~2k-token /decide went from 169 to 92 ms server-side;
- short 3-rotation batches cost ~3-6 ms more.
triton builds a C shim at runtime, so the image carries gcc. Without it the
warm-up failed and startup failed closed.

Heid bug-hunt panel (4/4 arms, thread 01M3H3F4RR7XBP90KQ3A39H4SX), folded:
- Startup validation: VRAM cap 0 no longer means uncapped (C1); limits must be
  >= 1 (S1); the token must be visible ASCII (S2); the calibration file must
  exist and parse, with T in [0.05, 20] (S8, and C3's NaN leg).
- The body limit is checked before a chunk is kept, and a Unicode-digit
  Content-Length no longer crashes (C2, S3).
- Failures while building the response now get the 500 envelope (C3).
- 429 busy past SEMIF_MAX_QUEUE requests in progress (C6).
- The engine releases memory on every non-validation failure, unchained after
  gc; an empty OOM message is handled; 'out of memory' RuntimeErrors map to 503
  (C4, C5, S9).
- The entry point forces HF_HUB_OFFLINE (S10). README wording fixed (S5, S6).
- New guard tests close the gaps the arms' mutation grids exposed: early stop of
  the body read, a shared-route lock, calibration pass-through, the gc cycle,
  the exact caps, TorchEngine.load's arch and device checks, and the offline
  entry point.
86 tests.

Deployed on fv-ml1 GPU 1: parity 144/144, OOM and burst release verified, shared
capacity 63/51/26/16 rows at ~140/520/1960/3900 prefix tokens.
2026-09-27 03:27:15 -07:00

133 lines
6.3 KiB
Python

"""Order averaging (0.1.3). Contract: semif-serve.contract.md § Order averaging."""
import math
import pytest
from fastapi.testclient import TestClient
from semif_serve.app import create_app
from semif_serve.config import Settings
TOKEN = "t" * 40
AUTH = {"Authorization": f"Bearer {TOKEN}"}
OPTS = [{"id": "casual", "description": "A casual outing"},
{"id": "date", "description": "A romantic date"},
{"id": "booty", "description": "A booty call"}]
ROW = {"id": "q", "state": "It's 2AM and I'm bored.", "question": "What is this?", "options": OPTS}
BASE = {"casual": 1.0, "date": -1.0, "booty": 0.5} # what the model "really" thinks
BIAS = 2.5 # added to whichever option is listed first
def softmax(xs):
top = max(xs)
w = [math.exp(x - top) for x in xs]
return [v / sum(w) for v in w]
class BiasedEngine:
"""Scores each option as BASE[id], plus BIAS for the first-listed option, the way a small model leans."""
def __init__(self):
self.shared_calls, self.direct_calls = [], []
def health(self):
return {}
def _score(self, row):
ids = [o["id"] for o in row["options"]]
logits = [BASE[i] + (BIAS if k == 0 else 0.0) for k, i in enumerate(ids)]
return {"id": row["id"], "option_ids": ids, "probabilities": softmax(logits), "option_logits": logits,
"prompt_sha256": row["id"], "probability_status": "uncalibrated"}
def direct(self, row):
self.direct_calls.append(row)
return self._score(row)
def shared(self, rows):
self.shared_calls.append(rows)
return [self._score(r) for r in rows], {"batch_size": len(rows)}
def client(engine, **kw):
return TestClient(create_app(Settings(api_token=TOKEN, **kw), engine))
def test_rotations_send_one_shared_call_with_every_option_once_per_position_and_cancel_the_bias():
engine = BiasedEngine()
body = client(engine).post("/decide", json={**ROW, "orderings": "rotations"}, headers=AUTH).json()
assert engine.direct_calls == [] and len(engine.shared_calls) == 1
rows = engine.shared_calls[0]
assert [r["id"] for r in rows] == ["q#o0", "q#o1", "q#o2"]
assert [[o["id"] for o in r["options"]] for r in rows] == [
["casual", "date", "booty"], ["date", "booty", "casual"], ["booty", "casual", "date"]]
assert all(r["state"] == ROW["state"] and r["question"] == ROW["question"] for r in rows)
assert body["id"] == "q" and body["option_ids"] == ["casual", "date", "booty"]
combined = body["combined"]
assert combined["method"] == "rotations" and combined["orderings"] == 3
# the first-position bias is additive, and every option sat first exactly once, so it cancels exactly
assert combined["probabilities"] == pytest.approx(softmax([BASE["casual"], BASE["date"], BASE["booty"]]))
assert combined["top"] == "casual"
# the bias is strong enough that every ordering's first option wins: casual, date, booty
tops = [max(zip(r["option_logits"], r["option_ids"]))[1] for r in body["orderings"]]
assert tops == ["casual", "date", "booty"]
assert combined["agreement"] == pytest.approx(1 / 3)
assert body["orderings"] == [engine._score(r) for r in rows] # native results, unchanged
for oid in ("casual", "date", "booty"):
ps = [dict(zip(r["option_ids"], r["probabilities"]))[oid] for r in body["orderings"]]
assert combined["spread"][oid] == pytest.approx([min(ps), max(ps)])
def test_all_sends_every_permutation_with_the_callers_order_first():
engine = BiasedEngine()
body = client(engine).post("/decide", json={**ROW, "orderings": "all"}, headers=AUTH).json()
rows = engine.shared_calls[0]
orders = [tuple(o["id"] for o in r["options"]) for r in rows]
assert len(orders) == 6 and len(set(orders)) == 6 and orders[0] == ("casual", "date", "booty")
assert body["combined"]["orderings"] == 6 and body["combined"]["method"] == "all"
def test_all_above_four_options_is_422_before_the_engine_runs():
engine = BiasedEngine()
five = [{"id": f"o{i}", "description": f"Option {i}"} for i in range(5)]
r = client(engine).post("/decide", json={**ROW, "options": five, "orderings": "all"}, headers=AUTH)
assert r.status_code == 422 and "rotations" in r.json()["error"]["message"]
assert engine.shared_calls == []
def test_a_mixed_shared_request_is_one_engine_call_with_results_in_request_order():
engine = BiasedEngine()
body = {"state": ROW["state"], "decisions": [
{"id": "plain", "question": "Q1?", "options": OPTS},
{"id": "avg", "question": "Q2?", "options": OPTS, "orderings": "rotations"},
{"id": "plain2", "question": "Q3?", "options": OPTS[:2]}]}
out = client(engine).post("/decide/shared", json=body, headers=AUTH).json()
assert len(engine.shared_calls) == 1
assert [r["id"] for r in engine.shared_calls[0]] == ["plain", "avg#o0", "avg#o1", "avg#o2", "plain2"]
ids = [r["id"] for r in out["results"]]
assert ids == ["plain", "avg", "plain2"]
assert out["results"][0] == engine._score(engine.shared_calls[0][0]) # plain results keep the old shape
assert "combined" in out["results"][1] and "combined" not in out["results"][2]
assert out["timing"] == {"batch_size": 5}
def test_expanded_rows_count_toward_the_cap():
engine = BiasedEngine()
r = client(engine, max_decisions=5).post("/decide/shared", json={"state": "s", "decisions": [
{"id": "a", "question": "Q?", "options": OPTS, "orderings": "rotations"},
{"id": "b", "question": "Q?", "options": OPTS, "orderings": "rotations"}]}, headers=AUTH)
assert r.status_code == 422 and "6 scored rows" in r.json()["error"]["message"]
assert engine.shared_calls == []
def test_workload_with_orderings_is_422_and_workload_alone_still_calibrates():
engine = BiasedEngine()
c = client(engine, calibration={"triage": 2.0})
r = c.post("/decide", json={**ROW, "orderings": "rotations", "workload": "triage"}, headers=AUTH)
assert r.status_code == 422 and engine.shared_calls == []
assert "calibrated" in c.post("/decide", json={**ROW, "workload": "triage"}, headers=AUTH).json()
def test_an_unknown_orderings_value_is_422():
r = client(BiasedEngine()).post("/decide", json={**ROW, "orderings": "shuffle"}, headers=AUTH)
assert r.status_code == 422