Order averaging (Prime, after the 739aa03 spike):
- A decision may set orderings: rotations|all (all only for <= 4 options). Every
ordering goes to the engine in one shared batch.
- The reply keeps each native result and adds combined {probabilities (log-mean),
top, agreement, spread}.
- Through the service on SemIf's labelled sets (252 rows): 78.6% -> 88.1%
(group-bootstrap 95% CI +5.1..+14.3). Unanimous agreement is 94.5% accurate.
Fast kernels: flash-linear-attention 0.5.2 and causal-conv1d 1.7.0 are now the
default build. A/B on the empty GPU 3:
- parity with upstream went from 142/144 to 144/144;
- a ~2k-token /decide went from 169 to 92 ms server-side;
- short 3-rotation batches cost ~3-6 ms more.
triton builds a C shim at runtime, so the image carries gcc. Without it the
warm-up failed and startup failed closed.
Heid bug-hunt panel (4/4 arms, thread 01M3H3F4RR7XBP90KQ3A39H4SX), folded:
- Startup validation: VRAM cap 0 no longer means uncapped (C1); limits must be
>= 1 (S1); the token must be visible ASCII (S2); the calibration file must
exist and parse, with T in [0.05, 20] (S8, and C3's NaN leg).
- The body limit is checked before a chunk is kept, and a Unicode-digit
Content-Length no longer crashes (C2, S3).
- Failures while building the response now get the 500 envelope (C3).
- 429 busy past SEMIF_MAX_QUEUE requests in progress (C6).
- The engine releases memory on every non-validation failure, unchained after
gc; an empty OOM message is handled; 'out of memory' RuntimeErrors map to 503
(C4, C5, S9).
- The entry point forces HF_HUB_OFFLINE (S10). README wording fixed (S5, S6).
- New guard tests close the gaps the arms' mutation grids exposed: early stop of
the body read, a shared-route lock, calibration pass-through, the gc cycle,
the exact caps, TorchEngine.load's arch and device checks, and the offline
entry point.
86 tests.
Deployed on fv-ml1 GPU 1: parity 144/144, OOM and burst release verified, shared
capacity 63/51/26/16 rows at ~140/520/1960/3900 prefix tokens.
133 lines
6.3 KiB
Python
133 lines
6.3 KiB
Python
"""Order averaging (0.1.3). Contract: semif-serve.contract.md § Order averaging."""
|
|
import math
|
|
|
|
import pytest
|
|
from fastapi.testclient import TestClient
|
|
|
|
from semif_serve.app import create_app
|
|
from semif_serve.config import Settings
|
|
|
|
TOKEN = "t" * 40
|
|
AUTH = {"Authorization": f"Bearer {TOKEN}"}
|
|
OPTS = [{"id": "casual", "description": "A casual outing"},
|
|
{"id": "date", "description": "A romantic date"},
|
|
{"id": "booty", "description": "A booty call"}]
|
|
ROW = {"id": "q", "state": "It's 2AM and I'm bored.", "question": "What is this?", "options": OPTS}
|
|
BASE = {"casual": 1.0, "date": -1.0, "booty": 0.5} # what the model "really" thinks
|
|
BIAS = 2.5 # added to whichever option is listed first
|
|
|
|
|
|
def softmax(xs):
|
|
top = max(xs)
|
|
w = [math.exp(x - top) for x in xs]
|
|
return [v / sum(w) for v in w]
|
|
|
|
|
|
class BiasedEngine:
|
|
"""Scores each option as BASE[id], plus BIAS for the first-listed option, the way a small model leans."""
|
|
|
|
def __init__(self):
|
|
self.shared_calls, self.direct_calls = [], []
|
|
|
|
def health(self):
|
|
return {}
|
|
|
|
def _score(self, row):
|
|
ids = [o["id"] for o in row["options"]]
|
|
logits = [BASE[i] + (BIAS if k == 0 else 0.0) for k, i in enumerate(ids)]
|
|
return {"id": row["id"], "option_ids": ids, "probabilities": softmax(logits), "option_logits": logits,
|
|
"prompt_sha256": row["id"], "probability_status": "uncalibrated"}
|
|
|
|
def direct(self, row):
|
|
self.direct_calls.append(row)
|
|
return self._score(row)
|
|
|
|
def shared(self, rows):
|
|
self.shared_calls.append(rows)
|
|
return [self._score(r) for r in rows], {"batch_size": len(rows)}
|
|
|
|
|
|
def client(engine, **kw):
|
|
return TestClient(create_app(Settings(api_token=TOKEN, **kw), engine))
|
|
|
|
|
|
def test_rotations_send_one_shared_call_with_every_option_once_per_position_and_cancel_the_bias():
|
|
engine = BiasedEngine()
|
|
body = client(engine).post("/decide", json={**ROW, "orderings": "rotations"}, headers=AUTH).json()
|
|
assert engine.direct_calls == [] and len(engine.shared_calls) == 1
|
|
rows = engine.shared_calls[0]
|
|
assert [r["id"] for r in rows] == ["q#o0", "q#o1", "q#o2"]
|
|
assert [[o["id"] for o in r["options"]] for r in rows] == [
|
|
["casual", "date", "booty"], ["date", "booty", "casual"], ["booty", "casual", "date"]]
|
|
assert all(r["state"] == ROW["state"] and r["question"] == ROW["question"] for r in rows)
|
|
|
|
assert body["id"] == "q" and body["option_ids"] == ["casual", "date", "booty"]
|
|
combined = body["combined"]
|
|
assert combined["method"] == "rotations" and combined["orderings"] == 3
|
|
# the first-position bias is additive, and every option sat first exactly once, so it cancels exactly
|
|
assert combined["probabilities"] == pytest.approx(softmax([BASE["casual"], BASE["date"], BASE["booty"]]))
|
|
assert combined["top"] == "casual"
|
|
# the bias is strong enough that every ordering's first option wins: casual, date, booty
|
|
tops = [max(zip(r["option_logits"], r["option_ids"]))[1] for r in body["orderings"]]
|
|
assert tops == ["casual", "date", "booty"]
|
|
assert combined["agreement"] == pytest.approx(1 / 3)
|
|
assert body["orderings"] == [engine._score(r) for r in rows] # native results, unchanged
|
|
for oid in ("casual", "date", "booty"):
|
|
ps = [dict(zip(r["option_ids"], r["probabilities"]))[oid] for r in body["orderings"]]
|
|
assert combined["spread"][oid] == pytest.approx([min(ps), max(ps)])
|
|
|
|
|
|
def test_all_sends_every_permutation_with_the_callers_order_first():
|
|
engine = BiasedEngine()
|
|
body = client(engine).post("/decide", json={**ROW, "orderings": "all"}, headers=AUTH).json()
|
|
rows = engine.shared_calls[0]
|
|
orders = [tuple(o["id"] for o in r["options"]) for r in rows]
|
|
assert len(orders) == 6 and len(set(orders)) == 6 and orders[0] == ("casual", "date", "booty")
|
|
assert body["combined"]["orderings"] == 6 and body["combined"]["method"] == "all"
|
|
|
|
|
|
def test_all_above_four_options_is_422_before_the_engine_runs():
|
|
engine = BiasedEngine()
|
|
five = [{"id": f"o{i}", "description": f"Option {i}"} for i in range(5)]
|
|
r = client(engine).post("/decide", json={**ROW, "options": five, "orderings": "all"}, headers=AUTH)
|
|
assert r.status_code == 422 and "rotations" in r.json()["error"]["message"]
|
|
assert engine.shared_calls == []
|
|
|
|
|
|
def test_a_mixed_shared_request_is_one_engine_call_with_results_in_request_order():
|
|
engine = BiasedEngine()
|
|
body = {"state": ROW["state"], "decisions": [
|
|
{"id": "plain", "question": "Q1?", "options": OPTS},
|
|
{"id": "avg", "question": "Q2?", "options": OPTS, "orderings": "rotations"},
|
|
{"id": "plain2", "question": "Q3?", "options": OPTS[:2]}]}
|
|
out = client(engine).post("/decide/shared", json=body, headers=AUTH).json()
|
|
assert len(engine.shared_calls) == 1
|
|
assert [r["id"] for r in engine.shared_calls[0]] == ["plain", "avg#o0", "avg#o1", "avg#o2", "plain2"]
|
|
ids = [r["id"] for r in out["results"]]
|
|
assert ids == ["plain", "avg", "plain2"]
|
|
assert out["results"][0] == engine._score(engine.shared_calls[0][0]) # plain results keep the old shape
|
|
assert "combined" in out["results"][1] and "combined" not in out["results"][2]
|
|
assert out["timing"] == {"batch_size": 5}
|
|
|
|
|
|
def test_expanded_rows_count_toward_the_cap():
|
|
engine = BiasedEngine()
|
|
r = client(engine, max_decisions=5).post("/decide/shared", json={"state": "s", "decisions": [
|
|
{"id": "a", "question": "Q?", "options": OPTS, "orderings": "rotations"},
|
|
{"id": "b", "question": "Q?", "options": OPTS, "orderings": "rotations"}]}, headers=AUTH)
|
|
assert r.status_code == 422 and "6 scored rows" in r.json()["error"]["message"]
|
|
assert engine.shared_calls == []
|
|
|
|
|
|
def test_workload_with_orderings_is_422_and_workload_alone_still_calibrates():
|
|
engine = BiasedEngine()
|
|
c = client(engine, calibration={"triage": 2.0})
|
|
r = c.post("/decide", json={**ROW, "orderings": "rotations", "workload": "triage"}, headers=AUTH)
|
|
assert r.status_code == 422 and engine.shared_calls == []
|
|
assert "calibrated" in c.post("/decide", json={**ROW, "workload": "triage"}, headers=AUTH).json()
|
|
|
|
|
|
def test_an_unknown_orderings_value_is_422():
|
|
r = client(BiasedEngine()).post("/decide", json={**ROW, "orderings": "shuffle"}, headers=AUTH)
|
|
assert r.status_code == 422
|