"""Order averaging (0.1.3). Contract: semif-serve.contract.md ยง Order averaging.""" import math import pytest from fastapi.testclient import TestClient from semif_serve.app import create_app from semif_serve.config import Settings TOKEN = "t" * 40 AUTH = {"Authorization": f"Bearer {TOKEN}"} OPTS = [{"id": "casual", "description": "A casual outing"}, {"id": "date", "description": "A romantic date"}, {"id": "booty", "description": "A booty call"}] ROW = {"id": "q", "state": "It's 2AM and I'm bored.", "question": "What is this?", "options": OPTS} BASE = {"casual": 1.0, "date": -1.0, "booty": 0.5} # what the model "really" thinks BIAS = 2.5 # added to whichever option is listed first def softmax(xs): top = max(xs) w = [math.exp(x - top) for x in xs] return [v / sum(w) for v in w] class BiasedEngine: """Scores each option as BASE[id], plus BIAS for the first-listed option, the way a small model leans.""" def __init__(self): self.shared_calls, self.direct_calls = [], [] def health(self): return {} def _score(self, row): ids = [o["id"] for o in row["options"]] logits = [BASE[i] + (BIAS if k == 0 else 0.0) for k, i in enumerate(ids)] return {"id": row["id"], "option_ids": ids, "probabilities": softmax(logits), "option_logits": logits, "prompt_sha256": row["id"], "probability_status": "uncalibrated"} def direct(self, row): self.direct_calls.append(row) return self._score(row) def shared(self, rows): self.shared_calls.append(rows) return [self._score(r) for r in rows], {"batch_size": len(rows)} def client(engine, **kw): return TestClient(create_app(Settings(api_token=TOKEN, **kw), engine)) def test_rotations_send_one_shared_call_with_every_option_once_per_position_and_cancel_the_bias(): engine = BiasedEngine() body = client(engine).post("/decide", json={**ROW, "orderings": "rotations"}, headers=AUTH).json() assert engine.direct_calls == [] and len(engine.shared_calls) == 1 rows = engine.shared_calls[0] assert [r["id"] for r in rows] == ["q#o0", "q#o1", "q#o2"] assert [[o["id"] for o in r["options"]] for r in rows] == [ ["casual", "date", "booty"], ["date", "booty", "casual"], ["booty", "casual", "date"]] assert all(r["state"] == ROW["state"] and r["question"] == ROW["question"] for r in rows) assert body["id"] == "q" and body["option_ids"] == ["casual", "date", "booty"] combined = body["combined"] assert combined["method"] == "rotations" and combined["orderings"] == 3 # the first-position bias is additive, and every option sat first exactly once, so it cancels exactly assert combined["probabilities"] == pytest.approx(softmax([BASE["casual"], BASE["date"], BASE["booty"]])) assert combined["top"] == "casual" # the bias is strong enough that every ordering's first option wins: casual, date, booty tops = [max(zip(r["option_logits"], r["option_ids"]))[1] for r in body["orderings"]] assert tops == ["casual", "date", "booty"] assert combined["agreement"] == pytest.approx(1 / 3) assert body["orderings"] == [engine._score(r) for r in rows] # native results, unchanged for oid in ("casual", "date", "booty"): ps = [dict(zip(r["option_ids"], r["probabilities"]))[oid] for r in body["orderings"]] assert combined["spread"][oid] == pytest.approx([min(ps), max(ps)]) def test_all_sends_every_permutation_with_the_callers_order_first(): engine = BiasedEngine() body = client(engine).post("/decide", json={**ROW, "orderings": "all"}, headers=AUTH).json() rows = engine.shared_calls[0] orders = [tuple(o["id"] for o in r["options"]) for r in rows] assert len(orders) == 6 and len(set(orders)) == 6 and orders[0] == ("casual", "date", "booty") assert body["combined"]["orderings"] == 6 and body["combined"]["method"] == "all" def test_all_above_four_options_is_422_before_the_engine_runs(): engine = BiasedEngine() five = [{"id": f"o{i}", "description": f"Option {i}"} for i in range(5)] r = client(engine).post("/decide", json={**ROW, "options": five, "orderings": "all"}, headers=AUTH) assert r.status_code == 422 and "rotations" in r.json()["error"]["message"] assert engine.shared_calls == [] def test_a_mixed_shared_request_is_one_engine_call_with_results_in_request_order(): engine = BiasedEngine() body = {"state": ROW["state"], "decisions": [ {"id": "plain", "question": "Q1?", "options": OPTS}, {"id": "avg", "question": "Q2?", "options": OPTS, "orderings": "rotations"}, {"id": "plain2", "question": "Q3?", "options": OPTS[:2]}]} out = client(engine).post("/decide/shared", json=body, headers=AUTH).json() assert len(engine.shared_calls) == 1 assert [r["id"] for r in engine.shared_calls[0]] == ["plain", "avg#o0", "avg#o1", "avg#o2", "plain2"] ids = [r["id"] for r in out["results"]] assert ids == ["plain", "avg", "plain2"] assert out["results"][0] == engine._score(engine.shared_calls[0][0]) # plain results keep the old shape assert "combined" in out["results"][1] and "combined" not in out["results"][2] assert out["timing"] == {"batch_size": 5} def test_expanded_rows_count_toward_the_cap(): engine = BiasedEngine() r = client(engine, max_decisions=5).post("/decide/shared", json={"state": "s", "decisions": [ {"id": "a", "question": "Q?", "options": OPTS, "orderings": "rotations"}, {"id": "b", "question": "Q?", "options": OPTS, "orderings": "rotations"}]}, headers=AUTH) assert r.status_code == 422 and "6 scored rows" in r.json()["error"]["message"] assert engine.shared_calls == [] def test_workload_with_orderings_is_422_and_workload_alone_still_calibrates(): engine = BiasedEngine() c = client(engine, calibration={"triage": 2.0}) r = c.post("/decide", json={**ROW, "orderings": "rotations", "workload": "triage"}, headers=AUTH) assert r.status_code == 422 and engine.shared_calls == [] assert "calibrated" in c.post("/decide", json={**ROW, "workload": "triage"}, headers=AUTH).json() def test_an_unknown_orderings_value_is_422(): r = client(BiasedEngine()).post("/decide", json={**ROW, "orderings": "shuffle"}, headers=AUTH) assert r.status_code == 422