"""semif-serve acceptance on the real card. Contract § Acceptance. Needs SemIf's authored144 rows and their committed torch predictions (same model revision, torch 2.10.0+cu128) from a SemIf checkout at the pinned commit: SEMIF_DIR= SEMIF_URL=http://10.251.50.54:8032 SEMIF_TOKEN=... \ uv run --with httpx python accept.py out.json 1 parity our /decide vs their direct-authored144 predictions: top choice, max |Δp|, prompt hash 2 noise floor the same 144 rows again (A-vs-A) 3 negative option descriptions rotated one place: agreement with the reference MUST drop 4 shared one state, many criteria through /decide/shared vs /decide on the same rows 5 speed 21 binary criteria over one state, shared vs 21 sequential /decide, 3 runs, after warm-up """ import json import os import statistics import sys import time from pathlib import Path import httpx URL, TOKEN, SEMIF = os.environ["SEMIF_URL"], os.environ["SEMIF_TOKEN"], Path(os.environ["SEMIF_DIR"]) H = {"Authorization": f"Bearer {TOKEN}"} client = httpx.Client(timeout=300) def row(r): return {k: r[k] for k in ("id", "state", "question", "options")} def decide(r): resp = client.post(f"{URL}/decide", json=row(r), headers=H) resp.raise_for_status() return resp.json() def argmax(p): return max(range(len(p)), key=p.__getitem__) def compare(ours, ref): agree = sum(argmax(ours[i]["probabilities"]) == argmax(ref[i]["probabilities"]) for i in ref) gap = max(max(abs(a - b) for a, b in zip(ours[i]["probabilities"], ref[i]["probabilities"])) for i in ref) return {"rows": len(ref), "top_choice_agree": agree, "max_abs_prob_gap": gap} rows = [json.loads(l) for l in (SEMIF / "benchmarks/data/authored144.jsonl").read_text().splitlines() if l.strip()] ref = {p["id"]: p for p in map(json.loads, (SEMIF / "results/raw/predictions/direct-authored144.jsonl").read_text().splitlines()) if p} report = {"url": URL, "health": client.get(f"{URL}/health").json()} run_a = {r["id"]: decide(r) for r in rows} run_b = {r["id"]: decide(r) for r in rows} report["1_parity_vs_upstream"] = compare(run_a, ref) report["1_prompt_sha256_equal"] = sum(run_a[i]["prompt_sha256"] == ref[i]["prompt_sha256"] for i in ref) report["2_noise_floor_a_vs_b"] = compare(run_a, run_b) rotated = [] for r in rows: descs = [o["description"] for o in r["options"]] descs = descs[1:] + descs[:1] rotated.append({**r, "options": [{**o, "description": d} for o, d in zip(r["options"], descs)]}) report["3_negative_rotated_options"] = compare({r["id"]: decide(r) for r in rotated}, ref) # 4: group authored144 by identical state; score every multi-row group both ways. groups = {} for r in rows: groups.setdefault(json.dumps(r["state"], sort_keys=True), []).append(r) multi = [g for g in groups.values() if len(g) > 1] shared_out = {} for g in multi: resp = client.post(f"{URL}/decide/shared", headers=H, json={ "state": g[0]["state"], "decisions": [{k: r[k] for k in ("id", "question", "options")} for r in g]}) resp.raise_for_status() shared_out.update({res["id"]: res for res in resp.json()["results"]}) report["4_shared_vs_direct"] = {"groups": len(multi), **compare(shared_out, {i: run_a[i] for i in shared_out})} # 5: 21 binary criteria over one state state = rows[0]["state"] crit = [{"id": f"c{i}", "question": f"Does the evidence mention item number {i}?", "options": [{"id": "yes", "description": "Yes"}, {"id": "no", "description": "No"}]} for i in range(21)] for _ in range(2): # warm-up client.post(f"{URL}/decide/shared", headers=H, json={"state": state, "decisions": crit}).raise_for_status() shared_s, seq_s, prefix = [], [], None for _ in range(3): t = time.perf_counter() resp = client.post(f"{URL}/decide/shared", headers=H, json={"state": state, "decisions": crit}) resp.raise_for_status() shared_s.append(time.perf_counter() - t) prefix = resp.json()["timing"]["prefix_tokens"] t = time.perf_counter() for c in crit: decide({**c, "state": state}) seq_s.append(time.perf_counter() - t) report["5_speed_21_binary"] = { "prefix_tokens": prefix, "shared_s": {"runs": shared_s, "median": statistics.median(shared_s)}, "sequential_decide_s": {"runs": seq_s, "median": statistics.median(seq_s)}, } json.dump(report, open(sys.argv[1], "w"), indent=1) print(json.dumps({k: v for k, v in report.items() if k != "health"}, indent=1))