"""Order-averaging acceptance through the SERVICE (0.1.3): the same 252 labelled rows as the spike (SemIf authored144 + perturbations108), each scored twice: plain /decide (the caller's order), and /decide with orderings=rotations (combined.top). Paired group bootstrap for the accuracy delta. SEMIF_DIR=... SEMIF_URL=... SEMIF_TOKEN=... uv run --with httpx python averaging.py out.json """ import json, os, random, statistics as st, sys from pathlib import Path import httpx S, U = Path(os.environ["SEMIF_DIR"]), os.environ["SEMIF_URL"] H = {"Authorization": f"Bearer {os.environ['SEMIF_TOKEN']}"} rows = [] for name in ("authored144", "perturbations108"): rows += [dict(json.loads(l), set=name) for l in (S / f"benchmarks/data/{name}.jsonl").read_text().splitlines() if l.strip()] out = [] with httpx.Client(timeout=120) as c: for r in rows: base = {k: r[k] for k in ("id", "state", "question", "options")} plain = c.post(f"{U}/decide", headers=H, json=base).json() avg = c.post(f"{U}/decide", headers=H, json={**base, "orderings": "rotations"}).json() gold = r["options"][r["label"]]["id"] top_plain = plain["option_ids"][plain["probabilities"].index(max(plain["probabilities"]))] out.append({"group": r["group_id"], "plain": top_plain == gold, "rotations": avg["combined"]["top"] == gold, "agreement": avg["combined"]["agreement"]}) groups = {} for o in out: groups.setdefault(o["group"], []).append(o) rng, keys, deltas = random.Random(7), list(groups), [] for _ in range(10000): sample = [o for g in (rng.choice(keys) for _ in keys) for o in groups[g]] deltas.append(st.fmean(o["rotations"] for o in sample) - st.fmean(o["plain"] for o in sample)) deltas.sort() unan = [o for o in out if o["agreement"] == 1.0] split = [o for o in out if o["agreement"] < 1.0] report = {"rows": len(out), "groups": len(groups), "accuracy_plain": round(st.fmean(o["plain"] for o in out), 4), "accuracy_rotations": round(st.fmean(o["rotations"] for o in out), 4), "delta_95ci": [round(deltas[250], 4), round(deltas[9750], 4)], "unanimous": {"rows": len(unan), "accuracy": round(st.fmean(o["rotations"] for o in unan), 4)}, "split": {"rows": len(split), "accuracy": round(st.fmean(o["rotations"] for o in split), 4) if split else None}} json.dump(report, open(sys.argv[1], "w"), indent=1) print(json.dumps(report))