"""SPIKE (2026-09-27): does averaging over option orderings help, and is agreement a useful ambiguity signal? Throwaway measurement, no service change: every labelled row goes out as ONE /decide/shared request carrying all 3! = 6 orderings of its options. Data: SemIf authored144 + perturbations108 (labelled, 3 options each), pinned commit. Conditions derived from the same 6 scored orderings per row: single-original the caller's own order (what the service does today) single-expected mean accuracy over the 6 single orderings (a caller's expected luck) rotations the 3 cyclic shifts of the original order, log-mean combined all-6 all 6 permutations, log-mean combined Paired group-bootstrap (resampling group_id, 10k) for each combined condition minus single-original, so the delta is judged against its own sampling noise. SEMIF_DIR=... SEMIF_URL=... SEMIF_TOKEN=... uv run --with httpx python averaging_spike.py out.json """ import itertools, json, math, os, random, statistics as st, sys, time from pathlib import Path import httpx S, U = Path(os.environ["SEMIF_DIR"]), os.environ["SEMIF_URL"] H = {"Authorization": f"Bearer {os.environ['SEMIF_TOKEN']}"} rows = [] for name in ("authored144", "perturbations108"): rows += [dict(json.loads(l), set=name) for l in (S / f"benchmarks/data/{name}.jsonl").read_text().splitlines() if l.strip()] def logmean(dists): # dists: list of {option_id: p}; mean log p per option, renormalised ids = dists[0].keys() m = {k: st.fmean(math.log(max(d[k], 1e-12)) for d in dists) for k in ids} top = max(m.values()) z = sum(math.exp(v - top) for v in m.values()) return {k: math.exp(v - top) / z for k, v in m.items()} argmax = lambda d: max(d, key=d.get) out, t0 = [], time.time() for r in rows: opts = r["options"] perms = list(itertools.permutations(range(len(opts)))) body = {"state": r["state"], "decisions": [ {"id": f"p{i}", "question": r["question"], "options": [opts[j] for j in p]} for i, p in enumerate(perms)]} res = httpx.post(f"{U}/decide/shared", headers=H, json=body, timeout=120) res.raise_for_status() dists = [dict(zip(x["option_ids"], x["probabilities"])) for x in res.json()["results"]] by_perm = dict(zip(perms, dists)) ident = tuple(range(len(opts))) rot = [tuple((k + i) % len(opts) for k in ident) for i in range(len(opts))] gold = opts[r["label"]]["id"] rot_d, all_d = logmean([by_perm[p] for p in rot]), logmean(dists) out.append({ "id": r["id"], "set": r["set"], "group": r["group_id"], "gold": gold, "single_original": argmax(by_perm[ident]) == gold, "single_expected": st.fmean(argmax(d) == gold for d in dists), "rotations": argmax(rot_d) == gold, "all6": argmax(all_d) == gold, "agree_rot": sum(argmax(by_perm[p]) == argmax(rot_d) for p in rot) / len(rot), "agree_all6": sum(argmax(d) == argmax(all_d) for d in dists) / len(dists), "first_position_wins": sum(argmax(d) == opts[p[0]]["id"] for p, d in by_perm.items()) / len(perms), }) wall = time.time() - t0 def boot(key, reps=10000, seed=7): groups = {} for o in out: groups.setdefault(o["group"], []).append(o) keys, rng, deltas = list(groups), random.Random(seed), [] for _ in range(reps): sample = [o for g in (rng.choice(keys) for _ in keys) for o in groups[g]] deltas.append(st.fmean(o[key] for o in sample) - st.fmean(o["single_original"] for o in sample)) deltas.sort() return round(deltas[int(0.025 * reps)], 4), round(deltas[int(0.975 * reps)], 4) acc = lambda key, sub=out: round(st.fmean(o[key] for o in sub), 4) report = {"rows": len(out), "groups": len({o['group'] for o in out}), "wall_s": round(wall, 1), "accuracy": {k: acc(k) for k in ("single_original", "single_expected", "rotations", "all6")}, "delta_vs_single_original_95ci": {k: boot(k) for k in ("rotations", "all6")}, "first_position_win_rate_mean": acc("first_position_wins"), "by_set": {s: {k: acc(k, [o for o in out if o["set"] == s]) for k in ("single_original", "rotations", "all6")} for s in ("authored144", "perturbations108")}} for key in ("agree_rot", "agree_all6"): unan = [o for o in out if o[key] == 1.0] split = [o for o in out if o[key] < 1.0] cond = "rotations" if key == "agree_rot" else "all6" report[f"{key}: accuracy when unanimous vs split"] = { "unanimous": {"rows": len(unan), "accuracy": acc(cond, unan) if unan else None}, "split": {"rows": len(split), "accuracy": acc(cond, split) if split else None}} json.dump({"report": report, "rows": out}, open(sys.argv[1], "w"), indent=1) print(json.dumps(report, indent=1))