"""Score intern-decision-serve acceptance runs of the 2026-09-30 bench harness (bench_sets.py --backend semif, i.e. through the service's semif-compatible API) exactly as the bench's analyze.py scored them, and compare them row by row with the bench's own native Intern-Decision rows. python3 compare.py --ours r1/sets.json r2/sets.json r3/sets.json \ --bench ../../semif-serve/bench-jev-2026-09-30/raw/out/intern-decision-4b-native > compare.json Definitions (from analyze.py): a row counts when ok and labelled; right = top in gold. pooled = authored144 + cicada-w1 + wyrd evidence rows (259). The negative control is read against the same run's single condition. """ from __future__ import annotations import argparse import json import statistics as st from pathlib import Path SETS = ("authored144", "perturbations108", "cicada-w1", "cicada-w2", "wyrd") POOLED = ("authored144", "cicada-w1", "wyrd") NAMES = ("pooled",) + SETS + ("wyrd:place", "wyrd:place2", "wyrd:exit", "wyrd:exit2") def rows(run: dict, cond: str) -> dict: c = run["conditions"].get(cond) return {x["id"]: x for x in c["rows"]} if c else {} def acc(rs) -> tuple[int, int]: lab = [x for x in rs if x["ok"] and x["gold"]] return sum(x["top"] in x["gold"] for x in lab), len(lab) def set_scores(run: dict) -> dict: out = {} for cond in ("single", "rotations", "multifield"): rr = list(rows(run, cond).values()) if not rr: continue for name in NAMES: base, _, dec = name.partition(":") wanted = set(POOLED) if base == "pooled" else {base} sub = [x for x in rr if x["set"] in wanted and (not dec or x["decision"] == dec) and x["cond"] == "evidence"] c, n = acc(sub) if n: out[f"{cond}/{name}"] = [c, n] out[f"{cond}/failures"] = sum(not x["ok"] for x in rr) return out def negative(run: dict) -> dict: s, neg = rows(run, "single"), rows(run, "negative") ids = [i for i in neg if i in s and s[i]["ok"] and neg[i]["ok"]] return {"n": len(ids), "same_top_as_unrotated": sum(neg[i]["top"] == s[i]["top"] for i in ids), "follows_description": sum(neg[i]["top"] == neg[i]["gold_desc_id"] for i in ids), "vs_original_gold": sum(neg[i]["top"] in neg[i]["gold"] for i in ids)} def pdiff(a: dict, b: dict) -> float: return max(abs(x - y) for x, y in zip(a["probs"], b["probs"])) def pair(a: dict, b: dict, cond_a: str, cond_b: str) -> dict: ra, rb = rows(a, cond_a), rows(b, cond_b) ids = [i for i in ra if i in rb and ra[i]["ok"] and rb[i]["ok"]] flips = [i for i in ids if ra[i]["top"] != rb[i]["top"]] return {"rows": len(ids), "top_differs": len(flips), "flipped_ids": flips[:20], "max_dp": max((pdiff(ra[i], rb[i]) for i in ids), default=None)} def main() -> None: ap = argparse.ArgumentParser() ap.add_argument("--ours", nargs="+", required=True) ap.add_argument("--bench", required=True, help="raw/out/intern-decision-4b-native") args = ap.parse_args() ours = {Path(p).parent.name or p: json.loads(Path(p).read_text()) for p in args.ours} bench = {d.name: json.loads((d / "sets.json").read_text()) for d in sorted(Path(args.bench).glob("r*")) if (d / "sets.json").exists()} report = {"ours": {k: {"scores": set_scores(v), "negative": negative(v), "a_vs_a_in_process_authored144": pair(v, v, "single", "repeat")} for k, v in ours.items()}, "bench": {k: {"scores": set_scores(v), "negative": negative(v)} for k, v in bench.items()}} def bench_with(cond: str) -> str: """The first bench repeat that ran `cond` (multifield was added after the bench's r1).""" return next(k for k in sorted(bench) if cond in bench[k]["conditions"]) report["vs_bench_row_by_row"] = {k: {cond: {"bench_repeat": bench_with(cond), **pair(v, bench[bench_with(cond)], cond, cond)} for cond in ("single", "rotations", "negative", "multifield")} for k, v in ours.items()} names = sorted(ours) report["across_restarts"] = {f"{a}~{b}": {cond: pair(ours[a], ours[b], cond, cond) for cond in ("single", "rotations")} for i, a in enumerate(names) for b in names[i + 1:]} keys = sorted({k for v in report["ours"].values() for k in v["scores"] if not k.endswith("/failures")}) report["summary"] = {k: {"ours_median": st.median(v["scores"][k][0] for v in report["ours"].values() if k in v["scores"]), "ours_all": [v["scores"][k][0] for v in report["ours"].values() if k in v["scores"]], "bench_all": [v["scores"][k][0] for v in report["bench"].values() if k in v["scores"]], "n": next(v["scores"][k][1] for v in report["ours"].values() if k in v["scores"])} for k in keys} print(json.dumps(report, indent=1)) if __name__ == "__main__": main()