Files
esh-pfi-infrastructure/services/intern-decision-serve/acceptance/compare.py
T
vh a262477a61 feat(intern-decision): stack, DNS and GPU 3 acceptance for the SemIf replacement
stacks/intern-decision: compose (GPU 1, :8033, hard VRAM cap as the single .env knob,
healthcheck, Homepage group 'AI - Eval & Retrieval'), .env.example and README.
dns: intern-decision.fv.internal -> fv-ml1 (synced to ana/esh/nh3).
acceptance on fv-ml1 GPU 3, 3 fresh processes: bit-identical to the Jev bench's native rows
(pooled 240/259, Wyrd 79/84, 0/560 flips, Δp 0), negative control 10/122/14, 0 flips across
restarts; largest accepted request 200 at a 10,134 MiB card peak under a 9.25 GiB cap; 503 and
recovery proven at a tight cap. GPU 1 deploy held: nvidia-smi Free on GPU 1 is 15,442 MiB.
2026-09-30 09:38:00 -07:00

102 lines
5.0 KiB
Python

"""Score intern-decision-serve acceptance runs of the 2026-09-30 bench harness (bench_sets.py
--backend semif, i.e. through the service's semif-compatible API) exactly as the bench's analyze.py
scored them, and compare them row by row with the bench's own native Intern-Decision rows.
python3 compare.py --ours r1/sets.json r2/sets.json r3/sets.json \
--bench ../../semif-serve/bench-jev-2026-09-30/raw/out/intern-decision-4b-native > compare.json
Definitions (from analyze.py): a row counts when ok and labelled; right = top in gold. pooled =
authored144 + cicada-w1 + wyrd evidence rows (259). The negative control is read against the same
run's single condition.
"""
from __future__ import annotations
import argparse
import json
import statistics as st
from pathlib import Path
SETS = ("authored144", "perturbations108", "cicada-w1", "cicada-w2", "wyrd")
POOLED = ("authored144", "cicada-w1", "wyrd")
NAMES = ("pooled",) + SETS + ("wyrd:place", "wyrd:place2", "wyrd:exit", "wyrd:exit2")
def rows(run: dict, cond: str) -> dict:
c = run["conditions"].get(cond)
return {x["id"]: x for x in c["rows"]} if c else {}
def acc(rs) -> tuple[int, int]:
lab = [x for x in rs if x["ok"] and x["gold"]]
return sum(x["top"] in x["gold"] for x in lab), len(lab)
def set_scores(run: dict) -> dict:
out = {}
for cond in ("single", "rotations", "multifield"):
rr = list(rows(run, cond).values())
if not rr:
continue
for name in NAMES:
base, _, dec = name.partition(":")
wanted = set(POOLED) if base == "pooled" else {base}
sub = [x for x in rr if x["set"] in wanted and (not dec or x["decision"] == dec) and x["cond"] == "evidence"]
c, n = acc(sub)
if n:
out[f"{cond}/{name}"] = [c, n]
out[f"{cond}/failures"] = sum(not x["ok"] for x in rr)
return out
def negative(run: dict) -> dict:
s, neg = rows(run, "single"), rows(run, "negative")
ids = [i for i in neg if i in s and s[i]["ok"] and neg[i]["ok"]]
return {"n": len(ids), "same_top_as_unrotated": sum(neg[i]["top"] == s[i]["top"] for i in ids),
"follows_description": sum(neg[i]["top"] == neg[i]["gold_desc_id"] for i in ids),
"vs_original_gold": sum(neg[i]["top"] in neg[i]["gold"] for i in ids)}
def pdiff(a: dict, b: dict) -> float:
return max(abs(x - y) for x, y in zip(a["probs"], b["probs"]))
def pair(a: dict, b: dict, cond_a: str, cond_b: str) -> dict:
ra, rb = rows(a, cond_a), rows(b, cond_b)
ids = [i for i in ra if i in rb and ra[i]["ok"] and rb[i]["ok"]]
flips = [i for i in ids if ra[i]["top"] != rb[i]["top"]]
return {"rows": len(ids), "top_differs": len(flips), "flipped_ids": flips[:20],
"max_dp": max((pdiff(ra[i], rb[i]) for i in ids), default=None)}
def main() -> None:
ap = argparse.ArgumentParser()
ap.add_argument("--ours", nargs="+", required=True)
ap.add_argument("--bench", required=True, help="raw/out/intern-decision-4b-native")
args = ap.parse_args()
ours = {Path(p).parent.name or p: json.loads(Path(p).read_text()) for p in args.ours}
bench = {d.name: json.loads((d / "sets.json").read_text()) for d in sorted(Path(args.bench).glob("r*"))
if (d / "sets.json").exists()}
report = {"ours": {k: {"scores": set_scores(v), "negative": negative(v),
"a_vs_a_in_process_authored144": pair(v, v, "single", "repeat")} for k, v in ours.items()},
"bench": {k: {"scores": set_scores(v), "negative": negative(v)} for k, v in bench.items()}}
def bench_with(cond: str) -> str:
"""The first bench repeat that ran `cond` (multifield was added after the bench's r1)."""
return next(k for k in sorted(bench) if cond in bench[k]["conditions"])
report["vs_bench_row_by_row"] = {k: {cond: {"bench_repeat": bench_with(cond), **pair(v, bench[bench_with(cond)], cond, cond)}
for cond in ("single", "rotations", "negative", "multifield")}
for k, v in ours.items()}
names = sorted(ours)
report["across_restarts"] = {f"{a}~{b}": {cond: pair(ours[a], ours[b], cond, cond) for cond in ("single", "rotations")}
for i, a in enumerate(names) for b in names[i + 1:]}
keys = sorted({k for v in report["ours"].values() for k in v["scores"] if not k.endswith("/failures")})
report["summary"] = {k: {"ours_median": st.median(v["scores"][k][0] for v in report["ours"].values() if k in v["scores"]),
"ours_all": [v["scores"][k][0] for v in report["ours"].values() if k in v["scores"]],
"bench_all": [v["scores"][k][0] for v in report["bench"].values() if k in v["scores"]],
"n": next(v["scores"][k][1] for v in report["ours"].values() if k in v["scores"])}
for k in keys}
print(json.dumps(report, indent=1))
if __name__ == "__main__":
main()