stacks/intern-decision: compose (GPU 1, :8033, hard VRAM cap as the single .env knob, healthcheck, Homepage group 'AI - Eval & Retrieval'), .env.example and README. dns: intern-decision.fv.internal -> fv-ml1 (synced to ana/esh/nh3). acceptance on fv-ml1 GPU 3, 3 fresh processes: bit-identical to the Jev bench's native rows (pooled 240/259, Wyrd 79/84, 0/560 flips, Δp 0), negative control 10/122/14, 0 flips across restarts; largest accepted request 200 at a 10,134 MiB card peak under a 9.25 GiB cap; 503 and recovery proven at a tight cap. GPU 1 deploy held: nvidia-smi Free on GPU 1 is 15,442 MiB.
102 lines
5.0 KiB
Python
102 lines
5.0 KiB
Python
"""Score intern-decision-serve acceptance runs of the 2026-09-30 bench harness (bench_sets.py
|
|
--backend semif, i.e. through the service's semif-compatible API) exactly as the bench's analyze.py
|
|
scored them, and compare them row by row with the bench's own native Intern-Decision rows.
|
|
|
|
python3 compare.py --ours r1/sets.json r2/sets.json r3/sets.json \
|
|
--bench ../../semif-serve/bench-jev-2026-09-30/raw/out/intern-decision-4b-native > compare.json
|
|
|
|
Definitions (from analyze.py): a row counts when ok and labelled; right = top in gold. pooled =
|
|
authored144 + cicada-w1 + wyrd evidence rows (259). The negative control is read against the same
|
|
run's single condition.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import statistics as st
|
|
from pathlib import Path
|
|
|
|
SETS = ("authored144", "perturbations108", "cicada-w1", "cicada-w2", "wyrd")
|
|
POOLED = ("authored144", "cicada-w1", "wyrd")
|
|
NAMES = ("pooled",) + SETS + ("wyrd:place", "wyrd:place2", "wyrd:exit", "wyrd:exit2")
|
|
|
|
|
|
def rows(run: dict, cond: str) -> dict:
|
|
c = run["conditions"].get(cond)
|
|
return {x["id"]: x for x in c["rows"]} if c else {}
|
|
|
|
|
|
def acc(rs) -> tuple[int, int]:
|
|
lab = [x for x in rs if x["ok"] and x["gold"]]
|
|
return sum(x["top"] in x["gold"] for x in lab), len(lab)
|
|
|
|
|
|
def set_scores(run: dict) -> dict:
|
|
out = {}
|
|
for cond in ("single", "rotations", "multifield"):
|
|
rr = list(rows(run, cond).values())
|
|
if not rr:
|
|
continue
|
|
for name in NAMES:
|
|
base, _, dec = name.partition(":")
|
|
wanted = set(POOLED) if base == "pooled" else {base}
|
|
sub = [x for x in rr if x["set"] in wanted and (not dec or x["decision"] == dec) and x["cond"] == "evidence"]
|
|
c, n = acc(sub)
|
|
if n:
|
|
out[f"{cond}/{name}"] = [c, n]
|
|
out[f"{cond}/failures"] = sum(not x["ok"] for x in rr)
|
|
return out
|
|
|
|
|
|
def negative(run: dict) -> dict:
|
|
s, neg = rows(run, "single"), rows(run, "negative")
|
|
ids = [i for i in neg if i in s and s[i]["ok"] and neg[i]["ok"]]
|
|
return {"n": len(ids), "same_top_as_unrotated": sum(neg[i]["top"] == s[i]["top"] for i in ids),
|
|
"follows_description": sum(neg[i]["top"] == neg[i]["gold_desc_id"] for i in ids),
|
|
"vs_original_gold": sum(neg[i]["top"] in neg[i]["gold"] for i in ids)}
|
|
|
|
|
|
def pdiff(a: dict, b: dict) -> float:
|
|
return max(abs(x - y) for x, y in zip(a["probs"], b["probs"]))
|
|
|
|
|
|
def pair(a: dict, b: dict, cond_a: str, cond_b: str) -> dict:
|
|
ra, rb = rows(a, cond_a), rows(b, cond_b)
|
|
ids = [i for i in ra if i in rb and ra[i]["ok"] and rb[i]["ok"]]
|
|
flips = [i for i in ids if ra[i]["top"] != rb[i]["top"]]
|
|
return {"rows": len(ids), "top_differs": len(flips), "flipped_ids": flips[:20],
|
|
"max_dp": max((pdiff(ra[i], rb[i]) for i in ids), default=None)}
|
|
|
|
|
|
def main() -> None:
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--ours", nargs="+", required=True)
|
|
ap.add_argument("--bench", required=True, help="raw/out/intern-decision-4b-native")
|
|
args = ap.parse_args()
|
|
ours = {Path(p).parent.name or p: json.loads(Path(p).read_text()) for p in args.ours}
|
|
bench = {d.name: json.loads((d / "sets.json").read_text()) for d in sorted(Path(args.bench).glob("r*"))
|
|
if (d / "sets.json").exists()}
|
|
report = {"ours": {k: {"scores": set_scores(v), "negative": negative(v),
|
|
"a_vs_a_in_process_authored144": pair(v, v, "single", "repeat")} for k, v in ours.items()},
|
|
"bench": {k: {"scores": set_scores(v), "negative": negative(v)} for k, v in bench.items()}}
|
|
def bench_with(cond: str) -> str:
|
|
"""The first bench repeat that ran `cond` (multifield was added after the bench's r1)."""
|
|
return next(k for k in sorted(bench) if cond in bench[k]["conditions"])
|
|
report["vs_bench_row_by_row"] = {k: {cond: {"bench_repeat": bench_with(cond), **pair(v, bench[bench_with(cond)], cond, cond)}
|
|
for cond in ("single", "rotations", "negative", "multifield")}
|
|
for k, v in ours.items()}
|
|
names = sorted(ours)
|
|
report["across_restarts"] = {f"{a}~{b}": {cond: pair(ours[a], ours[b], cond, cond) for cond in ("single", "rotations")}
|
|
for i, a in enumerate(names) for b in names[i + 1:]}
|
|
keys = sorted({k for v in report["ours"].values() for k in v["scores"] if not k.endswith("/failures")})
|
|
report["summary"] = {k: {"ours_median": st.median(v["scores"][k][0] for v in report["ours"].values() if k in v["scores"]),
|
|
"ours_all": [v["scores"][k][0] for v in report["ours"].values() if k in v["scores"]],
|
|
"bench_all": [v["scores"][k][0] for v in report["bench"].values() if k in v["scores"]],
|
|
"n": next(v["scores"][k][1] for v in report["ours"].values() if k in v["scores"])}
|
|
for k in keys}
|
|
print(json.dumps(report, indent=1))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|