Files
esh-pfi-infrastructure/services/semif-serve/bench-jev-2026-09-30/code/analyze.py
T
vh 475d6d6bcb docs: Jev candidate bench vs SemIf (fv-ml1 GPU 3) — Intern-Decision-4B is the replacement if SemIf is displaced
Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231,
hard 0.613) reproduced exactly; negative control and a 4-restart noise floor
(0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with-
rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one
ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench
rank does not transfer. Raw per-item data kept out of git.
2026-09-30 05:00:45 -07:00

361 lines
16 KiB
Python

"""Jev-candidate bench (2026-09-30): turn the raw runs into the tables of the results doc.
python3 analyze.py <out dir> > summary.json (and prints markdown tables to stderr)
Every cell is a median over the repeats (one repeat = one process lifetime) with min..max.
The noise floors come from the same data:
* A-vs-A inside a process: authored144 "single" vs "repeat"
* across restarts: every pair of repeats' "single" rows (label flips, max |dp|)
Paired candidate-vs-SemIf comparisons use each row's majority top over the repeats (ties
never arise with 3 repeats and a strict majority rule; a row without a majority keeps r1).
"""
from __future__ import annotations
import csv
import itertools
import json
import math
import random
import statistics as st
import sys
from collections import Counter, defaultdict
from pathlib import Path
OUT = Path(sys.argv[1])
JB = OUT.parent / "src/jevbench-v1.2.16/datasets/public"
if not JB.exists():
JB = Path(sys.argv[2]) if len(sys.argv) > 2 else JB
TIERS = {}
for t in ("easy", "original", "hard"):
for line in (JB / f"{t}.jsonl").read_text().splitlines():
if line.strip():
TIERS[json.loads(line)["id"]] = t
def med(xs):
xs = [x for x in xs if x is not None]
if not xs:
return None
return {"median": st.median(xs), "min": min(xs), "max": max(xs), "n": len(xs)}
def load_runs(label):
runs = {}
for d in sorted((OUT / label).glob("r*")):
r = {"dir": d}
for name in ("sets", "shape"):
p = d / f"{name}.json"
r[name] = json.loads(p.read_text()) if p.exists() else None
p = d / "jevbench/results.jsonl"
r["jb"] = [json.loads(l) for l in p.read_text().splitlines() if l.strip()] if p.exists() else None
r["vram"] = []
p = d / "vram.csv"
if p.exists():
for row in csv.reader(p.read_text().splitlines()):
try:
r["vram"].append((float(row[0]), float(row[1])))
except (ValueError, IndexError):
pass
r["events"] = {}
p = d / "events.txt"
if p.exists():
for line in p.read_text().splitlines():
ts, name = line.split(" ", 1)
r["events"][name] = float(ts)
p = d / "up_at.txt"
if p.exists():
r["events"]["up"] = float(p.read_text().strip())
runs[d.name] = r
return runs
def jevbench(runs):
per = defaultdict(list)
for r in runs.values():
if not r["jb"]:
continue
c, n = Counter(), Counter()
for x in r["jb"]:
t = TIERS[x["task_id"]]
n[t] += 1
c[t] += bool(x["correct"])
for t in ("easy", "original", "hard"):
per[t].append(c[t] / n[t] if n[t] else None)
per[t + "_n"].append(c[t])
per["all"].append(sum(c.values()) / sum(n.values()))
per["all_n"].append(sum(c.values()))
per["failed"].append(sum(not x["ok"] for x in r["jb"]))
lat = [x["latency_s"] for x in r["jb"] if x.get("latency_s") is not None]
per["p50_ms"].append(st.median(lat) * 1000 if lat else None)
return {k: med(v) for k, v in per.items()}
def rows(run, cond):
s = run["sets"]
return {x["id"]: x for x in s["conditions"][cond]["rows"]} if s and cond in s["conditions"] else {}
SETS = ("authored144", "perturbations108", "cicada-w1", "cicada-w2", "wyrd")
POOLED = ("authored144", "cicada-w1", "wyrd") # the brief's three replaced-baseline sets, 259 rows
def acc(rs, key=None):
lab = [x for x in rs if x["ok"] and x["gold"]]
if key:
lab = [x for x in lab if key(x)]
return (sum(x["top"] in x["gold"] for x in lab), len(lab))
def set_table(runs):
out = {}
for cond in ("single", "rotations", "multifield"):
for name in ("pooled",) + SETS + ("wyrd:place", "wyrd:place2", "wyrd:exit", "wyrd:exit2"):
base, _, dec = name.partition(":")
wanted = set(POOLED) if base == "pooled" else {base}
vals, blind, ctrl, fails = [], [], [], []
for r in runs.values():
rr = list(rows(r, cond).values())
if not rr:
continue
sub = [x for x in rr if x["set"] in wanted and (not dec or x["decision"] == dec)]
c, n = acc([x for x in sub if x["cond"] == "evidence"])
vals.append((c, n))
b = acc([x for x in sub if x["cond"] == "blind"])
blind.append(b)
k = acc([x for x in sub if x["cond"] == "evidence" and x["tag"] == "control"])
ctrl.append(k)
fails.append(sum(not x["ok"] for x in sub))
if vals and vals[0][1]:
out[f"{cond}/{name}"] = {
"correct": med([c for c, _ in vals]), "n": vals[0][1],
"acc": med([c / n for c, n in vals if n]),
"blind_correct": med([c for c, _ in blind]) if blind and blind[0][1] else None,
"controls": med([c for c, _ in ctrl]) if ctrl and ctrl[0][1] else None,
"controls_n": ctrl[0][1] if ctrl else None,
"failures": med(fails)}
# agreement signal (rotations): accuracy of unanimous vs split rows, authored144+perturbations108
for r in runs.values():
rr = [x for x in rows(r, "rotations").values() if x["set"] in ("authored144", "perturbations108") and x["ok"]]
if rr:
un = [x for x in rr if x.get("agreement") == 1.0]
sp = [x for x in rr if x.get("agreement") is not None and x["agreement"] < 1.0]
out.setdefault("agreement_signal", []).append({"unanimous": acc(un), "split": acc(sp)})
return out
def top_of(x):
return x["top"] if x.get("ok") else None
def pdiff(a, b):
pa = dict(zip(a["option_ids"], a["probs"]))
pb = dict(zip(b["option_ids"], b["probs"]))
return max(abs(pa[i] - pb[i]) for i in pa)
def floors(runs):
out = {}
# A-vs-A inside the process
aa = []
for r in runs.values():
s, rep = rows(r, "single"), rows(r, "repeat")
ids = [i for i in rep if i in s and s[i]["ok"] and rep[i]["ok"]]
if ids:
aa.append({"flips": sum(s[i]["top"] != rep[i]["top"] for i in ids),
"max_dp": max(pdiff(s[i], rep[i]) for i in ids), "n": len(ids)})
out["a_vs_a_in_process"] = aa
# across restarts, every row of "single" and of "rotations"
keys = sorted(runs)
for cond in ("single", "rotations"):
pairs = []
for a, b in itertools.combinations(keys, 2):
ra, rb = rows(runs[a], cond), rows(runs[b], cond)
ids = [i for i in ra if i in rb and ra[i]["ok"] and rb[i]["ok"]]
if ids:
flips = [i for i in ids if ra[i]["top"] != rb[i]["top"]]
pairs.append({"pair": f"{a}-{b}", "n": len(ids), "flips": len(flips),
"flipped_ids": flips[:10], "max_dp": max(pdiff(ra[i], rb[i]) for i in ids)})
out[f"cross_restart/{cond}"] = pairs
# rows whose top differs between ANY two repeats, per set (labelled evidence rows only)
tops = defaultdict(set)
meta = {}
for k in keys:
for i, x in rows(runs[k], cond).items():
if x["ok"]:
tops[i].add(x["top"])
meta[i] = x
unstable = [i for i, t in tops.items() if len(t) > 1]
out[f"unstable_rows/{cond}"] = {
s: sum(1 for i in unstable if meta[i]["set"] == s and meta[i]["cond"] == "evidence" and meta[i]["gold"])
for s in SETS} | {"pooled": sum(1 for i in unstable if meta[i]["set"] in POOLED
and meta[i]["cond"] == "evidence" and meta[i]["gold"]),
"repeats": len(keys)}
return out
def order_sensitivity(runs):
out = defaultdict(list)
for r in runs.values():
s = rows(r, "single")
for cond in ("reversed", "shuffled"):
o = rows(r, cond)
ids = [i for i in o if i in s and s[i]["ok"] and o[i]["ok"]]
if ids:
out[cond].append({"n": len(ids), "label_changes": sum(s[i]["top"] != o[i]["top"] for i in ids),
"max_dp": max(pdiff(s[i], o[i]) for i in ids),
"median_max_dp": st.median(pdiff(s[i], o[i]) for i in ids),
"acc": acc([o[i] for i in ids])})
return {k: {"label_changes": med([x["label_changes"] for x in v]), "n": v[0]["n"],
"max_dp": med([x["max_dp"] for x in v]), "median_row_max_dp": med([x["median_max_dp"] for x in v]),
"acc": med([x["acc"][0] for x in v])} for k, v in out.items()}
def negative(runs):
res = []
for r in runs.values():
s, neg = rows(r, "single"), rows(r, "negative")
ids = [i for i in neg if i in s and s[i]["ok"] and neg[i]["ok"]]
if ids:
res.append({"n": len(ids),
"same_top_as_unrotated": sum(neg[i]["top"] == s[i]["top"] for i in ids),
"vs_original_gold": sum(neg[i]["top"] in neg[i]["gold"] for i in ids),
"follows_description": sum(neg[i]["top"] == neg[i]["gold_desc_id"] for i in ids)})
return {k: med([x[k] for x in res]) for k in ("same_top_as_unrotated", "vs_original_gold", "follows_description")} | (
{"n": res[0]["n"]} if res else {})
def majority_tops(runs, cond):
votes = defaultdict(list)
meta = {}
for k in sorted(runs):
for i, x in rows(runs[k], cond).items():
votes[i].append(top_of(x))
meta[i] = x
out = {}
for i, v in votes.items():
c = Counter(t for t in v if t is not None).most_common()
out[i] = c[0][0] if c and c[0][1] > len(v) / 2 else v[0]
return out, meta
def mcnemar_exact(b, c):
n = b + c
if n == 0:
return 1.0
k = min(b, c)
p = sum(math.comb(n, i) for i in range(0, k + 1)) / 2 ** n
return min(1.0, 2 * p)
def paired(base_runs, cand_runs, cond, setname, cand_cond=None):
bt, meta = majority_tops(base_runs, cond)
ct, _ = majority_tops(cand_runs, cand_cond or cond)
wanted = set(POOLED) if setname == "pooled" else {setname.split(":")[0]}
ids = [i for i in bt if i in ct and meta[i]["set"] in wanted and meta[i]["cond"] == "evidence"
and meta[i]["gold"] and (":" not in setname or meta[i]["decision"] == setname.split(":")[1])]
if not ids:
return None
b_ok = {i: bt[i] in meta[i]["gold"] for i in ids}
c_ok = {i: ct[i] in meta[i]["gold"] for i in ids}
fixed = sum(c_ok[i] and not b_ok[i] for i in ids)
broken = sum(b_ok[i] and not c_ok[i] for i in ids)
groups = defaultdict(list)
for i in ids:
groups[meta[i]["group"]].append(i)
rng, keys, deltas = random.Random(7), list(groups), []
for _ in range(5000):
sample = [i for g in (rng.choice(keys) for _ in keys) for i in groups[g]]
deltas.append((sum(c_ok[i] for i in sample) - sum(b_ok[i] for i in sample)) / len(sample))
deltas.sort()
return {"n": len(ids), "semif": sum(b_ok.values()), "cand": sum(c_ok.values()), "fixed": fixed,
"broken": broken, "mcnemar_p": round(mcnemar_exact(fixed, broken), 4),
"delta_pts": round(100 * (sum(c_ok.values()) - sum(b_ok.values())) / len(ids), 1) if ids else None,
"delta_95ci_pts": [round(100 * deltas[125], 1), round(100 * deltas[4875], 1)] if ids else None}
def vram(runs):
res = []
for r in runs.values():
ev, v = r["events"], r["vram"]
if not v:
continue
rest = [m for t, m in v if "rest:start" in ev and ev["rest:start"] <= t <= ev.get("rest:end", 0)]
if not rest and "up" in ev: # r1 of the SemIf baseline predates the rest window
rest = [m for t, m in v if ev["up"] <= t <= ev["up"] + 1.5]
serve0 = ev.get("serve:start", 0)
sets0 = ev.get("sets:start", ev.get("up", serve0))
shape = r["shape"]
cap0 = None
if shape:
cap_ev = [t for name, t in shape["events"] if name.startswith("capacity:")]
cap0 = min(cap_ev) if cap_ev else None
work = [m for t, m in v if t >= sets0 and (cap0 is None or t < cap0) and t <= ev.get("serve:end", 1e18)]
capw = [m for t, m in v if cap0 is not None and t >= cap0 and t <= ev.get("serve:end", 1e18)]
jb = [m for t, m in v if ev.get("jevbench:start", 1e18) <= t <= ev.get("jevbench:end", 0)]
res.append({"rest_mib": st.median(rest) if rest else None, "peak_work_mib": max(work) if work else None,
"peak_capacity_mib": max(capw) if capw else None, "peak_jevbench_mib": max(jb) if jb else None})
return {k: med([x[k] for x in res]) for k in ("rest_mib", "peak_work_mib", "peak_capacity_mib", "peak_jevbench_mib")}
def latency(runs):
out = defaultdict(lambda: defaultdict(list))
cap = {}
for k, r in sorted(runs.items()):
sh = r["shape"]
if not sh:
continue
for name, s in sh["shapes"].items():
out[name]["e2e"].append(s["median_e2e_ms"])
out[name]["server"].append(s["median_server_ms"])
out[name]["run_medians"].extend(s["run_medians_e2e_ms"])
out[name]["tokens"].append(s["tokens"])
out[name]["non_200"].append(s["non_200"])
if sh.get("capacity"):
cap[k] = {lab: [(x["rows"], x["status"]) for x in series] for lab, series in sh["capacity"].items()}
return {name: {"e2e_ms": med(v["e2e"]), "server_ms": med(v["server"]),
"run_medians_ms": [min(v["run_medians"]), max(v["run_medians"])],
"tokens": v["tokens"][0], "non_200": sum(v["non_200"])} for name, v in out.items()}, cap
def parity(runs):
"""semif-format runs only: authored144 'single' against SemIf's committed torch predictions."""
p = Path(__file__).resolve().parent / "data/direct-authored144.jsonl"
ref = {x["id"]: x for x in map(json.loads, p.read_text().splitlines()) if x}
res = []
for r in runs.values():
s = rows(r, "single")
ids = [i for i in ref if i in s and s[i].get("prompt_sha256")]
if not ids:
continue
top_ref = {i: ref[i]["option_ids"][max(range(len(ref[i]["probabilities"])), key=ref[i]["probabilities"].__getitem__)]
for i in ids}
res.append({"n": len(ids), "top_agree": sum(s[i]["top"] == top_ref[i] for i in ids),
"prompt_sha_equal": sum(s[i]["prompt_sha256"] == ref[i]["prompt_sha256"] for i in ids),
"max_dp": max(max(abs(a - b) for a, b in zip(s[i]["probs"], ref[i]["probabilities"])) for i in ids)})
return res
def main():
labels = [p.name for p in sorted(OUT.iterdir()) if p.is_dir() and any(p.glob("r*/"))]
allruns = {l: load_runs(l) for l in labels}
summary = {}
base = allruns.get("semif-qwen35-4b")
for l, runs in allruns.items():
lat, cap = latency(runs)
s = {"repeats": sorted(runs), "jevbench": jevbench(runs), "sets": set_table(runs), "floors": floors(runs),
"order": order_sensitivity(runs), "negative": negative(runs), "vram": vram(runs),
"latency": lat, "capacity": cap, "parity_vs_upstream_semif": parity(runs)}
if base and l != "semif-qwen35-4b":
s["paired_vs_semif"] = {f"{cond}/{name}": paired(base, runs, cond, name)
for cond in ("single", "rotations")
for name in ("pooled",) + SETS + ("wyrd:place2", "wyrd:exit")}
# the cost question: the candidate at ONE ordering against SemIf WITH rotations
s["paired_vs_semif"].update({f"cand-single-vs-semif-rotations/{name}": paired(base, runs, "rotations", name, "single")
for name in ("pooled",) + SETS})
summary[l] = s
json.dump(summary, sys.stdout, indent=1, default=str)
if __name__ == "__main__":
main()