Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231, hard 0.613) reproduced exactly; negative control and a 4-restart noise floor (0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with- rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench rank does not transfer. Raw per-item data kept out of git.
361 lines
16 KiB
Python
361 lines
16 KiB
Python
"""Jev-candidate bench (2026-09-30): turn the raw runs into the tables of the results doc.
|
|
|
|
python3 analyze.py <out dir> > summary.json (and prints markdown tables to stderr)
|
|
|
|
Every cell is a median over the repeats (one repeat = one process lifetime) with min..max.
|
|
The noise floors come from the same data:
|
|
* A-vs-A inside a process: authored144 "single" vs "repeat"
|
|
* across restarts: every pair of repeats' "single" rows (label flips, max |dp|)
|
|
Paired candidate-vs-SemIf comparisons use each row's majority top over the repeats (ties
|
|
never arise with 3 repeats and a strict majority rule; a row without a majority keeps r1).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
import itertools
|
|
import json
|
|
import math
|
|
import random
|
|
import statistics as st
|
|
import sys
|
|
from collections import Counter, defaultdict
|
|
from pathlib import Path
|
|
|
|
OUT = Path(sys.argv[1])
|
|
JB = OUT.parent / "src/jevbench-v1.2.16/datasets/public"
|
|
if not JB.exists():
|
|
JB = Path(sys.argv[2]) if len(sys.argv) > 2 else JB
|
|
TIERS = {}
|
|
for t in ("easy", "original", "hard"):
|
|
for line in (JB / f"{t}.jsonl").read_text().splitlines():
|
|
if line.strip():
|
|
TIERS[json.loads(line)["id"]] = t
|
|
|
|
|
|
def med(xs):
|
|
xs = [x for x in xs if x is not None]
|
|
if not xs:
|
|
return None
|
|
return {"median": st.median(xs), "min": min(xs), "max": max(xs), "n": len(xs)}
|
|
|
|
|
|
def load_runs(label):
|
|
runs = {}
|
|
for d in sorted((OUT / label).glob("r*")):
|
|
r = {"dir": d}
|
|
for name in ("sets", "shape"):
|
|
p = d / f"{name}.json"
|
|
r[name] = json.loads(p.read_text()) if p.exists() else None
|
|
p = d / "jevbench/results.jsonl"
|
|
r["jb"] = [json.loads(l) for l in p.read_text().splitlines() if l.strip()] if p.exists() else None
|
|
r["vram"] = []
|
|
p = d / "vram.csv"
|
|
if p.exists():
|
|
for row in csv.reader(p.read_text().splitlines()):
|
|
try:
|
|
r["vram"].append((float(row[0]), float(row[1])))
|
|
except (ValueError, IndexError):
|
|
pass
|
|
r["events"] = {}
|
|
p = d / "events.txt"
|
|
if p.exists():
|
|
for line in p.read_text().splitlines():
|
|
ts, name = line.split(" ", 1)
|
|
r["events"][name] = float(ts)
|
|
p = d / "up_at.txt"
|
|
if p.exists():
|
|
r["events"]["up"] = float(p.read_text().strip())
|
|
runs[d.name] = r
|
|
return runs
|
|
|
|
|
|
def jevbench(runs):
|
|
per = defaultdict(list)
|
|
for r in runs.values():
|
|
if not r["jb"]:
|
|
continue
|
|
c, n = Counter(), Counter()
|
|
for x in r["jb"]:
|
|
t = TIERS[x["task_id"]]
|
|
n[t] += 1
|
|
c[t] += bool(x["correct"])
|
|
for t in ("easy", "original", "hard"):
|
|
per[t].append(c[t] / n[t] if n[t] else None)
|
|
per[t + "_n"].append(c[t])
|
|
per["all"].append(sum(c.values()) / sum(n.values()))
|
|
per["all_n"].append(sum(c.values()))
|
|
per["failed"].append(sum(not x["ok"] for x in r["jb"]))
|
|
lat = [x["latency_s"] for x in r["jb"] if x.get("latency_s") is not None]
|
|
per["p50_ms"].append(st.median(lat) * 1000 if lat else None)
|
|
return {k: med(v) for k, v in per.items()}
|
|
|
|
|
|
def rows(run, cond):
|
|
s = run["sets"]
|
|
return {x["id"]: x for x in s["conditions"][cond]["rows"]} if s and cond in s["conditions"] else {}
|
|
|
|
|
|
SETS = ("authored144", "perturbations108", "cicada-w1", "cicada-w2", "wyrd")
|
|
POOLED = ("authored144", "cicada-w1", "wyrd") # the brief's three replaced-baseline sets, 259 rows
|
|
|
|
|
|
def acc(rs, key=None):
|
|
lab = [x for x in rs if x["ok"] and x["gold"]]
|
|
if key:
|
|
lab = [x for x in lab if key(x)]
|
|
return (sum(x["top"] in x["gold"] for x in lab), len(lab))
|
|
|
|
|
|
def set_table(runs):
|
|
out = {}
|
|
for cond in ("single", "rotations", "multifield"):
|
|
for name in ("pooled",) + SETS + ("wyrd:place", "wyrd:place2", "wyrd:exit", "wyrd:exit2"):
|
|
base, _, dec = name.partition(":")
|
|
wanted = set(POOLED) if base == "pooled" else {base}
|
|
vals, blind, ctrl, fails = [], [], [], []
|
|
for r in runs.values():
|
|
rr = list(rows(r, cond).values())
|
|
if not rr:
|
|
continue
|
|
sub = [x for x in rr if x["set"] in wanted and (not dec or x["decision"] == dec)]
|
|
c, n = acc([x for x in sub if x["cond"] == "evidence"])
|
|
vals.append((c, n))
|
|
b = acc([x for x in sub if x["cond"] == "blind"])
|
|
blind.append(b)
|
|
k = acc([x for x in sub if x["cond"] == "evidence" and x["tag"] == "control"])
|
|
ctrl.append(k)
|
|
fails.append(sum(not x["ok"] for x in sub))
|
|
if vals and vals[0][1]:
|
|
out[f"{cond}/{name}"] = {
|
|
"correct": med([c for c, _ in vals]), "n": vals[0][1],
|
|
"acc": med([c / n for c, n in vals if n]),
|
|
"blind_correct": med([c for c, _ in blind]) if blind and blind[0][1] else None,
|
|
"controls": med([c for c, _ in ctrl]) if ctrl and ctrl[0][1] else None,
|
|
"controls_n": ctrl[0][1] if ctrl else None,
|
|
"failures": med(fails)}
|
|
# agreement signal (rotations): accuracy of unanimous vs split rows, authored144+perturbations108
|
|
for r in runs.values():
|
|
rr = [x for x in rows(r, "rotations").values() if x["set"] in ("authored144", "perturbations108") and x["ok"]]
|
|
if rr:
|
|
un = [x for x in rr if x.get("agreement") == 1.0]
|
|
sp = [x for x in rr if x.get("agreement") is not None and x["agreement"] < 1.0]
|
|
out.setdefault("agreement_signal", []).append({"unanimous": acc(un), "split": acc(sp)})
|
|
return out
|
|
|
|
|
|
def top_of(x):
|
|
return x["top"] if x.get("ok") else None
|
|
|
|
|
|
def pdiff(a, b):
|
|
pa = dict(zip(a["option_ids"], a["probs"]))
|
|
pb = dict(zip(b["option_ids"], b["probs"]))
|
|
return max(abs(pa[i] - pb[i]) for i in pa)
|
|
|
|
|
|
def floors(runs):
|
|
out = {}
|
|
# A-vs-A inside the process
|
|
aa = []
|
|
for r in runs.values():
|
|
s, rep = rows(r, "single"), rows(r, "repeat")
|
|
ids = [i for i in rep if i in s and s[i]["ok"] and rep[i]["ok"]]
|
|
if ids:
|
|
aa.append({"flips": sum(s[i]["top"] != rep[i]["top"] for i in ids),
|
|
"max_dp": max(pdiff(s[i], rep[i]) for i in ids), "n": len(ids)})
|
|
out["a_vs_a_in_process"] = aa
|
|
# across restarts, every row of "single" and of "rotations"
|
|
keys = sorted(runs)
|
|
for cond in ("single", "rotations"):
|
|
pairs = []
|
|
for a, b in itertools.combinations(keys, 2):
|
|
ra, rb = rows(runs[a], cond), rows(runs[b], cond)
|
|
ids = [i for i in ra if i in rb and ra[i]["ok"] and rb[i]["ok"]]
|
|
if ids:
|
|
flips = [i for i in ids if ra[i]["top"] != rb[i]["top"]]
|
|
pairs.append({"pair": f"{a}-{b}", "n": len(ids), "flips": len(flips),
|
|
"flipped_ids": flips[:10], "max_dp": max(pdiff(ra[i], rb[i]) for i in ids)})
|
|
out[f"cross_restart/{cond}"] = pairs
|
|
# rows whose top differs between ANY two repeats, per set (labelled evidence rows only)
|
|
tops = defaultdict(set)
|
|
meta = {}
|
|
for k in keys:
|
|
for i, x in rows(runs[k], cond).items():
|
|
if x["ok"]:
|
|
tops[i].add(x["top"])
|
|
meta[i] = x
|
|
unstable = [i for i, t in tops.items() if len(t) > 1]
|
|
out[f"unstable_rows/{cond}"] = {
|
|
s: sum(1 for i in unstable if meta[i]["set"] == s and meta[i]["cond"] == "evidence" and meta[i]["gold"])
|
|
for s in SETS} | {"pooled": sum(1 for i in unstable if meta[i]["set"] in POOLED
|
|
and meta[i]["cond"] == "evidence" and meta[i]["gold"]),
|
|
"repeats": len(keys)}
|
|
return out
|
|
|
|
|
|
def order_sensitivity(runs):
|
|
out = defaultdict(list)
|
|
for r in runs.values():
|
|
s = rows(r, "single")
|
|
for cond in ("reversed", "shuffled"):
|
|
o = rows(r, cond)
|
|
ids = [i for i in o if i in s and s[i]["ok"] and o[i]["ok"]]
|
|
if ids:
|
|
out[cond].append({"n": len(ids), "label_changes": sum(s[i]["top"] != o[i]["top"] for i in ids),
|
|
"max_dp": max(pdiff(s[i], o[i]) for i in ids),
|
|
"median_max_dp": st.median(pdiff(s[i], o[i]) for i in ids),
|
|
"acc": acc([o[i] for i in ids])})
|
|
return {k: {"label_changes": med([x["label_changes"] for x in v]), "n": v[0]["n"],
|
|
"max_dp": med([x["max_dp"] for x in v]), "median_row_max_dp": med([x["median_max_dp"] for x in v]),
|
|
"acc": med([x["acc"][0] for x in v])} for k, v in out.items()}
|
|
|
|
|
|
def negative(runs):
|
|
res = []
|
|
for r in runs.values():
|
|
s, neg = rows(r, "single"), rows(r, "negative")
|
|
ids = [i for i in neg if i in s and s[i]["ok"] and neg[i]["ok"]]
|
|
if ids:
|
|
res.append({"n": len(ids),
|
|
"same_top_as_unrotated": sum(neg[i]["top"] == s[i]["top"] for i in ids),
|
|
"vs_original_gold": sum(neg[i]["top"] in neg[i]["gold"] for i in ids),
|
|
"follows_description": sum(neg[i]["top"] == neg[i]["gold_desc_id"] for i in ids)})
|
|
return {k: med([x[k] for x in res]) for k in ("same_top_as_unrotated", "vs_original_gold", "follows_description")} | (
|
|
{"n": res[0]["n"]} if res else {})
|
|
|
|
|
|
def majority_tops(runs, cond):
|
|
votes = defaultdict(list)
|
|
meta = {}
|
|
for k in sorted(runs):
|
|
for i, x in rows(runs[k], cond).items():
|
|
votes[i].append(top_of(x))
|
|
meta[i] = x
|
|
out = {}
|
|
for i, v in votes.items():
|
|
c = Counter(t for t in v if t is not None).most_common()
|
|
out[i] = c[0][0] if c and c[0][1] > len(v) / 2 else v[0]
|
|
return out, meta
|
|
|
|
|
|
def mcnemar_exact(b, c):
|
|
n = b + c
|
|
if n == 0:
|
|
return 1.0
|
|
k = min(b, c)
|
|
p = sum(math.comb(n, i) for i in range(0, k + 1)) / 2 ** n
|
|
return min(1.0, 2 * p)
|
|
|
|
|
|
def paired(base_runs, cand_runs, cond, setname, cand_cond=None):
|
|
bt, meta = majority_tops(base_runs, cond)
|
|
ct, _ = majority_tops(cand_runs, cand_cond or cond)
|
|
wanted = set(POOLED) if setname == "pooled" else {setname.split(":")[0]}
|
|
ids = [i for i in bt if i in ct and meta[i]["set"] in wanted and meta[i]["cond"] == "evidence"
|
|
and meta[i]["gold"] and (":" not in setname or meta[i]["decision"] == setname.split(":")[1])]
|
|
if not ids:
|
|
return None
|
|
b_ok = {i: bt[i] in meta[i]["gold"] for i in ids}
|
|
c_ok = {i: ct[i] in meta[i]["gold"] for i in ids}
|
|
fixed = sum(c_ok[i] and not b_ok[i] for i in ids)
|
|
broken = sum(b_ok[i] and not c_ok[i] for i in ids)
|
|
groups = defaultdict(list)
|
|
for i in ids:
|
|
groups[meta[i]["group"]].append(i)
|
|
rng, keys, deltas = random.Random(7), list(groups), []
|
|
for _ in range(5000):
|
|
sample = [i for g in (rng.choice(keys) for _ in keys) for i in groups[g]]
|
|
deltas.append((sum(c_ok[i] for i in sample) - sum(b_ok[i] for i in sample)) / len(sample))
|
|
deltas.sort()
|
|
return {"n": len(ids), "semif": sum(b_ok.values()), "cand": sum(c_ok.values()), "fixed": fixed,
|
|
"broken": broken, "mcnemar_p": round(mcnemar_exact(fixed, broken), 4),
|
|
"delta_pts": round(100 * (sum(c_ok.values()) - sum(b_ok.values())) / len(ids), 1) if ids else None,
|
|
"delta_95ci_pts": [round(100 * deltas[125], 1), round(100 * deltas[4875], 1)] if ids else None}
|
|
|
|
|
|
def vram(runs):
|
|
res = []
|
|
for r in runs.values():
|
|
ev, v = r["events"], r["vram"]
|
|
if not v:
|
|
continue
|
|
rest = [m for t, m in v if "rest:start" in ev and ev["rest:start"] <= t <= ev.get("rest:end", 0)]
|
|
if not rest and "up" in ev: # r1 of the SemIf baseline predates the rest window
|
|
rest = [m for t, m in v if ev["up"] <= t <= ev["up"] + 1.5]
|
|
serve0 = ev.get("serve:start", 0)
|
|
sets0 = ev.get("sets:start", ev.get("up", serve0))
|
|
shape = r["shape"]
|
|
cap0 = None
|
|
if shape:
|
|
cap_ev = [t for name, t in shape["events"] if name.startswith("capacity:")]
|
|
cap0 = min(cap_ev) if cap_ev else None
|
|
work = [m for t, m in v if t >= sets0 and (cap0 is None or t < cap0) and t <= ev.get("serve:end", 1e18)]
|
|
capw = [m for t, m in v if cap0 is not None and t >= cap0 and t <= ev.get("serve:end", 1e18)]
|
|
jb = [m for t, m in v if ev.get("jevbench:start", 1e18) <= t <= ev.get("jevbench:end", 0)]
|
|
res.append({"rest_mib": st.median(rest) if rest else None, "peak_work_mib": max(work) if work else None,
|
|
"peak_capacity_mib": max(capw) if capw else None, "peak_jevbench_mib": max(jb) if jb else None})
|
|
return {k: med([x[k] for x in res]) for k in ("rest_mib", "peak_work_mib", "peak_capacity_mib", "peak_jevbench_mib")}
|
|
|
|
|
|
def latency(runs):
|
|
out = defaultdict(lambda: defaultdict(list))
|
|
cap = {}
|
|
for k, r in sorted(runs.items()):
|
|
sh = r["shape"]
|
|
if not sh:
|
|
continue
|
|
for name, s in sh["shapes"].items():
|
|
out[name]["e2e"].append(s["median_e2e_ms"])
|
|
out[name]["server"].append(s["median_server_ms"])
|
|
out[name]["run_medians"].extend(s["run_medians_e2e_ms"])
|
|
out[name]["tokens"].append(s["tokens"])
|
|
out[name]["non_200"].append(s["non_200"])
|
|
if sh.get("capacity"):
|
|
cap[k] = {lab: [(x["rows"], x["status"]) for x in series] for lab, series in sh["capacity"].items()}
|
|
return {name: {"e2e_ms": med(v["e2e"]), "server_ms": med(v["server"]),
|
|
"run_medians_ms": [min(v["run_medians"]), max(v["run_medians"])],
|
|
"tokens": v["tokens"][0], "non_200": sum(v["non_200"])} for name, v in out.items()}, cap
|
|
|
|
|
|
def parity(runs):
|
|
"""semif-format runs only: authored144 'single' against SemIf's committed torch predictions."""
|
|
p = Path(__file__).resolve().parent / "data/direct-authored144.jsonl"
|
|
ref = {x["id"]: x for x in map(json.loads, p.read_text().splitlines()) if x}
|
|
res = []
|
|
for r in runs.values():
|
|
s = rows(r, "single")
|
|
ids = [i for i in ref if i in s and s[i].get("prompt_sha256")]
|
|
if not ids:
|
|
continue
|
|
top_ref = {i: ref[i]["option_ids"][max(range(len(ref[i]["probabilities"])), key=ref[i]["probabilities"].__getitem__)]
|
|
for i in ids}
|
|
res.append({"n": len(ids), "top_agree": sum(s[i]["top"] == top_ref[i] for i in ids),
|
|
"prompt_sha_equal": sum(s[i]["prompt_sha256"] == ref[i]["prompt_sha256"] for i in ids),
|
|
"max_dp": max(max(abs(a - b) for a, b in zip(s[i]["probs"], ref[i]["probabilities"])) for i in ids)})
|
|
return res
|
|
|
|
|
|
def main():
|
|
labels = [p.name for p in sorted(OUT.iterdir()) if p.is_dir() and any(p.glob("r*/"))]
|
|
allruns = {l: load_runs(l) for l in labels}
|
|
summary = {}
|
|
base = allruns.get("semif-qwen35-4b")
|
|
for l, runs in allruns.items():
|
|
lat, cap = latency(runs)
|
|
s = {"repeats": sorted(runs), "jevbench": jevbench(runs), "sets": set_table(runs), "floors": floors(runs),
|
|
"order": order_sensitivity(runs), "negative": negative(runs), "vram": vram(runs),
|
|
"latency": lat, "capacity": cap, "parity_vs_upstream_semif": parity(runs)}
|
|
if base and l != "semif-qwen35-4b":
|
|
s["paired_vs_semif"] = {f"{cond}/{name}": paired(base, runs, cond, name)
|
|
for cond in ("single", "rotations")
|
|
for name in ("pooled",) + SETS + ("wyrd:place2", "wyrd:exit")}
|
|
# the cost question: the candidate at ONE ordering against SemIf WITH rotations
|
|
s["paired_vs_semif"].update({f"cand-single-vs-semif-rotations/{name}": paired(base, runs, "rotations", name, "single")
|
|
for name in ("pooled",) + SETS})
|
|
summary[l] = s
|
|
json.dump(summary, sys.stdout, indent=1, default=str)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|