docs: Jev candidate bench vs SemIf (fv-ml1 GPU 3) — Intern-Decision-4B is the replacement if SemIf is displaced
Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231, hard 0.613) reproduced exactly; negative control and a 4-restart noise floor (0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with- rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench rank does not transfer. Raw per-item data kept out of git.
This commit is contained in:
@@ -0,0 +1,360 @@
|
||||
"""Jev-candidate bench (2026-09-30): turn the raw runs into the tables of the results doc.
|
||||
|
||||
python3 analyze.py <out dir> > summary.json (and prints markdown tables to stderr)
|
||||
|
||||
Every cell is a median over the repeats (one repeat = one process lifetime) with min..max.
|
||||
The noise floors come from the same data:
|
||||
* A-vs-A inside a process: authored144 "single" vs "repeat"
|
||||
* across restarts: every pair of repeats' "single" rows (label flips, max |dp|)
|
||||
Paired candidate-vs-SemIf comparisons use each row's majority top over the repeats (ties
|
||||
never arise with 3 repeats and a strict majority rule; a row without a majority keeps r1).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import csv
|
||||
import itertools
|
||||
import json
|
||||
import math
|
||||
import random
|
||||
import statistics as st
|
||||
import sys
|
||||
from collections import Counter, defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
OUT = Path(sys.argv[1])
|
||||
JB = OUT.parent / "src/jevbench-v1.2.16/datasets/public"
|
||||
if not JB.exists():
|
||||
JB = Path(sys.argv[2]) if len(sys.argv) > 2 else JB
|
||||
TIERS = {}
|
||||
for t in ("easy", "original", "hard"):
|
||||
for line in (JB / f"{t}.jsonl").read_text().splitlines():
|
||||
if line.strip():
|
||||
TIERS[json.loads(line)["id"]] = t
|
||||
|
||||
|
||||
def med(xs):
|
||||
xs = [x for x in xs if x is not None]
|
||||
if not xs:
|
||||
return None
|
||||
return {"median": st.median(xs), "min": min(xs), "max": max(xs), "n": len(xs)}
|
||||
|
||||
|
||||
def load_runs(label):
|
||||
runs = {}
|
||||
for d in sorted((OUT / label).glob("r*")):
|
||||
r = {"dir": d}
|
||||
for name in ("sets", "shape"):
|
||||
p = d / f"{name}.json"
|
||||
r[name] = json.loads(p.read_text()) if p.exists() else None
|
||||
p = d / "jevbench/results.jsonl"
|
||||
r["jb"] = [json.loads(l) for l in p.read_text().splitlines() if l.strip()] if p.exists() else None
|
||||
r["vram"] = []
|
||||
p = d / "vram.csv"
|
||||
if p.exists():
|
||||
for row in csv.reader(p.read_text().splitlines()):
|
||||
try:
|
||||
r["vram"].append((float(row[0]), float(row[1])))
|
||||
except (ValueError, IndexError):
|
||||
pass
|
||||
r["events"] = {}
|
||||
p = d / "events.txt"
|
||||
if p.exists():
|
||||
for line in p.read_text().splitlines():
|
||||
ts, name = line.split(" ", 1)
|
||||
r["events"][name] = float(ts)
|
||||
p = d / "up_at.txt"
|
||||
if p.exists():
|
||||
r["events"]["up"] = float(p.read_text().strip())
|
||||
runs[d.name] = r
|
||||
return runs
|
||||
|
||||
|
||||
def jevbench(runs):
|
||||
per = defaultdict(list)
|
||||
for r in runs.values():
|
||||
if not r["jb"]:
|
||||
continue
|
||||
c, n = Counter(), Counter()
|
||||
for x in r["jb"]:
|
||||
t = TIERS[x["task_id"]]
|
||||
n[t] += 1
|
||||
c[t] += bool(x["correct"])
|
||||
for t in ("easy", "original", "hard"):
|
||||
per[t].append(c[t] / n[t] if n[t] else None)
|
||||
per[t + "_n"].append(c[t])
|
||||
per["all"].append(sum(c.values()) / sum(n.values()))
|
||||
per["all_n"].append(sum(c.values()))
|
||||
per["failed"].append(sum(not x["ok"] for x in r["jb"]))
|
||||
lat = [x["latency_s"] for x in r["jb"] if x.get("latency_s") is not None]
|
||||
per["p50_ms"].append(st.median(lat) * 1000 if lat else None)
|
||||
return {k: med(v) for k, v in per.items()}
|
||||
|
||||
|
||||
def rows(run, cond):
|
||||
s = run["sets"]
|
||||
return {x["id"]: x for x in s["conditions"][cond]["rows"]} if s and cond in s["conditions"] else {}
|
||||
|
||||
|
||||
SETS = ("authored144", "perturbations108", "cicada-w1", "cicada-w2", "wyrd")
|
||||
POOLED = ("authored144", "cicada-w1", "wyrd") # the brief's three replaced-baseline sets, 259 rows
|
||||
|
||||
|
||||
def acc(rs, key=None):
|
||||
lab = [x for x in rs if x["ok"] and x["gold"]]
|
||||
if key:
|
||||
lab = [x for x in lab if key(x)]
|
||||
return (sum(x["top"] in x["gold"] for x in lab), len(lab))
|
||||
|
||||
|
||||
def set_table(runs):
|
||||
out = {}
|
||||
for cond in ("single", "rotations", "multifield"):
|
||||
for name in ("pooled",) + SETS + ("wyrd:place", "wyrd:place2", "wyrd:exit", "wyrd:exit2"):
|
||||
base, _, dec = name.partition(":")
|
||||
wanted = set(POOLED) if base == "pooled" else {base}
|
||||
vals, blind, ctrl, fails = [], [], [], []
|
||||
for r in runs.values():
|
||||
rr = list(rows(r, cond).values())
|
||||
if not rr:
|
||||
continue
|
||||
sub = [x for x in rr if x["set"] in wanted and (not dec or x["decision"] == dec)]
|
||||
c, n = acc([x for x in sub if x["cond"] == "evidence"])
|
||||
vals.append((c, n))
|
||||
b = acc([x for x in sub if x["cond"] == "blind"])
|
||||
blind.append(b)
|
||||
k = acc([x for x in sub if x["cond"] == "evidence" and x["tag"] == "control"])
|
||||
ctrl.append(k)
|
||||
fails.append(sum(not x["ok"] for x in sub))
|
||||
if vals and vals[0][1]:
|
||||
out[f"{cond}/{name}"] = {
|
||||
"correct": med([c for c, _ in vals]), "n": vals[0][1],
|
||||
"acc": med([c / n for c, n in vals if n]),
|
||||
"blind_correct": med([c for c, _ in blind]) if blind and blind[0][1] else None,
|
||||
"controls": med([c for c, _ in ctrl]) if ctrl and ctrl[0][1] else None,
|
||||
"controls_n": ctrl[0][1] if ctrl else None,
|
||||
"failures": med(fails)}
|
||||
# agreement signal (rotations): accuracy of unanimous vs split rows, authored144+perturbations108
|
||||
for r in runs.values():
|
||||
rr = [x for x in rows(r, "rotations").values() if x["set"] in ("authored144", "perturbations108") and x["ok"]]
|
||||
if rr:
|
||||
un = [x for x in rr if x.get("agreement") == 1.0]
|
||||
sp = [x for x in rr if x.get("agreement") is not None and x["agreement"] < 1.0]
|
||||
out.setdefault("agreement_signal", []).append({"unanimous": acc(un), "split": acc(sp)})
|
||||
return out
|
||||
|
||||
|
||||
def top_of(x):
|
||||
return x["top"] if x.get("ok") else None
|
||||
|
||||
|
||||
def pdiff(a, b):
|
||||
pa = dict(zip(a["option_ids"], a["probs"]))
|
||||
pb = dict(zip(b["option_ids"], b["probs"]))
|
||||
return max(abs(pa[i] - pb[i]) for i in pa)
|
||||
|
||||
|
||||
def floors(runs):
|
||||
out = {}
|
||||
# A-vs-A inside the process
|
||||
aa = []
|
||||
for r in runs.values():
|
||||
s, rep = rows(r, "single"), rows(r, "repeat")
|
||||
ids = [i for i in rep if i in s and s[i]["ok"] and rep[i]["ok"]]
|
||||
if ids:
|
||||
aa.append({"flips": sum(s[i]["top"] != rep[i]["top"] for i in ids),
|
||||
"max_dp": max(pdiff(s[i], rep[i]) for i in ids), "n": len(ids)})
|
||||
out["a_vs_a_in_process"] = aa
|
||||
# across restarts, every row of "single" and of "rotations"
|
||||
keys = sorted(runs)
|
||||
for cond in ("single", "rotations"):
|
||||
pairs = []
|
||||
for a, b in itertools.combinations(keys, 2):
|
||||
ra, rb = rows(runs[a], cond), rows(runs[b], cond)
|
||||
ids = [i for i in ra if i in rb and ra[i]["ok"] and rb[i]["ok"]]
|
||||
if ids:
|
||||
flips = [i for i in ids if ra[i]["top"] != rb[i]["top"]]
|
||||
pairs.append({"pair": f"{a}-{b}", "n": len(ids), "flips": len(flips),
|
||||
"flipped_ids": flips[:10], "max_dp": max(pdiff(ra[i], rb[i]) for i in ids)})
|
||||
out[f"cross_restart/{cond}"] = pairs
|
||||
# rows whose top differs between ANY two repeats, per set (labelled evidence rows only)
|
||||
tops = defaultdict(set)
|
||||
meta = {}
|
||||
for k in keys:
|
||||
for i, x in rows(runs[k], cond).items():
|
||||
if x["ok"]:
|
||||
tops[i].add(x["top"])
|
||||
meta[i] = x
|
||||
unstable = [i for i, t in tops.items() if len(t) > 1]
|
||||
out[f"unstable_rows/{cond}"] = {
|
||||
s: sum(1 for i in unstable if meta[i]["set"] == s and meta[i]["cond"] == "evidence" and meta[i]["gold"])
|
||||
for s in SETS} | {"pooled": sum(1 for i in unstable if meta[i]["set"] in POOLED
|
||||
and meta[i]["cond"] == "evidence" and meta[i]["gold"]),
|
||||
"repeats": len(keys)}
|
||||
return out
|
||||
|
||||
|
||||
def order_sensitivity(runs):
|
||||
out = defaultdict(list)
|
||||
for r in runs.values():
|
||||
s = rows(r, "single")
|
||||
for cond in ("reversed", "shuffled"):
|
||||
o = rows(r, cond)
|
||||
ids = [i for i in o if i in s and s[i]["ok"] and o[i]["ok"]]
|
||||
if ids:
|
||||
out[cond].append({"n": len(ids), "label_changes": sum(s[i]["top"] != o[i]["top"] for i in ids),
|
||||
"max_dp": max(pdiff(s[i], o[i]) for i in ids),
|
||||
"median_max_dp": st.median(pdiff(s[i], o[i]) for i in ids),
|
||||
"acc": acc([o[i] for i in ids])})
|
||||
return {k: {"label_changes": med([x["label_changes"] for x in v]), "n": v[0]["n"],
|
||||
"max_dp": med([x["max_dp"] for x in v]), "median_row_max_dp": med([x["median_max_dp"] for x in v]),
|
||||
"acc": med([x["acc"][0] for x in v])} for k, v in out.items()}
|
||||
|
||||
|
||||
def negative(runs):
|
||||
res = []
|
||||
for r in runs.values():
|
||||
s, neg = rows(r, "single"), rows(r, "negative")
|
||||
ids = [i for i in neg if i in s and s[i]["ok"] and neg[i]["ok"]]
|
||||
if ids:
|
||||
res.append({"n": len(ids),
|
||||
"same_top_as_unrotated": sum(neg[i]["top"] == s[i]["top"] for i in ids),
|
||||
"vs_original_gold": sum(neg[i]["top"] in neg[i]["gold"] for i in ids),
|
||||
"follows_description": sum(neg[i]["top"] == neg[i]["gold_desc_id"] for i in ids)})
|
||||
return {k: med([x[k] for x in res]) for k in ("same_top_as_unrotated", "vs_original_gold", "follows_description")} | (
|
||||
{"n": res[0]["n"]} if res else {})
|
||||
|
||||
|
||||
def majority_tops(runs, cond):
|
||||
votes = defaultdict(list)
|
||||
meta = {}
|
||||
for k in sorted(runs):
|
||||
for i, x in rows(runs[k], cond).items():
|
||||
votes[i].append(top_of(x))
|
||||
meta[i] = x
|
||||
out = {}
|
||||
for i, v in votes.items():
|
||||
c = Counter(t for t in v if t is not None).most_common()
|
||||
out[i] = c[0][0] if c and c[0][1] > len(v) / 2 else v[0]
|
||||
return out, meta
|
||||
|
||||
|
||||
def mcnemar_exact(b, c):
|
||||
n = b + c
|
||||
if n == 0:
|
||||
return 1.0
|
||||
k = min(b, c)
|
||||
p = sum(math.comb(n, i) for i in range(0, k + 1)) / 2 ** n
|
||||
return min(1.0, 2 * p)
|
||||
|
||||
|
||||
def paired(base_runs, cand_runs, cond, setname, cand_cond=None):
|
||||
bt, meta = majority_tops(base_runs, cond)
|
||||
ct, _ = majority_tops(cand_runs, cand_cond or cond)
|
||||
wanted = set(POOLED) if setname == "pooled" else {setname.split(":")[0]}
|
||||
ids = [i for i in bt if i in ct and meta[i]["set"] in wanted and meta[i]["cond"] == "evidence"
|
||||
and meta[i]["gold"] and (":" not in setname or meta[i]["decision"] == setname.split(":")[1])]
|
||||
if not ids:
|
||||
return None
|
||||
b_ok = {i: bt[i] in meta[i]["gold"] for i in ids}
|
||||
c_ok = {i: ct[i] in meta[i]["gold"] for i in ids}
|
||||
fixed = sum(c_ok[i] and not b_ok[i] for i in ids)
|
||||
broken = sum(b_ok[i] and not c_ok[i] for i in ids)
|
||||
groups = defaultdict(list)
|
||||
for i in ids:
|
||||
groups[meta[i]["group"]].append(i)
|
||||
rng, keys, deltas = random.Random(7), list(groups), []
|
||||
for _ in range(5000):
|
||||
sample = [i for g in (rng.choice(keys) for _ in keys) for i in groups[g]]
|
||||
deltas.append((sum(c_ok[i] for i in sample) - sum(b_ok[i] for i in sample)) / len(sample))
|
||||
deltas.sort()
|
||||
return {"n": len(ids), "semif": sum(b_ok.values()), "cand": sum(c_ok.values()), "fixed": fixed,
|
||||
"broken": broken, "mcnemar_p": round(mcnemar_exact(fixed, broken), 4),
|
||||
"delta_pts": round(100 * (sum(c_ok.values()) - sum(b_ok.values())) / len(ids), 1) if ids else None,
|
||||
"delta_95ci_pts": [round(100 * deltas[125], 1), round(100 * deltas[4875], 1)] if ids else None}
|
||||
|
||||
|
||||
def vram(runs):
|
||||
res = []
|
||||
for r in runs.values():
|
||||
ev, v = r["events"], r["vram"]
|
||||
if not v:
|
||||
continue
|
||||
rest = [m for t, m in v if "rest:start" in ev and ev["rest:start"] <= t <= ev.get("rest:end", 0)]
|
||||
if not rest and "up" in ev: # r1 of the SemIf baseline predates the rest window
|
||||
rest = [m for t, m in v if ev["up"] <= t <= ev["up"] + 1.5]
|
||||
serve0 = ev.get("serve:start", 0)
|
||||
sets0 = ev.get("sets:start", ev.get("up", serve0))
|
||||
shape = r["shape"]
|
||||
cap0 = None
|
||||
if shape:
|
||||
cap_ev = [t for name, t in shape["events"] if name.startswith("capacity:")]
|
||||
cap0 = min(cap_ev) if cap_ev else None
|
||||
work = [m for t, m in v if t >= sets0 and (cap0 is None or t < cap0) and t <= ev.get("serve:end", 1e18)]
|
||||
capw = [m for t, m in v if cap0 is not None and t >= cap0 and t <= ev.get("serve:end", 1e18)]
|
||||
jb = [m for t, m in v if ev.get("jevbench:start", 1e18) <= t <= ev.get("jevbench:end", 0)]
|
||||
res.append({"rest_mib": st.median(rest) if rest else None, "peak_work_mib": max(work) if work else None,
|
||||
"peak_capacity_mib": max(capw) if capw else None, "peak_jevbench_mib": max(jb) if jb else None})
|
||||
return {k: med([x[k] for x in res]) for k in ("rest_mib", "peak_work_mib", "peak_capacity_mib", "peak_jevbench_mib")}
|
||||
|
||||
|
||||
def latency(runs):
|
||||
out = defaultdict(lambda: defaultdict(list))
|
||||
cap = {}
|
||||
for k, r in sorted(runs.items()):
|
||||
sh = r["shape"]
|
||||
if not sh:
|
||||
continue
|
||||
for name, s in sh["shapes"].items():
|
||||
out[name]["e2e"].append(s["median_e2e_ms"])
|
||||
out[name]["server"].append(s["median_server_ms"])
|
||||
out[name]["run_medians"].extend(s["run_medians_e2e_ms"])
|
||||
out[name]["tokens"].append(s["tokens"])
|
||||
out[name]["non_200"].append(s["non_200"])
|
||||
if sh.get("capacity"):
|
||||
cap[k] = {lab: [(x["rows"], x["status"]) for x in series] for lab, series in sh["capacity"].items()}
|
||||
return {name: {"e2e_ms": med(v["e2e"]), "server_ms": med(v["server"]),
|
||||
"run_medians_ms": [min(v["run_medians"]), max(v["run_medians"])],
|
||||
"tokens": v["tokens"][0], "non_200": sum(v["non_200"])} for name, v in out.items()}, cap
|
||||
|
||||
|
||||
def parity(runs):
|
||||
"""semif-format runs only: authored144 'single' against SemIf's committed torch predictions."""
|
||||
p = Path(__file__).resolve().parent / "data/direct-authored144.jsonl"
|
||||
ref = {x["id"]: x for x in map(json.loads, p.read_text().splitlines()) if x}
|
||||
res = []
|
||||
for r in runs.values():
|
||||
s = rows(r, "single")
|
||||
ids = [i for i in ref if i in s and s[i].get("prompt_sha256")]
|
||||
if not ids:
|
||||
continue
|
||||
top_ref = {i: ref[i]["option_ids"][max(range(len(ref[i]["probabilities"])), key=ref[i]["probabilities"].__getitem__)]
|
||||
for i in ids}
|
||||
res.append({"n": len(ids), "top_agree": sum(s[i]["top"] == top_ref[i] for i in ids),
|
||||
"prompt_sha_equal": sum(s[i]["prompt_sha256"] == ref[i]["prompt_sha256"] for i in ids),
|
||||
"max_dp": max(max(abs(a - b) for a, b in zip(s[i]["probs"], ref[i]["probabilities"])) for i in ids)})
|
||||
return res
|
||||
|
||||
|
||||
def main():
|
||||
labels = [p.name for p in sorted(OUT.iterdir()) if p.is_dir() and any(p.glob("r*/"))]
|
||||
allruns = {l: load_runs(l) for l in labels}
|
||||
summary = {}
|
||||
base = allruns.get("semif-qwen35-4b")
|
||||
for l, runs in allruns.items():
|
||||
lat, cap = latency(runs)
|
||||
s = {"repeats": sorted(runs), "jevbench": jevbench(runs), "sets": set_table(runs), "floors": floors(runs),
|
||||
"order": order_sensitivity(runs), "negative": negative(runs), "vram": vram(runs),
|
||||
"latency": lat, "capacity": cap, "parity_vs_upstream_semif": parity(runs)}
|
||||
if base and l != "semif-qwen35-4b":
|
||||
s["paired_vs_semif"] = {f"{cond}/{name}": paired(base, runs, cond, name)
|
||||
for cond in ("single", "rotations")
|
||||
for name in ("pooled",) + SETS + ("wyrd:place2", "wyrd:exit")}
|
||||
# the cost question: the candidate at ONE ordering against SemIf WITH rotations
|
||||
s["paired_vs_semif"].update({f"cand-single-vs-semif-rotations/{name}": paired(base, runs, "rotations", name, "single")
|
||||
for name in ("pooled",) + SETS})
|
||||
summary[l] = s
|
||||
json.dump(summary, sys.stdout, indent=1, default=str)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user