"""Jev-candidate bench (2026-09-30): turn the raw runs into the tables of the results doc. python3 analyze.py > summary.json (and prints markdown tables to stderr) Every cell is a median over the repeats (one repeat = one process lifetime) with min..max. The noise floors come from the same data: * A-vs-A inside a process: authored144 "single" vs "repeat" * across restarts: every pair of repeats' "single" rows (label flips, max |dp|) Paired candidate-vs-SemIf comparisons use each row's majority top over the repeats (ties never arise with 3 repeats and a strict majority rule; a row without a majority keeps r1). """ from __future__ import annotations import csv import itertools import json import math import random import statistics as st import sys from collections import Counter, defaultdict from pathlib import Path OUT = Path(sys.argv[1]) JB = OUT.parent / "src/jevbench-v1.2.16/datasets/public" if not JB.exists(): JB = Path(sys.argv[2]) if len(sys.argv) > 2 else JB TIERS = {} for t in ("easy", "original", "hard"): for line in (JB / f"{t}.jsonl").read_text().splitlines(): if line.strip(): TIERS[json.loads(line)["id"]] = t def med(xs): xs = [x for x in xs if x is not None] if not xs: return None return {"median": st.median(xs), "min": min(xs), "max": max(xs), "n": len(xs)} def load_runs(label): runs = {} for d in sorted((OUT / label).glob("r*")): r = {"dir": d} for name in ("sets", "shape"): p = d / f"{name}.json" r[name] = json.loads(p.read_text()) if p.exists() else None p = d / "jevbench/results.jsonl" r["jb"] = [json.loads(l) for l in p.read_text().splitlines() if l.strip()] if p.exists() else None r["vram"] = [] p = d / "vram.csv" if p.exists(): for row in csv.reader(p.read_text().splitlines()): try: r["vram"].append((float(row[0]), float(row[1]))) except (ValueError, IndexError): pass r["events"] = {} p = d / "events.txt" if p.exists(): for line in p.read_text().splitlines(): ts, name = line.split(" ", 1) r["events"][name] = float(ts) p = d / "up_at.txt" if p.exists(): r["events"]["up"] = float(p.read_text().strip()) runs[d.name] = r return runs def jevbench(runs): per = defaultdict(list) for r in runs.values(): if not r["jb"]: continue c, n = Counter(), Counter() for x in r["jb"]: t = TIERS[x["task_id"]] n[t] += 1 c[t] += bool(x["correct"]) for t in ("easy", "original", "hard"): per[t].append(c[t] / n[t] if n[t] else None) per[t + "_n"].append(c[t]) per["all"].append(sum(c.values()) / sum(n.values())) per["all_n"].append(sum(c.values())) per["failed"].append(sum(not x["ok"] for x in r["jb"])) lat = [x["latency_s"] for x in r["jb"] if x.get("latency_s") is not None] per["p50_ms"].append(st.median(lat) * 1000 if lat else None) return {k: med(v) for k, v in per.items()} def rows(run, cond): s = run["sets"] return {x["id"]: x for x in s["conditions"][cond]["rows"]} if s and cond in s["conditions"] else {} SETS = ("authored144", "perturbations108", "cicada-w1", "cicada-w2", "wyrd") POOLED = ("authored144", "cicada-w1", "wyrd") # the brief's three replaced-baseline sets, 259 rows def acc(rs, key=None): lab = [x for x in rs if x["ok"] and x["gold"]] if key: lab = [x for x in lab if key(x)] return (sum(x["top"] in x["gold"] for x in lab), len(lab)) def set_table(runs): out = {} for cond in ("single", "rotations", "multifield"): for name in ("pooled",) + SETS + ("wyrd:place", "wyrd:place2", "wyrd:exit", "wyrd:exit2"): base, _, dec = name.partition(":") wanted = set(POOLED) if base == "pooled" else {base} vals, blind, ctrl, fails = [], [], [], [] for r in runs.values(): rr = list(rows(r, cond).values()) if not rr: continue sub = [x for x in rr if x["set"] in wanted and (not dec or x["decision"] == dec)] c, n = acc([x for x in sub if x["cond"] == "evidence"]) vals.append((c, n)) b = acc([x for x in sub if x["cond"] == "blind"]) blind.append(b) k = acc([x for x in sub if x["cond"] == "evidence" and x["tag"] == "control"]) ctrl.append(k) fails.append(sum(not x["ok"] for x in sub)) if vals and vals[0][1]: out[f"{cond}/{name}"] = { "correct": med([c for c, _ in vals]), "n": vals[0][1], "acc": med([c / n for c, n in vals if n]), "blind_correct": med([c for c, _ in blind]) if blind and blind[0][1] else None, "controls": med([c for c, _ in ctrl]) if ctrl and ctrl[0][1] else None, "controls_n": ctrl[0][1] if ctrl else None, "failures": med(fails)} # agreement signal (rotations): accuracy of unanimous vs split rows, authored144+perturbations108 for r in runs.values(): rr = [x for x in rows(r, "rotations").values() if x["set"] in ("authored144", "perturbations108") and x["ok"]] if rr: un = [x for x in rr if x.get("agreement") == 1.0] sp = [x for x in rr if x.get("agreement") is not None and x["agreement"] < 1.0] out.setdefault("agreement_signal", []).append({"unanimous": acc(un), "split": acc(sp)}) return out def top_of(x): return x["top"] if x.get("ok") else None def pdiff(a, b): pa = dict(zip(a["option_ids"], a["probs"])) pb = dict(zip(b["option_ids"], b["probs"])) return max(abs(pa[i] - pb[i]) for i in pa) def floors(runs): out = {} # A-vs-A inside the process aa = [] for r in runs.values(): s, rep = rows(r, "single"), rows(r, "repeat") ids = [i for i in rep if i in s and s[i]["ok"] and rep[i]["ok"]] if ids: aa.append({"flips": sum(s[i]["top"] != rep[i]["top"] for i in ids), "max_dp": max(pdiff(s[i], rep[i]) for i in ids), "n": len(ids)}) out["a_vs_a_in_process"] = aa # across restarts, every row of "single" and of "rotations" keys = sorted(runs) for cond in ("single", "rotations"): pairs = [] for a, b in itertools.combinations(keys, 2): ra, rb = rows(runs[a], cond), rows(runs[b], cond) ids = [i for i in ra if i in rb and ra[i]["ok"] and rb[i]["ok"]] if ids: flips = [i for i in ids if ra[i]["top"] != rb[i]["top"]] pairs.append({"pair": f"{a}-{b}", "n": len(ids), "flips": len(flips), "flipped_ids": flips[:10], "max_dp": max(pdiff(ra[i], rb[i]) for i in ids)}) out[f"cross_restart/{cond}"] = pairs # rows whose top differs between ANY two repeats, per set (labelled evidence rows only) tops = defaultdict(set) meta = {} for k in keys: for i, x in rows(runs[k], cond).items(): if x["ok"]: tops[i].add(x["top"]) meta[i] = x unstable = [i for i, t in tops.items() if len(t) > 1] out[f"unstable_rows/{cond}"] = { s: sum(1 for i in unstable if meta[i]["set"] == s and meta[i]["cond"] == "evidence" and meta[i]["gold"]) for s in SETS} | {"pooled": sum(1 for i in unstable if meta[i]["set"] in POOLED and meta[i]["cond"] == "evidence" and meta[i]["gold"]), "repeats": len(keys)} return out def order_sensitivity(runs): out = defaultdict(list) for r in runs.values(): s = rows(r, "single") for cond in ("reversed", "shuffled"): o = rows(r, cond) ids = [i for i in o if i in s and s[i]["ok"] and o[i]["ok"]] if ids: out[cond].append({"n": len(ids), "label_changes": sum(s[i]["top"] != o[i]["top"] for i in ids), "max_dp": max(pdiff(s[i], o[i]) for i in ids), "median_max_dp": st.median(pdiff(s[i], o[i]) for i in ids), "acc": acc([o[i] for i in ids])}) return {k: {"label_changes": med([x["label_changes"] for x in v]), "n": v[0]["n"], "max_dp": med([x["max_dp"] for x in v]), "median_row_max_dp": med([x["median_max_dp"] for x in v]), "acc": med([x["acc"][0] for x in v])} for k, v in out.items()} def negative(runs): res = [] for r in runs.values(): s, neg = rows(r, "single"), rows(r, "negative") ids = [i for i in neg if i in s and s[i]["ok"] and neg[i]["ok"]] if ids: res.append({"n": len(ids), "same_top_as_unrotated": sum(neg[i]["top"] == s[i]["top"] for i in ids), "vs_original_gold": sum(neg[i]["top"] in neg[i]["gold"] for i in ids), "follows_description": sum(neg[i]["top"] == neg[i]["gold_desc_id"] for i in ids)}) return {k: med([x[k] for x in res]) for k in ("same_top_as_unrotated", "vs_original_gold", "follows_description")} | ( {"n": res[0]["n"]} if res else {}) def majority_tops(runs, cond): votes = defaultdict(list) meta = {} for k in sorted(runs): for i, x in rows(runs[k], cond).items(): votes[i].append(top_of(x)) meta[i] = x out = {} for i, v in votes.items(): c = Counter(t for t in v if t is not None).most_common() out[i] = c[0][0] if c and c[0][1] > len(v) / 2 else v[0] return out, meta def mcnemar_exact(b, c): n = b + c if n == 0: return 1.0 k = min(b, c) p = sum(math.comb(n, i) for i in range(0, k + 1)) / 2 ** n return min(1.0, 2 * p) def paired(base_runs, cand_runs, cond, setname, cand_cond=None): bt, meta = majority_tops(base_runs, cond) ct, _ = majority_tops(cand_runs, cand_cond or cond) wanted = set(POOLED) if setname == "pooled" else {setname.split(":")[0]} ids = [i for i in bt if i in ct and meta[i]["set"] in wanted and meta[i]["cond"] == "evidence" and meta[i]["gold"] and (":" not in setname or meta[i]["decision"] == setname.split(":")[1])] if not ids: return None b_ok = {i: bt[i] in meta[i]["gold"] for i in ids} c_ok = {i: ct[i] in meta[i]["gold"] for i in ids} fixed = sum(c_ok[i] and not b_ok[i] for i in ids) broken = sum(b_ok[i] and not c_ok[i] for i in ids) groups = defaultdict(list) for i in ids: groups[meta[i]["group"]].append(i) rng, keys, deltas = random.Random(7), list(groups), [] for _ in range(5000): sample = [i for g in (rng.choice(keys) for _ in keys) for i in groups[g]] deltas.append((sum(c_ok[i] for i in sample) - sum(b_ok[i] for i in sample)) / len(sample)) deltas.sort() return {"n": len(ids), "semif": sum(b_ok.values()), "cand": sum(c_ok.values()), "fixed": fixed, "broken": broken, "mcnemar_p": round(mcnemar_exact(fixed, broken), 4), "delta_pts": round(100 * (sum(c_ok.values()) - sum(b_ok.values())) / len(ids), 1) if ids else None, "delta_95ci_pts": [round(100 * deltas[125], 1), round(100 * deltas[4875], 1)] if ids else None} def vram(runs): res = [] for r in runs.values(): ev, v = r["events"], r["vram"] if not v: continue rest = [m for t, m in v if "rest:start" in ev and ev["rest:start"] <= t <= ev.get("rest:end", 0)] if not rest and "up" in ev: # r1 of the SemIf baseline predates the rest window rest = [m for t, m in v if ev["up"] <= t <= ev["up"] + 1.5] serve0 = ev.get("serve:start", 0) sets0 = ev.get("sets:start", ev.get("up", serve0)) shape = r["shape"] cap0 = None if shape: cap_ev = [t for name, t in shape["events"] if name.startswith("capacity:")] cap0 = min(cap_ev) if cap_ev else None work = [m for t, m in v if t >= sets0 and (cap0 is None or t < cap0) and t <= ev.get("serve:end", 1e18)] capw = [m for t, m in v if cap0 is not None and t >= cap0 and t <= ev.get("serve:end", 1e18)] jb = [m for t, m in v if ev.get("jevbench:start", 1e18) <= t <= ev.get("jevbench:end", 0)] res.append({"rest_mib": st.median(rest) if rest else None, "peak_work_mib": max(work) if work else None, "peak_capacity_mib": max(capw) if capw else None, "peak_jevbench_mib": max(jb) if jb else None}) return {k: med([x[k] for x in res]) for k in ("rest_mib", "peak_work_mib", "peak_capacity_mib", "peak_jevbench_mib")} def latency(runs): out = defaultdict(lambda: defaultdict(list)) cap = {} for k, r in sorted(runs.items()): sh = r["shape"] if not sh: continue for name, s in sh["shapes"].items(): out[name]["e2e"].append(s["median_e2e_ms"]) out[name]["server"].append(s["median_server_ms"]) out[name]["run_medians"].extend(s["run_medians_e2e_ms"]) out[name]["tokens"].append(s["tokens"]) out[name]["non_200"].append(s["non_200"]) if sh.get("capacity"): cap[k] = {lab: [(x["rows"], x["status"]) for x in series] for lab, series in sh["capacity"].items()} return {name: {"e2e_ms": med(v["e2e"]), "server_ms": med(v["server"]), "run_medians_ms": [min(v["run_medians"]), max(v["run_medians"])], "tokens": v["tokens"][0], "non_200": sum(v["non_200"])} for name, v in out.items()}, cap def parity(runs): """semif-format runs only: authored144 'single' against SemIf's committed torch predictions.""" p = Path(__file__).resolve().parent / "data/direct-authored144.jsonl" ref = {x["id"]: x for x in map(json.loads, p.read_text().splitlines()) if x} res = [] for r in runs.values(): s = rows(r, "single") ids = [i for i in ref if i in s and s[i].get("prompt_sha256")] if not ids: continue top_ref = {i: ref[i]["option_ids"][max(range(len(ref[i]["probabilities"])), key=ref[i]["probabilities"].__getitem__)] for i in ids} res.append({"n": len(ids), "top_agree": sum(s[i]["top"] == top_ref[i] for i in ids), "prompt_sha_equal": sum(s[i]["prompt_sha256"] == ref[i]["prompt_sha256"] for i in ids), "max_dp": max(max(abs(a - b) for a, b in zip(s[i]["probs"], ref[i]["probabilities"])) for i in ids)}) return res def main(): labels = [p.name for p in sorted(OUT.iterdir()) if p.is_dir() and any(p.glob("r*/"))] allruns = {l: load_runs(l) for l in labels} summary = {} base = allruns.get("semif-qwen35-4b") for l, runs in allruns.items(): lat, cap = latency(runs) s = {"repeats": sorted(runs), "jevbench": jevbench(runs), "sets": set_table(runs), "floors": floors(runs), "order": order_sensitivity(runs), "negative": negative(runs), "vram": vram(runs), "latency": lat, "capacity": cap, "parity_vs_upstream_semif": parity(runs)} if base and l != "semif-qwen35-4b": s["paired_vs_semif"] = {f"{cond}/{name}": paired(base, runs, cond, name) for cond in ("single", "rotations") for name in ("pooled",) + SETS + ("wyrd:place2", "wyrd:exit")} # the cost question: the candidate at ONE ordering against SemIf WITH rotations s["paired_vs_semif"].update({f"cand-single-vs-semif-rotations/{name}": paired(base, runs, "rotations", name, "single") for name in ("pooled",) + SETS}) summary[l] = s json.dump(summary, sys.stdout, indent=1, default=str) if __name__ == "__main__": main()