"""Render analyze.py's summary.json as the markdown tables of the results doc. python3 tables.py summary.json > tables.md Cells: median over repeats, with min..max when the repeats differ; n = repeats.""" import json import sys S = json.load(open(sys.argv[1])) ORDER = ["semif-qwen35-4b", "plumb-4b-native", "plumb-4b-dropin", "imajev-4b-native", "intern-decision-4b-native", "intern-decision-4b-dropin", "jevk5-v02-native", "jevk5-v02-dropin", "jevk5-v03-native", "jevk5-v03-dropin"] NAMES = {"semif-qwen35-4b": "SemIf (Qwen3.5-4B), baseline", "plumb-4b-native": "Plumb-4B, native", "plumb-4b-dropin": "Plumb-4B, drop-in", "imajev-4b-native": "Imajev-4B, native", "intern-decision-4b-native": "Intern-Decision-4B, native", "intern-decision-4b-dropin": "Intern-Decision-4B, drop-in", "jevk5-v02-native": "JevK5 v0.2, native", "jevk5-v02-dropin": "JevK5 v0.2, drop-in", "jevk5-v03-native": "JevK5 v0.3, native", "jevk5-v03-dropin": "JevK5 v0.3, drop-in"} labels = [l for l in ORDER if l in S] + [l for l in S if l not in ORDER] def cell(m, fmt="{:.0f}", scale=1): if not m: return "–" a, lo, hi = m["median"] * scale, m["min"] * scale, m["max"] * scale s = fmt.format(a) if fmt.format(lo) != fmt.format(hi): s += f" ({fmt.format(lo)}–{fmt.format(hi)})" return s def reps(l): return len(S[l]["repeats"]) out = [] w = out.append w("### Key table\n") w("| system | fits 12 GiB? rest / peak MiB | JevBench all /231 · hard /111 | pooled /259 single · rot | Wyrd /84 single · rot | Δ pooled vs SemIf single (95% CI) | Δ pooled vs SemIf rot (95% CI) | 21 criteria, ms | 16 × ~3,900 tok, ms |") w("|---|---|---|---|---|---|---|---|---|") for l in labels: x = S[l] v, j, t, L = x["vram"], x["jevbench"], x["sets"], x["latency"] if "all_n" not in j or "single/pooled" not in t or "rotations/pooled" not in t: continue pv = x.get("paired_vs_semif") or {} def d(k): p = pv.get(k) return f"{p['delta_pts']:+.1f} ({p['delta_95ci_pts'][0]:+.1f}..{p['delta_95ci_pts'][1]:+.1f})" if p else "baseline" fits = "yes" if v["peak_work_mib"] and v["peak_work_mib"]["max"] < 13500 else "?" w(f"| {NAMES.get(l, l)} | {fits}: {cell(v['rest_mib'])} / {cell(v['peak_work_mib'])} | {cell(j['all_n'])} · {cell(j['hard_n'])} | " f"{cell(t['single/pooled']['correct'])} · {cell(t['rotations/pooled']['correct'])} | " f"{cell(t['single/wyrd']['correct'])} · {cell(t['rotations/wyrd']['correct'])} | {d('single/pooled')} | {d('rotations/pooled')} | " f"{cell(L['crit21']['e2e_ms'], '{:.0f}') if 'crit21' in L else '–'} | {cell(L['long16']['e2e_ms'], '{:.0f}') if 'long16' in L else '–'} |") w("") w("### JevBench v1.2.16 public items (positive control + candidate score)\n") w("| system | repeats | easy /48 | standard /72 | hard /111 | all /231 | all | p50 ms |") w("|---|---|---|---|---|---|---|---|") for l in labels: j = S[l]["jevbench"] if not j: continue w(f"| {NAMES.get(l, l)} | {j['all']['n']} | {cell(j['easy_n'])} | {cell(j['original_n'])} | {cell(j['hard_n'])} | " f"{cell(j['all_n'])} | {cell(j['all'], '{:.3f}')} | {cell(j['p50_ms'], '{:.0f}')} |") for cond in ("single", "rotations"): w(f"\n### Replaced-baseline sets, {cond} ({'caller order' if cond == 'single' else 'n rotations averaged'})\n") w("| system | pooled /259 | authored144 | perturb108 | Cicada w1 /31 | Cicada w2 /31 | Wyrd /84 | Wyrd place2 /21 | Wyrd exit /21 |") w("|---|---|---|---|---|---|---|---|---|") for l in labels: t = S[l]["sets"] if f"{cond}/pooled" not in t: continue g = lambda k: cell(t[f"{cond}/{k}"]["correct"]) if f"{cond}/{k}" in t else "–" w(f"| {NAMES.get(l, l)} | {g('pooled')} | {g('authored144')} | {g('perturbations108')} | {g('cicada-w1')} | " f"{g('cicada-w2')} | {g('wyrd')} | {g('wyrd:place2')} | {g('wyrd:exit')} |") w("\n### Wyrd as one request per turn (4 decisions over one state: SemIf /decide/shared, native multi-question request)\n") w("| system | repeats | Wyrd /84 | place /21 | place2 /21 | exit /21 | exit2 /21 |") w("|---|---|---|---|---|---|---|") for l in labels: t = S[l]["sets"] if "multifield/wyrd" not in t: continue g = lambda k: cell(t[f"multifield/{k}"]["correct"]) if f"multifield/{k}" in t else "–" w(f"| {NAMES.get(l, l)} | {t['multifield/wyrd']['correct']['n']} | {g('wyrd')} | {g('wyrd:place')} | {g('wyrd:place2')} | {g('wyrd:exit')} | {g('wyrd:exit2')} |") w("\n### Null control (content-free state) and positive controls inside the spike sets, single ordering\n") w("| system | Cicada w1 blind /31 | Wyrd blind /84 | Cicada w1 controls /5 | Wyrd controls /28 |") w("|---|---|---|---|---|") for l in labels: t = S[l]["sets"] if "single/cicada-w1" not in t: continue w(f"| {NAMES.get(l, l)} | {cell(t['single/cicada-w1']['blind_correct'])} | {cell(t['single/wyrd']['blind_correct'])} | " f"{cell(t['single/cicada-w1']['controls'])} | {cell(t['single/wyrd']['controls'])} |") w("\n### Paired against SemIf (each row's majority top over the repeats; group bootstrap 95% CI; exact McNemar)\n") w("| system | set | cond | SemIf | cand | fixed | broken | Δ pts | 95% CI | p |") w("|---|---|---|---|---|---|---|---|---|---|") for l in labels: pv = S[l].get("paired_vs_semif") or {} for k in ("single/pooled", "rotations/pooled", "single/authored144", "rotations/authored144", "single/cicada-w1", "rotations/cicada-w1", "single/wyrd", "rotations/wyrd", "single/perturbations108", "rotations/perturbations108", "single/cicada-w2", "rotations/cicada-w2", "cand-single-vs-semif-rotations/pooled", "cand-single-vs-semif-rotations/wyrd"): p = pv.get(k) if not p: continue cond, name = k.split("/") w(f"| {NAMES.get(l, l)} | {name} | {cond} | {p['semif']}/{p['n']} | {p['cand']}/{p['n']} | {p['fixed']} | " f"{p['broken']} | {p['delta_pts']:+.1f} | {p['delta_95ci_pts'][0]:+.1f}..{p['delta_95ci_pts'][1]:+.1f} | {p['mcnemar_p']} |") w("\n### Noise floors\n") w("| system | A-vs-A in process (authored144): flips, max Δp | across restarts, single (560 rows/pair): flips per pair, max Δp | across restarts, rotations: flips per pair | labelled rows whose top moved in ANY restart pair, single: pooled /259 · perturb /108 · Cicada w2 /31 |") w("|---|---|---|---|---|") for l in labels: f = S[l]["floors"] aa = f["a_vs_a_in_process"] aa_s = ", ".join(f"{x['flips']}/{x['n']} Δp≤{x['max_dp']:.3f}" for x in aa) or "–" cr = f.get("cross_restart/single") or [] cr_s = ", ".join(f"{x['flips']} (Δp≤{x['max_dp']:.3f})" for x in cr) or "–" crr = f.get("cross_restart/rotations") or [] crr_s = ", ".join(str(x["flips"]) for x in crr) or "–" u = f.get("unstable_rows/single") or {} u_s = f"{u.get('pooled', '–')} · {u.get('perturbations108', '–')} · {u.get('cicada-w2', '–')} ({u.get('repeats', '?')} repeats)" if u else "–" w(f"| {NAMES.get(l, l)} | {aa_s} | {cr_s} | {crr_s} | {u_s} |") w("\n### Order sensitivity (authored144, single ordering vs the same request reordered) and the negative control\n") w("| system | reversed: label changes /144 | reversed: max Δp | shuffled: label changes | shuffled: max Δp | NEG: same top as unrotated /144 | NEG: follows the description | NEG: right vs original gold |") w("|---|---|---|---|---|---|---|---|") for l in labels: o, n = S[l]["order"], S[l]["negative"] if not o: continue w(f"| {NAMES.get(l, l)} | {cell(o['reversed']['label_changes'])} | {cell(o['reversed']['max_dp'], '{:.2f}')} | " f"{cell(o['shuffled']['label_changes'])} | {cell(o['shuffled']['max_dp'], '{:.2f}')} | " f"{cell(n.get('same_top_as_unrotated'))} | {cell(n.get('follows_description'))} | {cell(n.get('vs_original_gold'))} |") w("\n### Latency at our shape (loopback on fv-ml1, GPU 3; ms; median of all requests, run-median range)\n") w("| system | short1 e2e | crit21 e2e | crit21 server | long1 e2e | long16 e2e | long16 server | tokens crit21 / long16 |") w("|---|---|---|---|---|---|---|---|") for l in labels: L = S[l]["latency"] if not L: continue g = lambda k, f="e2e_ms": cell(L[k][f], "{:.0f}") if k in L else "–" w(f"| {NAMES.get(l, l)} | {g('short1')} | {g('crit21')} | {g('crit21', 'server_ms')} | {g('long1')} | {g('long16')} | " f"{g('long16', 'server_ms')} | {L.get('crit21', {}).get('tokens')} / {L.get('long16', {}).get('tokens')} |") w("\n### VRAM (nvidia-smi, whole GPU 3, only our process on it) and capacity under the 12 GiB cap\n") w("| system | rest after warm-up MiB | peak in sets+shapes MiB | peak in capacity sweep MiB | max rows @ short state | max rows @ ~3,900 tok |") w("|---|---|---|---|---|---|") for l in labels: v, cap = S[l]["vram"], S[l]["capacity"] def mx(lab): vals = [] for r, c in cap.items(): ok = [n for n, s in c.get(lab, []) if s == 200] bad = [(n, s) for n, s in c.get(lab, []) if s != 200] vals.append(f"{max(ok) if ok else 0}" + (f" (next {bad[0][0]}: {bad[0][1]})" if bad else " (no failure up to the last step)")) return "; ".join(vals) or "–" w(f"| {NAMES.get(l, l)} | {cell(v['rest_mib'])} | {cell(v['peak_work_mib'])} | {cell(v['peak_capacity_mib'])} | {mx('short')} | {mx('long')} |") print("\n".join(out))