Files
esh-pfi-infrastructure/services/semif-serve/bench-jev-2026-09-30/code/tables.py
T
vh 475d6d6bcb docs: Jev candidate bench vs SemIf (fv-ml1 GPU 3) — Intern-Decision-4B is the replacement if SemIf is displaced
Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231,
hard 0.613) reproduced exactly; negative control and a 4-restart noise floor
(0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with-
rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one
ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench
rank does not transfer. Raw per-item data kept out of git.
2026-09-30 05:00:45 -07:00

165 lines
9.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Render analyze.py's summary.json as the markdown tables of the results doc.
python3 tables.py summary.json > tables.md
Cells: median over repeats, with min..max when the repeats differ; n = repeats."""
import json
import sys
S = json.load(open(sys.argv[1]))
ORDER = ["semif-qwen35-4b", "plumb-4b-native", "plumb-4b-dropin", "imajev-4b-native",
"intern-decision-4b-native", "intern-decision-4b-dropin", "jevk5-v02-native", "jevk5-v02-dropin",
"jevk5-v03-native", "jevk5-v03-dropin"]
NAMES = {"semif-qwen35-4b": "SemIf (Qwen3.5-4B), baseline", "plumb-4b-native": "Plumb-4B, native",
"plumb-4b-dropin": "Plumb-4B, drop-in", "imajev-4b-native": "Imajev-4B, native",
"intern-decision-4b-native": "Intern-Decision-4B, native", "intern-decision-4b-dropin": "Intern-Decision-4B, drop-in",
"jevk5-v02-native": "JevK5 v0.2, native", "jevk5-v02-dropin": "JevK5 v0.2, drop-in",
"jevk5-v03-native": "JevK5 v0.3, native", "jevk5-v03-dropin": "JevK5 v0.3, drop-in"}
labels = [l for l in ORDER if l in S] + [l for l in S if l not in ORDER]
def cell(m, fmt="{:.0f}", scale=1):
if not m:
return "–"
a, lo, hi = m["median"] * scale, m["min"] * scale, m["max"] * scale
s = fmt.format(a)
if fmt.format(lo) != fmt.format(hi):
s += f" ({fmt.format(lo)}–{fmt.format(hi)})"
return s
def reps(l):
return len(S[l]["repeats"])
out = []
w = out.append
w("### Key table\n")
w("| system | fits 12 GiB? rest / peak MiB | JevBench all /231 · hard /111 | pooled /259 single · rot | Wyrd /84 single · rot | Δ pooled vs SemIf single (95% CI) | Δ pooled vs SemIf rot (95% CI) | 21 criteria, ms | 16 × ~3,900 tok, ms |")
w("|---|---|---|---|---|---|---|---|---|")
for l in labels:
x = S[l]
v, j, t, L = x["vram"], x["jevbench"], x["sets"], x["latency"]
if "all_n" not in j or "single/pooled" not in t or "rotations/pooled" not in t:
continue
pv = x.get("paired_vs_semif") or {}
def d(k):
p = pv.get(k)
return f"{p['delta_pts']:+.1f} ({p['delta_95ci_pts'][0]:+.1f}..{p['delta_95ci_pts'][1]:+.1f})" if p else "baseline"
fits = "yes" if v["peak_work_mib"] and v["peak_work_mib"]["max"] < 13500 else "?"
w(f"| {NAMES.get(l, l)} | {fits}: {cell(v['rest_mib'])} / {cell(v['peak_work_mib'])} | {cell(j['all_n'])} · {cell(j['hard_n'])} | "
f"{cell(t['single/pooled']['correct'])} · {cell(t['rotations/pooled']['correct'])} | "
f"{cell(t['single/wyrd']['correct'])} · {cell(t['rotations/wyrd']['correct'])} | {d('single/pooled')} | {d('rotations/pooled')} | "
f"{cell(L['crit21']['e2e_ms'], '{:.0f}') if 'crit21' in L else '–'} | {cell(L['long16']['e2e_ms'], '{:.0f}') if 'long16' in L else '–'} |")
w("")
w("### JevBench v1.2.16 public items (positive control + candidate score)\n")
w("| system | repeats | easy /48 | standard /72 | hard /111 | all /231 | all | p50 ms |")
w("|---|---|---|---|---|---|---|---|")
for l in labels:
j = S[l]["jevbench"]
if not j:
continue
w(f"| {NAMES.get(l, l)} | {j['all']['n']} | {cell(j['easy_n'])} | {cell(j['original_n'])} | {cell(j['hard_n'])} | "
f"{cell(j['all_n'])} | {cell(j['all'], '{:.3f}')} | {cell(j['p50_ms'], '{:.0f}')} |")
for cond in ("single", "rotations"):
w(f"\n### Replaced-baseline sets, {cond} ({'caller order' if cond == 'single' else 'n rotations averaged'})\n")
w("| system | pooled /259 | authored144 | perturb108 | Cicada w1 /31 | Cicada w2 /31 | Wyrd /84 | Wyrd place2 /21 | Wyrd exit /21 |")
w("|---|---|---|---|---|---|---|---|---|")
for l in labels:
t = S[l]["sets"]
if f"{cond}/pooled" not in t:
continue
g = lambda k: cell(t[f"{cond}/{k}"]["correct"]) if f"{cond}/{k}" in t else "–"
w(f"| {NAMES.get(l, l)} | {g('pooled')} | {g('authored144')} | {g('perturbations108')} | {g('cicada-w1')} | "
f"{g('cicada-w2')} | {g('wyrd')} | {g('wyrd:place2')} | {g('wyrd:exit')} |")
w("\n### Wyrd as one request per turn (4 decisions over one state: SemIf /decide/shared, native multi-question request)\n")
w("| system | repeats | Wyrd /84 | place /21 | place2 /21 | exit /21 | exit2 /21 |")
w("|---|---|---|---|---|---|---|")
for l in labels:
t = S[l]["sets"]
if "multifield/wyrd" not in t:
continue
g = lambda k: cell(t[f"multifield/{k}"]["correct"]) if f"multifield/{k}" in t else "–"
w(f"| {NAMES.get(l, l)} | {t['multifield/wyrd']['correct']['n']} | {g('wyrd')} | {g('wyrd:place')} | {g('wyrd:place2')} | {g('wyrd:exit')} | {g('wyrd:exit2')} |")
w("\n### Null control (content-free state) and positive controls inside the spike sets, single ordering\n")
w("| system | Cicada w1 blind /31 | Wyrd blind /84 | Cicada w1 controls /5 | Wyrd controls /28 |")
w("|---|---|---|---|---|")
for l in labels:
t = S[l]["sets"]
if "single/cicada-w1" not in t:
continue
w(f"| {NAMES.get(l, l)} | {cell(t['single/cicada-w1']['blind_correct'])} | {cell(t['single/wyrd']['blind_correct'])} | "
f"{cell(t['single/cicada-w1']['controls'])} | {cell(t['single/wyrd']['controls'])} |")
w("\n### Paired against SemIf (each row's majority top over the repeats; group bootstrap 95% CI; exact McNemar)\n")
w("| system | set | cond | SemIf | cand | fixed | broken | Δ pts | 95% CI | p |")
w("|---|---|---|---|---|---|---|---|---|---|")
for l in labels:
pv = S[l].get("paired_vs_semif") or {}
for k in ("single/pooled", "rotations/pooled", "single/authored144", "rotations/authored144",
"single/cicada-w1", "rotations/cicada-w1", "single/wyrd", "rotations/wyrd",
"single/perturbations108", "rotations/perturbations108", "single/cicada-w2", "rotations/cicada-w2",
"cand-single-vs-semif-rotations/pooled", "cand-single-vs-semif-rotations/wyrd"):
p = pv.get(k)
if not p:
continue
cond, name = k.split("/")
w(f"| {NAMES.get(l, l)} | {name} | {cond} | {p['semif']}/{p['n']} | {p['cand']}/{p['n']} | {p['fixed']} | "
f"{p['broken']} | {p['delta_pts']:+.1f} | {p['delta_95ci_pts'][0]:+.1f}..{p['delta_95ci_pts'][1]:+.1f} | {p['mcnemar_p']} |")
w("\n### Noise floors\n")
w("| system | A-vs-A in process (authored144): flips, max Δp | across restarts, single (560 rows/pair): flips per pair, max Δp | across restarts, rotations: flips per pair | labelled rows whose top moved in ANY restart pair, single: pooled /259 · perturb /108 · Cicada w2 /31 |")
w("|---|---|---|---|---|")
for l in labels:
f = S[l]["floors"]
aa = f["a_vs_a_in_process"]
aa_s = ", ".join(f"{x['flips']}/{x['n']} Δp≤{x['max_dp']:.3f}" for x in aa) or "–"
cr = f.get("cross_restart/single") or []
cr_s = ", ".join(f"{x['flips']} (Δp≤{x['max_dp']:.3f})" for x in cr) or "–"
crr = f.get("cross_restart/rotations") or []
crr_s = ", ".join(str(x["flips"]) for x in crr) or "–"
u = f.get("unstable_rows/single") or {}
u_s = f"{u.get('pooled', '–')} · {u.get('perturbations108', '–')} · {u.get('cicada-w2', '–')} ({u.get('repeats', '?')} repeats)" if u else "–"
w(f"| {NAMES.get(l, l)} | {aa_s} | {cr_s} | {crr_s} | {u_s} |")
w("\n### Order sensitivity (authored144, single ordering vs the same request reordered) and the negative control\n")
w("| system | reversed: label changes /144 | reversed: max Δp | shuffled: label changes | shuffled: max Δp | NEG: same top as unrotated /144 | NEG: follows the description | NEG: right vs original gold |")
w("|---|---|---|---|---|---|---|---|")
for l in labels:
o, n = S[l]["order"], S[l]["negative"]
if not o:
continue
w(f"| {NAMES.get(l, l)} | {cell(o['reversed']['label_changes'])} | {cell(o['reversed']['max_dp'], '{:.2f}')} | "
f"{cell(o['shuffled']['label_changes'])} | {cell(o['shuffled']['max_dp'], '{:.2f}')} | "
f"{cell(n.get('same_top_as_unrotated'))} | {cell(n.get('follows_description'))} | {cell(n.get('vs_original_gold'))} |")
w("\n### Latency at our shape (loopback on fv-ml1, GPU 3; ms; median of all requests, run-median range)\n")
w("| system | short1 e2e | crit21 e2e | crit21 server | long1 e2e | long16 e2e | long16 server | tokens crit21 / long16 |")
w("|---|---|---|---|---|---|---|---|")
for l in labels:
L = S[l]["latency"]
if not L:
continue
g = lambda k, f="e2e_ms": cell(L[k][f], "{:.0f}") if k in L else "–"
w(f"| {NAMES.get(l, l)} | {g('short1')} | {g('crit21')} | {g('crit21', 'server_ms')} | {g('long1')} | {g('long16')} | "
f"{g('long16', 'server_ms')} | {L.get('crit21', {}).get('tokens')} / {L.get('long16', {}).get('tokens')} |")
w("\n### VRAM (nvidia-smi, whole GPU 3, only our process on it) and capacity under the 12 GiB cap\n")
w("| system | rest after warm-up MiB | peak in sets+shapes MiB | peak in capacity sweep MiB | max rows @ short state | max rows @ ~3,900 tok |")
w("|---|---|---|---|---|---|")
for l in labels:
v, cap = S[l]["vram"], S[l]["capacity"]
def mx(lab):
vals = []
for r, c in cap.items():
ok = [n for n, s in c.get(lab, []) if s == 200]
bad = [(n, s) for n, s in c.get(lab, []) if s != 200]
vals.append(f"{max(ok) if ok else 0}" + (f" (next {bad[0][0]}: {bad[0][1]})" if bad else " (no failure up to the last step)"))
return "; ".join(vals) or "–"
w(f"| {NAMES.get(l, l)} | {cell(v['rest_mib'])} | {cell(v['peak_work_mib'])} | {cell(v['peak_capacity_mib'])} | {mx('short')} | {mx('long')} |")
print("\n".join(out))