Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231, hard 0.613) reproduced exactly; negative control and a 4-restart noise floor (0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with- rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench rank does not transfer. Raw per-item data kept out of git.
165 lines
9.3 KiB
Python
165 lines
9.3 KiB
Python
"""Render analyze.py's summary.json as the markdown tables of the results doc.
|
||
python3 tables.py summary.json > tables.md
|
||
Cells: median over repeats, with min..max when the repeats differ; n = repeats."""
|
||
import json
|
||
import sys
|
||
|
||
S = json.load(open(sys.argv[1]))
|
||
ORDER = ["semif-qwen35-4b", "plumb-4b-native", "plumb-4b-dropin", "imajev-4b-native",
|
||
"intern-decision-4b-native", "intern-decision-4b-dropin", "jevk5-v02-native", "jevk5-v02-dropin",
|
||
"jevk5-v03-native", "jevk5-v03-dropin"]
|
||
NAMES = {"semif-qwen35-4b": "SemIf (Qwen3.5-4B), baseline", "plumb-4b-native": "Plumb-4B, native",
|
||
"plumb-4b-dropin": "Plumb-4B, drop-in", "imajev-4b-native": "Imajev-4B, native",
|
||
"intern-decision-4b-native": "Intern-Decision-4B, native", "intern-decision-4b-dropin": "Intern-Decision-4B, drop-in",
|
||
"jevk5-v02-native": "JevK5 v0.2, native", "jevk5-v02-dropin": "JevK5 v0.2, drop-in",
|
||
"jevk5-v03-native": "JevK5 v0.3, native", "jevk5-v03-dropin": "JevK5 v0.3, drop-in"}
|
||
labels = [l for l in ORDER if l in S] + [l for l in S if l not in ORDER]
|
||
|
||
|
||
def cell(m, fmt="{:.0f}", scale=1):
|
||
if not m:
|
||
return "–"
|
||
a, lo, hi = m["median"] * scale, m["min"] * scale, m["max"] * scale
|
||
s = fmt.format(a)
|
||
if fmt.format(lo) != fmt.format(hi):
|
||
s += f" ({fmt.format(lo)}–{fmt.format(hi)})"
|
||
return s
|
||
|
||
|
||
def reps(l):
|
||
return len(S[l]["repeats"])
|
||
|
||
|
||
out = []
|
||
w = out.append
|
||
|
||
w("### Key table\n")
|
||
w("| system | fits 12 GiB? rest / peak MiB | JevBench all /231 · hard /111 | pooled /259 single · rot | Wyrd /84 single · rot | Δ pooled vs SemIf single (95% CI) | Δ pooled vs SemIf rot (95% CI) | 21 criteria, ms | 16 × ~3,900 tok, ms |")
|
||
w("|---|---|---|---|---|---|---|---|---|")
|
||
for l in labels:
|
||
x = S[l]
|
||
v, j, t, L = x["vram"], x["jevbench"], x["sets"], x["latency"]
|
||
if "all_n" not in j or "single/pooled" not in t or "rotations/pooled" not in t:
|
||
continue
|
||
pv = x.get("paired_vs_semif") or {}
|
||
def d(k):
|
||
p = pv.get(k)
|
||
return f"{p['delta_pts']:+.1f} ({p['delta_95ci_pts'][0]:+.1f}..{p['delta_95ci_pts'][1]:+.1f})" if p else "baseline"
|
||
fits = "yes" if v["peak_work_mib"] and v["peak_work_mib"]["max"] < 13500 else "?"
|
||
w(f"| {NAMES.get(l, l)} | {fits}: {cell(v['rest_mib'])} / {cell(v['peak_work_mib'])} | {cell(j['all_n'])} · {cell(j['hard_n'])} | "
|
||
f"{cell(t['single/pooled']['correct'])} · {cell(t['rotations/pooled']['correct'])} | "
|
||
f"{cell(t['single/wyrd']['correct'])} · {cell(t['rotations/wyrd']['correct'])} | {d('single/pooled')} | {d('rotations/pooled')} | "
|
||
f"{cell(L['crit21']['e2e_ms'], '{:.0f}') if 'crit21' in L else '–'} | {cell(L['long16']['e2e_ms'], '{:.0f}') if 'long16' in L else '–'} |")
|
||
w("")
|
||
|
||
w("### JevBench v1.2.16 public items (positive control + candidate score)\n")
|
||
w("| system | repeats | easy /48 | standard /72 | hard /111 | all /231 | all | p50 ms |")
|
||
w("|---|---|---|---|---|---|---|---|")
|
||
for l in labels:
|
||
j = S[l]["jevbench"]
|
||
if not j:
|
||
continue
|
||
w(f"| {NAMES.get(l, l)} | {j['all']['n']} | {cell(j['easy_n'])} | {cell(j['original_n'])} | {cell(j['hard_n'])} | "
|
||
f"{cell(j['all_n'])} | {cell(j['all'], '{:.3f}')} | {cell(j['p50_ms'], '{:.0f}')} |")
|
||
|
||
for cond in ("single", "rotations"):
|
||
w(f"\n### Replaced-baseline sets, {cond} ({'caller order' if cond == 'single' else 'n rotations averaged'})\n")
|
||
w("| system | pooled /259 | authored144 | perturb108 | Cicada w1 /31 | Cicada w2 /31 | Wyrd /84 | Wyrd place2 /21 | Wyrd exit /21 |")
|
||
w("|---|---|---|---|---|---|---|---|---|")
|
||
for l in labels:
|
||
t = S[l]["sets"]
|
||
if f"{cond}/pooled" not in t:
|
||
continue
|
||
g = lambda k: cell(t[f"{cond}/{k}"]["correct"]) if f"{cond}/{k}" in t else "–"
|
||
w(f"| {NAMES.get(l, l)} | {g('pooled')} | {g('authored144')} | {g('perturbations108')} | {g('cicada-w1')} | "
|
||
f"{g('cicada-w2')} | {g('wyrd')} | {g('wyrd:place2')} | {g('wyrd:exit')} |")
|
||
|
||
w("\n### Wyrd as one request per turn (4 decisions over one state: SemIf /decide/shared, native multi-question request)\n")
|
||
w("| system | repeats | Wyrd /84 | place /21 | place2 /21 | exit /21 | exit2 /21 |")
|
||
w("|---|---|---|---|---|---|---|")
|
||
for l in labels:
|
||
t = S[l]["sets"]
|
||
if "multifield/wyrd" not in t:
|
||
continue
|
||
g = lambda k: cell(t[f"multifield/{k}"]["correct"]) if f"multifield/{k}" in t else "–"
|
||
w(f"| {NAMES.get(l, l)} | {t['multifield/wyrd']['correct']['n']} | {g('wyrd')} | {g('wyrd:place')} | {g('wyrd:place2')} | {g('wyrd:exit')} | {g('wyrd:exit2')} |")
|
||
|
||
w("\n### Null control (content-free state) and positive controls inside the spike sets, single ordering\n")
|
||
w("| system | Cicada w1 blind /31 | Wyrd blind /84 | Cicada w1 controls /5 | Wyrd controls /28 |")
|
||
w("|---|---|---|---|---|")
|
||
for l in labels:
|
||
t = S[l]["sets"]
|
||
if "single/cicada-w1" not in t:
|
||
continue
|
||
w(f"| {NAMES.get(l, l)} | {cell(t['single/cicada-w1']['blind_correct'])} | {cell(t['single/wyrd']['blind_correct'])} | "
|
||
f"{cell(t['single/cicada-w1']['controls'])} | {cell(t['single/wyrd']['controls'])} |")
|
||
|
||
w("\n### Paired against SemIf (each row's majority top over the repeats; group bootstrap 95% CI; exact McNemar)\n")
|
||
w("| system | set | cond | SemIf | cand | fixed | broken | Δ pts | 95% CI | p |")
|
||
w("|---|---|---|---|---|---|---|---|---|---|")
|
||
for l in labels:
|
||
pv = S[l].get("paired_vs_semif") or {}
|
||
for k in ("single/pooled", "rotations/pooled", "single/authored144", "rotations/authored144",
|
||
"single/cicada-w1", "rotations/cicada-w1", "single/wyrd", "rotations/wyrd",
|
||
"single/perturbations108", "rotations/perturbations108", "single/cicada-w2", "rotations/cicada-w2",
|
||
"cand-single-vs-semif-rotations/pooled", "cand-single-vs-semif-rotations/wyrd"):
|
||
p = pv.get(k)
|
||
if not p:
|
||
continue
|
||
cond, name = k.split("/")
|
||
w(f"| {NAMES.get(l, l)} | {name} | {cond} | {p['semif']}/{p['n']} | {p['cand']}/{p['n']} | {p['fixed']} | "
|
||
f"{p['broken']} | {p['delta_pts']:+.1f} | {p['delta_95ci_pts'][0]:+.1f}..{p['delta_95ci_pts'][1]:+.1f} | {p['mcnemar_p']} |")
|
||
|
||
w("\n### Noise floors\n")
|
||
w("| system | A-vs-A in process (authored144): flips, max Δp | across restarts, single (560 rows/pair): flips per pair, max Δp | across restarts, rotations: flips per pair | labelled rows whose top moved in ANY restart pair, single: pooled /259 · perturb /108 · Cicada w2 /31 |")
|
||
w("|---|---|---|---|---|")
|
||
for l in labels:
|
||
f = S[l]["floors"]
|
||
aa = f["a_vs_a_in_process"]
|
||
aa_s = ", ".join(f"{x['flips']}/{x['n']} Δp≤{x['max_dp']:.3f}" for x in aa) or "–"
|
||
cr = f.get("cross_restart/single") or []
|
||
cr_s = ", ".join(f"{x['flips']} (Δp≤{x['max_dp']:.3f})" for x in cr) or "–"
|
||
crr = f.get("cross_restart/rotations") or []
|
||
crr_s = ", ".join(str(x["flips"]) for x in crr) or "–"
|
||
u = f.get("unstable_rows/single") or {}
|
||
u_s = f"{u.get('pooled', '–')} · {u.get('perturbations108', '–')} · {u.get('cicada-w2', '–')} ({u.get('repeats', '?')} repeats)" if u else "–"
|
||
w(f"| {NAMES.get(l, l)} | {aa_s} | {cr_s} | {crr_s} | {u_s} |")
|
||
|
||
w("\n### Order sensitivity (authored144, single ordering vs the same request reordered) and the negative control\n")
|
||
w("| system | reversed: label changes /144 | reversed: max Δp | shuffled: label changes | shuffled: max Δp | NEG: same top as unrotated /144 | NEG: follows the description | NEG: right vs original gold |")
|
||
w("|---|---|---|---|---|---|---|---|")
|
||
for l in labels:
|
||
o, n = S[l]["order"], S[l]["negative"]
|
||
if not o:
|
||
continue
|
||
w(f"| {NAMES.get(l, l)} | {cell(o['reversed']['label_changes'])} | {cell(o['reversed']['max_dp'], '{:.2f}')} | "
|
||
f"{cell(o['shuffled']['label_changes'])} | {cell(o['shuffled']['max_dp'], '{:.2f}')} | "
|
||
f"{cell(n.get('same_top_as_unrotated'))} | {cell(n.get('follows_description'))} | {cell(n.get('vs_original_gold'))} |")
|
||
|
||
w("\n### Latency at our shape (loopback on fv-ml1, GPU 3; ms; median of all requests, run-median range)\n")
|
||
w("| system | short1 e2e | crit21 e2e | crit21 server | long1 e2e | long16 e2e | long16 server | tokens crit21 / long16 |")
|
||
w("|---|---|---|---|---|---|---|---|")
|
||
for l in labels:
|
||
L = S[l]["latency"]
|
||
if not L:
|
||
continue
|
||
g = lambda k, f="e2e_ms": cell(L[k][f], "{:.0f}") if k in L else "–"
|
||
w(f"| {NAMES.get(l, l)} | {g('short1')} | {g('crit21')} | {g('crit21', 'server_ms')} | {g('long1')} | {g('long16')} | "
|
||
f"{g('long16', 'server_ms')} | {L.get('crit21', {}).get('tokens')} / {L.get('long16', {}).get('tokens')} |")
|
||
|
||
w("\n### VRAM (nvidia-smi, whole GPU 3, only our process on it) and capacity under the 12 GiB cap\n")
|
||
w("| system | rest after warm-up MiB | peak in sets+shapes MiB | peak in capacity sweep MiB | max rows @ short state | max rows @ ~3,900 tok |")
|
||
w("|---|---|---|---|---|---|")
|
||
for l in labels:
|
||
v, cap = S[l]["vram"], S[l]["capacity"]
|
||
def mx(lab):
|
||
vals = []
|
||
for r, c in cap.items():
|
||
ok = [n for n, s in c.get(lab, []) if s == 200]
|
||
bad = [(n, s) for n, s in c.get(lab, []) if s != 200]
|
||
vals.append(f"{max(ok) if ok else 0}" + (f" (next {bad[0][0]}: {bad[0][1]})" if bad else " (no failure up to the last step)"))
|
||
return "; ".join(vals) or "–"
|
||
w(f"| {NAMES.get(l, l)} | {cell(v['rest_mib'])} | {cell(v['peak_work_mib'])} | {cell(v['peak_capacity_mib'])} | {mx('short')} | {mx('long')} |")
|
||
|
||
print("\n".join(out))
|