docs: Jev candidate bench vs SemIf (fv-ml1 GPU 3) — Intern-Decision-4B is the replacement if SemIf is displaced

Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231,
hard 0.613) reproduced exactly; negative control and a 4-restart noise floor
(0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with-
rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one
ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench
rank does not transfer. Raw per-item data kept out of git.
This commit is contained in:
vh
2026-09-30 05:00:45 -07:00
parent 9a6ac59da7
commit 475d6d6bcb
27 changed files with 19068 additions and 0 deletions
@@ -0,0 +1,164 @@
"""Render analyze.py's summary.json as the markdown tables of the results doc.
python3 tables.py summary.json > tables.md
Cells: median over repeats, with min..max when the repeats differ; n = repeats."""
import json
import sys
S = json.load(open(sys.argv[1]))
ORDER = ["semif-qwen35-4b", "plumb-4b-native", "plumb-4b-dropin", "imajev-4b-native",
"intern-decision-4b-native", "intern-decision-4b-dropin", "jevk5-v02-native", "jevk5-v02-dropin",
"jevk5-v03-native", "jevk5-v03-dropin"]
NAMES = {"semif-qwen35-4b": "SemIf (Qwen3.5-4B), baseline", "plumb-4b-native": "Plumb-4B, native",
"plumb-4b-dropin": "Plumb-4B, drop-in", "imajev-4b-native": "Imajev-4B, native",
"intern-decision-4b-native": "Intern-Decision-4B, native", "intern-decision-4b-dropin": "Intern-Decision-4B, drop-in",
"jevk5-v02-native": "JevK5 v0.2, native", "jevk5-v02-dropin": "JevK5 v0.2, drop-in",
"jevk5-v03-native": "JevK5 v0.3, native", "jevk5-v03-dropin": "JevK5 v0.3, drop-in"}
labels = [l for l in ORDER if l in S] + [l for l in S if l not in ORDER]
def cell(m, fmt="{:.0f}", scale=1):
if not m:
return "–"
a, lo, hi = m["median"] * scale, m["min"] * scale, m["max"] * scale
s = fmt.format(a)
if fmt.format(lo) != fmt.format(hi):
s += f" ({fmt.format(lo)}–{fmt.format(hi)})"
return s
def reps(l):
return len(S[l]["repeats"])
out = []
w = out.append
w("### Key table\n")
w("| system | fits 12 GiB? rest / peak MiB | JevBench all /231 · hard /111 | pooled /259 single · rot | Wyrd /84 single · rot | Δ pooled vs SemIf single (95% CI) | Δ pooled vs SemIf rot (95% CI) | 21 criteria, ms | 16 × ~3,900 tok, ms |")
w("|---|---|---|---|---|---|---|---|---|")
for l in labels:
x = S[l]
v, j, t, L = x["vram"], x["jevbench"], x["sets"], x["latency"]
if "all_n" not in j or "single/pooled" not in t or "rotations/pooled" not in t:
continue
pv = x.get("paired_vs_semif") or {}
def d(k):
p = pv.get(k)
return f"{p['delta_pts']:+.1f} ({p['delta_95ci_pts'][0]:+.1f}..{p['delta_95ci_pts'][1]:+.1f})" if p else "baseline"
fits = "yes" if v["peak_work_mib"] and v["peak_work_mib"]["max"] < 13500 else "?"
w(f"| {NAMES.get(l, l)} | {fits}: {cell(v['rest_mib'])} / {cell(v['peak_work_mib'])} | {cell(j['all_n'])} · {cell(j['hard_n'])} | "
f"{cell(t['single/pooled']['correct'])} · {cell(t['rotations/pooled']['correct'])} | "
f"{cell(t['single/wyrd']['correct'])} · {cell(t['rotations/wyrd']['correct'])} | {d('single/pooled')} | {d('rotations/pooled')} | "
f"{cell(L['crit21']['e2e_ms'], '{:.0f}') if 'crit21' in L else '–'} | {cell(L['long16']['e2e_ms'], '{:.0f}') if 'long16' in L else '–'} |")
w("")
w("### JevBench v1.2.16 public items (positive control + candidate score)\n")
w("| system | repeats | easy /48 | standard /72 | hard /111 | all /231 | all | p50 ms |")
w("|---|---|---|---|---|---|---|---|")
for l in labels:
j = S[l]["jevbench"]
if not j:
continue
w(f"| {NAMES.get(l, l)} | {j['all']['n']} | {cell(j['easy_n'])} | {cell(j['original_n'])} | {cell(j['hard_n'])} | "
f"{cell(j['all_n'])} | {cell(j['all'], '{:.3f}')} | {cell(j['p50_ms'], '{:.0f}')} |")
for cond in ("single", "rotations"):
w(f"\n### Replaced-baseline sets, {cond} ({'caller order' if cond == 'single' else 'n rotations averaged'})\n")
w("| system | pooled /259 | authored144 | perturb108 | Cicada w1 /31 | Cicada w2 /31 | Wyrd /84 | Wyrd place2 /21 | Wyrd exit /21 |")
w("|---|---|---|---|---|---|---|---|---|")
for l in labels:
t = S[l]["sets"]
if f"{cond}/pooled" not in t:
continue
g = lambda k: cell(t[f"{cond}/{k}"]["correct"]) if f"{cond}/{k}" in t else "–"
w(f"| {NAMES.get(l, l)} | {g('pooled')} | {g('authored144')} | {g('perturbations108')} | {g('cicada-w1')} | "
f"{g('cicada-w2')} | {g('wyrd')} | {g('wyrd:place2')} | {g('wyrd:exit')} |")
w("\n### Wyrd as one request per turn (4 decisions over one state: SemIf /decide/shared, native multi-question request)\n")
w("| system | repeats | Wyrd /84 | place /21 | place2 /21 | exit /21 | exit2 /21 |")
w("|---|---|---|---|---|---|---|")
for l in labels:
t = S[l]["sets"]
if "multifield/wyrd" not in t:
continue
g = lambda k: cell(t[f"multifield/{k}"]["correct"]) if f"multifield/{k}" in t else "–"
w(f"| {NAMES.get(l, l)} | {t['multifield/wyrd']['correct']['n']} | {g('wyrd')} | {g('wyrd:place')} | {g('wyrd:place2')} | {g('wyrd:exit')} | {g('wyrd:exit2')} |")
w("\n### Null control (content-free state) and positive controls inside the spike sets, single ordering\n")
w("| system | Cicada w1 blind /31 | Wyrd blind /84 | Cicada w1 controls /5 | Wyrd controls /28 |")
w("|---|---|---|---|---|")
for l in labels:
t = S[l]["sets"]
if "single/cicada-w1" not in t:
continue
w(f"| {NAMES.get(l, l)} | {cell(t['single/cicada-w1']['blind_correct'])} | {cell(t['single/wyrd']['blind_correct'])} | "
f"{cell(t['single/cicada-w1']['controls'])} | {cell(t['single/wyrd']['controls'])} |")
w("\n### Paired against SemIf (each row's majority top over the repeats; group bootstrap 95% CI; exact McNemar)\n")
w("| system | set | cond | SemIf | cand | fixed | broken | Δ pts | 95% CI | p |")
w("|---|---|---|---|---|---|---|---|---|---|")
for l in labels:
pv = S[l].get("paired_vs_semif") or {}
for k in ("single/pooled", "rotations/pooled", "single/authored144", "rotations/authored144",
"single/cicada-w1", "rotations/cicada-w1", "single/wyrd", "rotations/wyrd",
"single/perturbations108", "rotations/perturbations108", "single/cicada-w2", "rotations/cicada-w2",
"cand-single-vs-semif-rotations/pooled", "cand-single-vs-semif-rotations/wyrd"):
p = pv.get(k)
if not p:
continue
cond, name = k.split("/")
w(f"| {NAMES.get(l, l)} | {name} | {cond} | {p['semif']}/{p['n']} | {p['cand']}/{p['n']} | {p['fixed']} | "
f"{p['broken']} | {p['delta_pts']:+.1f} | {p['delta_95ci_pts'][0]:+.1f}..{p['delta_95ci_pts'][1]:+.1f} | {p['mcnemar_p']} |")
w("\n### Noise floors\n")
w("| system | A-vs-A in process (authored144): flips, max Δp | across restarts, single (560 rows/pair): flips per pair, max Δp | across restarts, rotations: flips per pair | labelled rows whose top moved in ANY restart pair, single: pooled /259 · perturb /108 · Cicada w2 /31 |")
w("|---|---|---|---|---|")
for l in labels:
f = S[l]["floors"]
aa = f["a_vs_a_in_process"]
aa_s = ", ".join(f"{x['flips']}/{x['n']} Δp≤{x['max_dp']:.3f}" for x in aa) or "–"
cr = f.get("cross_restart/single") or []
cr_s = ", ".join(f"{x['flips']} (Δp≤{x['max_dp']:.3f})" for x in cr) or "–"
crr = f.get("cross_restart/rotations") or []
crr_s = ", ".join(str(x["flips"]) for x in crr) or "–"
u = f.get("unstable_rows/single") or {}
u_s = f"{u.get('pooled', '–')} · {u.get('perturbations108', '–')} · {u.get('cicada-w2', '–')} ({u.get('repeats', '?')} repeats)" if u else "–"
w(f"| {NAMES.get(l, l)} | {aa_s} | {cr_s} | {crr_s} | {u_s} |")
w("\n### Order sensitivity (authored144, single ordering vs the same request reordered) and the negative control\n")
w("| system | reversed: label changes /144 | reversed: max Δp | shuffled: label changes | shuffled: max Δp | NEG: same top as unrotated /144 | NEG: follows the description | NEG: right vs original gold |")
w("|---|---|---|---|---|---|---|---|")
for l in labels:
o, n = S[l]["order"], S[l]["negative"]
if not o:
continue
w(f"| {NAMES.get(l, l)} | {cell(o['reversed']['label_changes'])} | {cell(o['reversed']['max_dp'], '{:.2f}')} | "
f"{cell(o['shuffled']['label_changes'])} | {cell(o['shuffled']['max_dp'], '{:.2f}')} | "
f"{cell(n.get('same_top_as_unrotated'))} | {cell(n.get('follows_description'))} | {cell(n.get('vs_original_gold'))} |")
w("\n### Latency at our shape (loopback on fv-ml1, GPU 3; ms; median of all requests, run-median range)\n")
w("| system | short1 e2e | crit21 e2e | crit21 server | long1 e2e | long16 e2e | long16 server | tokens crit21 / long16 |")
w("|---|---|---|---|---|---|---|---|")
for l in labels:
L = S[l]["latency"]
if not L:
continue
g = lambda k, f="e2e_ms": cell(L[k][f], "{:.0f}") if k in L else "–"
w(f"| {NAMES.get(l, l)} | {g('short1')} | {g('crit21')} | {g('crit21', 'server_ms')} | {g('long1')} | {g('long16')} | "
f"{g('long16', 'server_ms')} | {L.get('crit21', {}).get('tokens')} / {L.get('long16', {}).get('tokens')} |")
w("\n### VRAM (nvidia-smi, whole GPU 3, only our process on it) and capacity under the 12 GiB cap\n")
w("| system | rest after warm-up MiB | peak in sets+shapes MiB | peak in capacity sweep MiB | max rows @ short state | max rows @ ~3,900 tok |")
w("|---|---|---|---|---|---|")
for l in labels:
v, cap = S[l]["vram"], S[l]["capacity"]
def mx(lab):
vals = []
for r, c in cap.items():
ok = [n for n, s in c.get(lab, []) if s == 200]
bad = [(n, s) for n, s in c.get(lab, []) if s != 200]
vals.append(f"{max(ok) if ok else 0}" + (f" (next {bad[0][0]}: {bad[0][1]})" if bad else " (no failure up to the last step)"))
return "; ".join(vals) or "–"
w(f"| {NAMES.get(l, l)} | {cell(v['rest_mib'])} | {cell(v['peak_work_mib'])} | {cell(v['peak_capacity_mib'])} | {mx('short')} | {mx('long')} |")
print("\n".join(out))