docs: Jev candidate bench vs SemIf (fv-ml1 GPU 3) — Intern-Decision-4B is the replacement if SemIf is displaced
Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231, hard 0.613) reproduced exactly; negative control and a 4-restart noise floor (0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with- rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench rank does not transfer. Raw per-item data kept out of git.
This commit is contained in:
@@ -0,0 +1,303 @@
|
||||
"""Jev-candidate bench (2026-09-30): the replaced-baseline sets through one backend.
|
||||
|
||||
Stdlib only (urllib), so it runs on fv-ml1's host python against a loopback server.
|
||||
|
||||
Backends
|
||||
semif semif-serve (SemIf's own scorer and prompt). /decide; order averaging uses the
|
||||
service's own `"orderings": "rotations"` (what a switch through the contract gets).
|
||||
systemone a TypeSafe-style POST /v1/systemone server (a candidate's NATIVE runtime and prompt).
|
||||
Each row is one `choice` question, criteria {option id: description} in the row's
|
||||
order. Order averaging is done here: the n cyclic rotations are n separate requests,
|
||||
combined exactly as semif-serve does (per-ordering log p, mean per option id,
|
||||
renormalised; agreement = share of orderings whose top equals the combined top).
|
||||
|
||||
Sets (all labelled rows; see ../README in the results doc)
|
||||
authored144, perturbations108 SemIf's own labelled sets (evidence interpretation, 3 options)
|
||||
cicada-w1, cicada-w2 the 2026-09-27 Cicada affect-gate spike, both wordings, 35 cases
|
||||
wyrd the 2026-09-27 Wyrd scene-change spike, 21 cases x 4 decisions
|
||||
Spike rows are also asked over a content-free state (the null control, cond=blind).
|
||||
|
||||
Conditions (one process lifetime = one repeat r)
|
||||
single every row, caller's order
|
||||
rotations every row, n cyclic rotations averaged
|
||||
repeat authored144 again, caller's order (A-vs-A inside the process)
|
||||
reversed authored144, options reversed (order sensitivity)
|
||||
shuffled authored144, options shuffled, fixed seed (order sensitivity)
|
||||
negative authored144, descriptions rotated one place, ids fixed (NEGATIVE CONTROL)
|
||||
|
||||
python3 bench_sets.py --backend semif --url http://127.0.0.1:18032 --token-file T --out out.json
|
||||
python3 bench_sets.py --backend systemone --url http://127.0.0.1:18090 --out out.json
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
import random
|
||||
import sys
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
DATA = Path(__file__).resolve().parent / "data"
|
||||
BLIND = "(No evidence is available for this turn.)"
|
||||
|
||||
|
||||
def load_items() -> list[dict]:
|
||||
items = []
|
||||
for name in ("authored144", "perturbations108"):
|
||||
for line in (DATA / f"{name}.jsonl").read_text().splitlines():
|
||||
if not line.strip():
|
||||
continue
|
||||
r = json.loads(line)
|
||||
items.append({"set": name, "id": r["id"], "cond": "evidence", "state": r["state"],
|
||||
"question": r["question"],
|
||||
"options": [{"id": o["id"], "description": o["description"]} for o in r["options"]],
|
||||
"gold": [r["options"][r["label"]]["id"]], "group": r["group_id"], "tag": "case",
|
||||
"decision": "evidence", "forbid": []})
|
||||
for f in sorted((DATA / "spike").glob("*.json")):
|
||||
scen = json.loads(f.read_text())
|
||||
name = scen["name"]
|
||||
setname = ("cicada-w1" if "-w1-" in name else "cicada-w2") if name.startswith("cicada") else "wyrd"
|
||||
for case in scen["cases"]:
|
||||
for d in scen["decisions"]:
|
||||
lab = case.get("labels", {}).get(d["id"])
|
||||
gold = None if lab is None else ([lab] if isinstance(lab, str) else list(lab))
|
||||
base = {"set": setname, "decision": d["id"], "question": d["question"],
|
||||
"options": d["options"], "gold": gold, "group": f"{name}/{case['id']}",
|
||||
"tag": case.get("tag", "case"), "forbid": case.get("forbid", {}).get(d["id"], [])}
|
||||
items.append({**base, "id": f"{name}/{case['id']}/{d['id']}", "cond": "evidence",
|
||||
"state": case["state"]})
|
||||
items.append({**base, "id": f"{name}/{case['id']}/{d['id']}#blind", "cond": "blind",
|
||||
"state": BLIND})
|
||||
return items
|
||||
|
||||
|
||||
def post(url: str, body: dict, headers: dict, timeout: float = 300) -> tuple[int, dict, float]:
|
||||
data = json.dumps(body).encode()
|
||||
req = urllib.request.Request(url, data=data, headers={"Content-Type": "application/json", **headers})
|
||||
t = time.perf_counter()
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
status, raw = resp.status, resp.read()
|
||||
except urllib.error.HTTPError as e:
|
||||
status, raw = e.code, e.read()
|
||||
ms = (time.perf_counter() - t) * 1000
|
||||
try:
|
||||
parsed = json.loads(raw)
|
||||
except ValueError:
|
||||
parsed = {"raw": raw[:500].decode(errors="replace")}
|
||||
return status, parsed, ms
|
||||
|
||||
|
||||
def combine(per_ordering: list[dict[str, float]], ids: list[str]) -> dict:
|
||||
"""semif-serve's method: mean of log p per option id, renormalised."""
|
||||
logs = {i: sum(math.log(max(p[i], 1e-300)) for p in per_ordering) / len(per_ordering) for i in ids}
|
||||
m = max(logs.values())
|
||||
w = {i: math.exp(v - m) for i, v in logs.items()}
|
||||
z = sum(w.values())
|
||||
probs = {i: w[i] / z for i in ids}
|
||||
top = max(ids, key=lambda i: probs[i])
|
||||
tops = [max(ids, key=lambda i: p[i]) for p in per_ordering]
|
||||
return {"probabilities": [probs[i] for i in ids], "top": top,
|
||||
"agreement": sum(t == top for t in tops) / len(tops), "orderings": len(per_ordering)}
|
||||
|
||||
|
||||
class Semif:
|
||||
def __init__(self, url: str, token: str):
|
||||
self.url, self.h = url.rstrip("/"), {"Authorization": f"Bearer {token}"}
|
||||
|
||||
def score(self, item: dict, options: list[dict], averaged: bool) -> dict:
|
||||
body = {"id": item["id"], "state": item["state"], "question": item["question"], "options": options}
|
||||
if averaged:
|
||||
body["orderings"] = "rotations"
|
||||
status, j, ms = post(f"{self.url}/decide", body, self.h)
|
||||
if status != 200:
|
||||
return {"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms}
|
||||
ids = [o["id"] for o in options]
|
||||
if averaged:
|
||||
c = j["combined"]
|
||||
probs = dict(zip(j["option_ids"], c["probabilities"]))
|
||||
return {"ok": True, "e2e_ms": ms, "probs": [probs[i] for i in ids], "top": c["top"],
|
||||
"agreement": c["agreement"],
|
||||
"input_tokens": j["orderings"][0]["input_tokens"]}
|
||||
probs = dict(zip(j["option_ids"], j["probabilities"]))
|
||||
return {"ok": True, "e2e_ms": ms, "probs": [probs[i] for i in ids],
|
||||
"top": max(ids, key=lambda i: probs[i]), "option_logits": j.get("option_logits"),
|
||||
"prompt_sha256": j.get("prompt_sha256"), "input_tokens": j.get("input_tokens"),
|
||||
"server_ms": round(j.get("total_seconds", 0) * 1000, 2)}
|
||||
|
||||
def score_many(self, items: list[dict]) -> list[dict]:
|
||||
"""All decisions over one state in ONE /decide/shared (SemIf's shared-prefix path)."""
|
||||
body = {"state": items[0]["state"], "decisions": [
|
||||
{"id": i["id"], "question": i["question"], "options": i["options"]} for i in items]}
|
||||
status, j, ms = post(f"{self.url}/decide/shared", body, self.h)
|
||||
if status != 200:
|
||||
return [{"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms} for _ in items]
|
||||
out = []
|
||||
for item, res in zip(items, j["results"]):
|
||||
p = dict(zip(res["option_ids"], res["probabilities"]))
|
||||
ids = [o["id"] for o in item["options"]]
|
||||
out.append({"ok": True, "e2e_ms": ms, "probs": [p[i] for i in ids], "top": max(ids, key=lambda i: p[i])})
|
||||
return out
|
||||
|
||||
|
||||
class SystemOne:
|
||||
def __init__(self, url: str):
|
||||
self.url = url.rstrip("/")
|
||||
|
||||
def _one(self, item: dict, options: list[dict]) -> dict:
|
||||
body = {"state": item["state"], "questions": {"q": {
|
||||
"type": "choice", "instructions": item["question"],
|
||||
"criteria": {o["id"]: o["description"] for o in options}}}}
|
||||
status, j, ms = post(f"{self.url}/v1/systemone", body, {})
|
||||
if status != 200:
|
||||
return {"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms}
|
||||
a = j["answers"]["q"]
|
||||
p = a["probabilities"]
|
||||
missing = [o["id"] for o in options if o["id"] not in p]
|
||||
if missing:
|
||||
return {"ok": False, "status": status, "error": f"missing ids {missing}", "e2e_ms": ms}
|
||||
out = {"ok": True, "e2e_ms": ms, "p": {o["id"]: float(p[o["id"]]) for o in options},
|
||||
"input_tokens": (j.get("usage") or {}).get("input_tokens")}
|
||||
for k in ("unknown_probability", "abstained"):
|
||||
if k in a:
|
||||
out[k] = a[k]
|
||||
return out
|
||||
|
||||
def score_many(self, items: list[dict]) -> list[dict]:
|
||||
"""All decisions over one state as ONE request with several questions (the runtime's own
|
||||
multi-question mode; Intern-Decision scores them in one forward over one prompt)."""
|
||||
body = {"state": items[0]["state"], "questions": {
|
||||
it["decision"]: {"type": "choice", "instructions": it["question"],
|
||||
"criteria": {o["id"]: o["description"] for o in it["options"]}} for it in items}}
|
||||
status, j, ms = post(f"{self.url}/v1/systemone", body, {})
|
||||
if status != 200:
|
||||
return [{"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms} for _ in items]
|
||||
out = []
|
||||
for it in items:
|
||||
p = j["answers"][it["decision"]]["probabilities"]
|
||||
ids = [o["id"] for o in it["options"]]
|
||||
out.append({"ok": True, "e2e_ms": ms, "probs": [float(p[i]) for i in ids],
|
||||
"top": max(ids, key=lambda i: float(p[i]))})
|
||||
return out
|
||||
|
||||
def score(self, item: dict, options: list[dict], averaged: bool) -> dict:
|
||||
ids = [o["id"] for o in options]
|
||||
if not averaged:
|
||||
r = self._one(item, options)
|
||||
if not r["ok"]:
|
||||
return r
|
||||
p = r.pop("p")
|
||||
return {**r, "probs": [p[i] for i in ids], "top": max(ids, key=lambda i: p[i])}
|
||||
per, ms, extra = [], 0.0, {}
|
||||
for k in range(len(options)):
|
||||
r = self._one(item, options[k:] + options[:k])
|
||||
if not r["ok"]:
|
||||
return r
|
||||
per.append(r["p"])
|
||||
ms += r["e2e_ms"]
|
||||
if "abstained" in r:
|
||||
extra.setdefault("abstained_orderings", 0)
|
||||
extra["abstained_orderings"] += bool(r["abstained"])
|
||||
c = combine(per, ids)
|
||||
return {"ok": True, "e2e_ms": ms, "probs": c["probabilities"], "top": c["top"],
|
||||
"agreement": c["agreement"], **extra}
|
||||
|
||||
|
||||
def shuffled(options: list[dict], key: str) -> list[dict]:
|
||||
rng = random.Random(f"jev-bench-2026-09-30/{key}")
|
||||
out = list(options)
|
||||
for _ in range(10):
|
||||
rng.shuffle(out)
|
||||
if [o["id"] for o in out] != [o["id"] for o in options]:
|
||||
break
|
||||
return out
|
||||
|
||||
|
||||
def rotated_descriptions(options: list[dict]) -> list[dict]:
|
||||
descs = [o["description"] for o in options]
|
||||
descs = descs[1:] + descs[:1]
|
||||
return [{**o, "description": d} for o, d in zip(options, descs)]
|
||||
|
||||
|
||||
def gold_desc_id(item: dict) -> str | None:
|
||||
"""After rotated_descriptions, position j carries the description that was at j+1, so the gold
|
||||
description (position g) now sits at g-1."""
|
||||
if not item["gold"]:
|
||||
return None
|
||||
ids = [o["id"] for o in item["options"]]
|
||||
g = ids.index(item["gold"][0])
|
||||
return ids[(g - 1) % len(ids)]
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--backend", choices=("semif", "systemone"), required=True)
|
||||
ap.add_argument("--url", required=True)
|
||||
ap.add_argument("--token-file")
|
||||
ap.add_argument("--label", required=True)
|
||||
ap.add_argument("--out", required=True)
|
||||
ap.add_argument("--conditions", default="single,rotations,repeat,reversed,shuffled,negative,multifield")
|
||||
args = ap.parse_args()
|
||||
backend = (Semif(args.url, Path(args.token_file).read_text().strip()) if args.backend == "semif"
|
||||
else SystemOne(args.url))
|
||||
items = load_items()
|
||||
a144 = [i for i in items if i["set"] == "authored144"]
|
||||
plan = {
|
||||
"single": [(i, i["options"], False) for i in items],
|
||||
"rotations": [(i, i["options"], True) for i in items],
|
||||
"repeat": [(i, i["options"], False) for i in a144],
|
||||
"reversed": [(i, list(reversed(i["options"])), False) for i in a144],
|
||||
"shuffled": [(i, shuffled(i["options"], i["id"]), False) for i in a144],
|
||||
"negative": [(i, rotated_descriptions(i["options"]), False) for i in a144],
|
||||
}
|
||||
report = {"label": args.label, "backend": args.backend, "url": args.url,
|
||||
"started_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), "conditions": {}}
|
||||
for cond in args.conditions.split(","):
|
||||
if cond == "multifield":
|
||||
# Wyrd's real shape: the 4 decisions of one turn over one state, in one request
|
||||
groups: dict[str, list[dict]] = {}
|
||||
for it in items:
|
||||
if it["set"] == "wyrd" and it["cond"] == "evidence":
|
||||
groups.setdefault(it["group"], []).append(it)
|
||||
rows, t0 = [], time.perf_counter()
|
||||
for g, its in groups.items():
|
||||
for it, r in zip(its, backend.score_many(its)):
|
||||
rows.append({"id": it["id"], "set": it["set"], "cond": it["cond"], "decision": it["decision"],
|
||||
"tag": it["tag"], "group": it["group"], "gold": it["gold"], "forbid": it["forbid"],
|
||||
"option_ids": [o["id"] for o in it["options"]], "gold_desc_id": None, **r})
|
||||
fails = sum(not r["ok"] for r in rows)
|
||||
report["conditions"][cond] = {"rows": rows, "seconds": round(time.perf_counter() - t0, 1), "failures": fails}
|
||||
ok = [x for x in rows if x["ok"] and x["gold"]]
|
||||
print(f"[{args.label}] {cond}: {len(rows)} rows in {len(groups)} requests, {fails} failed, "
|
||||
f"acc {sum(x['top'] in x['gold'] for x in ok) / max(1, len(ok)):.3f}", flush=True)
|
||||
continue
|
||||
rows, t0, fails = [], time.perf_counter(), 0
|
||||
for item, options, averaged in plan[cond]:
|
||||
r = backend.score(item, options, averaged)
|
||||
fails += not r["ok"]
|
||||
rows.append({"id": item["id"], "set": item["set"], "cond": item["cond"],
|
||||
"decision": item["decision"], "tag": item["tag"], "group": item["group"],
|
||||
"gold": item["gold"], "forbid": item["forbid"],
|
||||
"option_ids": [o["id"] for o in options],
|
||||
# negative control: the id that now carries the gold option's description
|
||||
"gold_desc_id": gold_desc_id(item) if cond == "negative" else None,
|
||||
**r})
|
||||
if fails >= 5 and fails == len(rows):
|
||||
print(f"[{args.label}] {cond}: first {fails} rows all failed: {r}", file=sys.stderr)
|
||||
return 2
|
||||
report["conditions"][cond] = {"rows": rows, "seconds": round(time.perf_counter() - t0, 1),
|
||||
"failures": fails}
|
||||
ok = [x for x in rows if x["ok"] and x["gold"] and x["cond"] == "evidence"]
|
||||
acc = sum(x["top"] in x["gold"] for x in ok) / len(ok) if ok else float("nan")
|
||||
print(f"[{args.label}] {cond}: {len(rows)} rows, {fails} failed, "
|
||||
f"{report['conditions'][cond]['seconds']}s, labelled-evidence acc {acc:.3f}", flush=True)
|
||||
report["finished_utc"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
||||
Path(args.out).write_text(json.dumps(report))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user