Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231, hard 0.613) reproduced exactly; negative control and a 4-restart noise floor (0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with- rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench rank does not transfer. Raw per-item data kept out of git.
304 lines
15 KiB
Python
304 lines
15 KiB
Python
"""Jev-candidate bench (2026-09-30): the replaced-baseline sets through one backend.
|
|
|
|
Stdlib only (urllib), so it runs on fv-ml1's host python against a loopback server.
|
|
|
|
Backends
|
|
semif semif-serve (SemIf's own scorer and prompt). /decide; order averaging uses the
|
|
service's own `"orderings": "rotations"` (what a switch through the contract gets).
|
|
systemone a TypeSafe-style POST /v1/systemone server (a candidate's NATIVE runtime and prompt).
|
|
Each row is one `choice` question, criteria {option id: description} in the row's
|
|
order. Order averaging is done here: the n cyclic rotations are n separate requests,
|
|
combined exactly as semif-serve does (per-ordering log p, mean per option id,
|
|
renormalised; agreement = share of orderings whose top equals the combined top).
|
|
|
|
Sets (all labelled rows; see ../README in the results doc)
|
|
authored144, perturbations108 SemIf's own labelled sets (evidence interpretation, 3 options)
|
|
cicada-w1, cicada-w2 the 2026-09-27 Cicada affect-gate spike, both wordings, 35 cases
|
|
wyrd the 2026-09-27 Wyrd scene-change spike, 21 cases x 4 decisions
|
|
Spike rows are also asked over a content-free state (the null control, cond=blind).
|
|
|
|
Conditions (one process lifetime = one repeat r)
|
|
single every row, caller's order
|
|
rotations every row, n cyclic rotations averaged
|
|
repeat authored144 again, caller's order (A-vs-A inside the process)
|
|
reversed authored144, options reversed (order sensitivity)
|
|
shuffled authored144, options shuffled, fixed seed (order sensitivity)
|
|
negative authored144, descriptions rotated one place, ids fixed (NEGATIVE CONTROL)
|
|
|
|
python3 bench_sets.py --backend semif --url http://127.0.0.1:18032 --token-file T --out out.json
|
|
python3 bench_sets.py --backend systemone --url http://127.0.0.1:18090 --out out.json
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import math
|
|
import random
|
|
import sys
|
|
import time
|
|
import urllib.error
|
|
import urllib.request
|
|
from pathlib import Path
|
|
|
|
DATA = Path(__file__).resolve().parent / "data"
|
|
BLIND = "(No evidence is available for this turn.)"
|
|
|
|
|
|
def load_items() -> list[dict]:
|
|
items = []
|
|
for name in ("authored144", "perturbations108"):
|
|
for line in (DATA / f"{name}.jsonl").read_text().splitlines():
|
|
if not line.strip():
|
|
continue
|
|
r = json.loads(line)
|
|
items.append({"set": name, "id": r["id"], "cond": "evidence", "state": r["state"],
|
|
"question": r["question"],
|
|
"options": [{"id": o["id"], "description": o["description"]} for o in r["options"]],
|
|
"gold": [r["options"][r["label"]]["id"]], "group": r["group_id"], "tag": "case",
|
|
"decision": "evidence", "forbid": []})
|
|
for f in sorted((DATA / "spike").glob("*.json")):
|
|
scen = json.loads(f.read_text())
|
|
name = scen["name"]
|
|
setname = ("cicada-w1" if "-w1-" in name else "cicada-w2") if name.startswith("cicada") else "wyrd"
|
|
for case in scen["cases"]:
|
|
for d in scen["decisions"]:
|
|
lab = case.get("labels", {}).get(d["id"])
|
|
gold = None if lab is None else ([lab] if isinstance(lab, str) else list(lab))
|
|
base = {"set": setname, "decision": d["id"], "question": d["question"],
|
|
"options": d["options"], "gold": gold, "group": f"{name}/{case['id']}",
|
|
"tag": case.get("tag", "case"), "forbid": case.get("forbid", {}).get(d["id"], [])}
|
|
items.append({**base, "id": f"{name}/{case['id']}/{d['id']}", "cond": "evidence",
|
|
"state": case["state"]})
|
|
items.append({**base, "id": f"{name}/{case['id']}/{d['id']}#blind", "cond": "blind",
|
|
"state": BLIND})
|
|
return items
|
|
|
|
|
|
def post(url: str, body: dict, headers: dict, timeout: float = 300) -> tuple[int, dict, float]:
|
|
data = json.dumps(body).encode()
|
|
req = urllib.request.Request(url, data=data, headers={"Content-Type": "application/json", **headers})
|
|
t = time.perf_counter()
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
status, raw = resp.status, resp.read()
|
|
except urllib.error.HTTPError as e:
|
|
status, raw = e.code, e.read()
|
|
ms = (time.perf_counter() - t) * 1000
|
|
try:
|
|
parsed = json.loads(raw)
|
|
except ValueError:
|
|
parsed = {"raw": raw[:500].decode(errors="replace")}
|
|
return status, parsed, ms
|
|
|
|
|
|
def combine(per_ordering: list[dict[str, float]], ids: list[str]) -> dict:
|
|
"""semif-serve's method: mean of log p per option id, renormalised."""
|
|
logs = {i: sum(math.log(max(p[i], 1e-300)) for p in per_ordering) / len(per_ordering) for i in ids}
|
|
m = max(logs.values())
|
|
w = {i: math.exp(v - m) for i, v in logs.items()}
|
|
z = sum(w.values())
|
|
probs = {i: w[i] / z for i in ids}
|
|
top = max(ids, key=lambda i: probs[i])
|
|
tops = [max(ids, key=lambda i: p[i]) for p in per_ordering]
|
|
return {"probabilities": [probs[i] for i in ids], "top": top,
|
|
"agreement": sum(t == top for t in tops) / len(tops), "orderings": len(per_ordering)}
|
|
|
|
|
|
class Semif:
|
|
def __init__(self, url: str, token: str):
|
|
self.url, self.h = url.rstrip("/"), {"Authorization": f"Bearer {token}"}
|
|
|
|
def score(self, item: dict, options: list[dict], averaged: bool) -> dict:
|
|
body = {"id": item["id"], "state": item["state"], "question": item["question"], "options": options}
|
|
if averaged:
|
|
body["orderings"] = "rotations"
|
|
status, j, ms = post(f"{self.url}/decide", body, self.h)
|
|
if status != 200:
|
|
return {"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms}
|
|
ids = [o["id"] for o in options]
|
|
if averaged:
|
|
c = j["combined"]
|
|
probs = dict(zip(j["option_ids"], c["probabilities"]))
|
|
return {"ok": True, "e2e_ms": ms, "probs": [probs[i] for i in ids], "top": c["top"],
|
|
"agreement": c["agreement"],
|
|
"input_tokens": j["orderings"][0]["input_tokens"]}
|
|
probs = dict(zip(j["option_ids"], j["probabilities"]))
|
|
return {"ok": True, "e2e_ms": ms, "probs": [probs[i] for i in ids],
|
|
"top": max(ids, key=lambda i: probs[i]), "option_logits": j.get("option_logits"),
|
|
"prompt_sha256": j.get("prompt_sha256"), "input_tokens": j.get("input_tokens"),
|
|
"server_ms": round(j.get("total_seconds", 0) * 1000, 2)}
|
|
|
|
def score_many(self, items: list[dict]) -> list[dict]:
|
|
"""All decisions over one state in ONE /decide/shared (SemIf's shared-prefix path)."""
|
|
body = {"state": items[0]["state"], "decisions": [
|
|
{"id": i["id"], "question": i["question"], "options": i["options"]} for i in items]}
|
|
status, j, ms = post(f"{self.url}/decide/shared", body, self.h)
|
|
if status != 200:
|
|
return [{"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms} for _ in items]
|
|
out = []
|
|
for item, res in zip(items, j["results"]):
|
|
p = dict(zip(res["option_ids"], res["probabilities"]))
|
|
ids = [o["id"] for o in item["options"]]
|
|
out.append({"ok": True, "e2e_ms": ms, "probs": [p[i] for i in ids], "top": max(ids, key=lambda i: p[i])})
|
|
return out
|
|
|
|
|
|
class SystemOne:
|
|
def __init__(self, url: str):
|
|
self.url = url.rstrip("/")
|
|
|
|
def _one(self, item: dict, options: list[dict]) -> dict:
|
|
body = {"state": item["state"], "questions": {"q": {
|
|
"type": "choice", "instructions": item["question"],
|
|
"criteria": {o["id"]: o["description"] for o in options}}}}
|
|
status, j, ms = post(f"{self.url}/v1/systemone", body, {})
|
|
if status != 200:
|
|
return {"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms}
|
|
a = j["answers"]["q"]
|
|
p = a["probabilities"]
|
|
missing = [o["id"] for o in options if o["id"] not in p]
|
|
if missing:
|
|
return {"ok": False, "status": status, "error": f"missing ids {missing}", "e2e_ms": ms}
|
|
out = {"ok": True, "e2e_ms": ms, "p": {o["id"]: float(p[o["id"]]) for o in options},
|
|
"input_tokens": (j.get("usage") or {}).get("input_tokens")}
|
|
for k in ("unknown_probability", "abstained"):
|
|
if k in a:
|
|
out[k] = a[k]
|
|
return out
|
|
|
|
def score_many(self, items: list[dict]) -> list[dict]:
|
|
"""All decisions over one state as ONE request with several questions (the runtime's own
|
|
multi-question mode; Intern-Decision scores them in one forward over one prompt)."""
|
|
body = {"state": items[0]["state"], "questions": {
|
|
it["decision"]: {"type": "choice", "instructions": it["question"],
|
|
"criteria": {o["id"]: o["description"] for o in it["options"]}} for it in items}}
|
|
status, j, ms = post(f"{self.url}/v1/systemone", body, {})
|
|
if status != 200:
|
|
return [{"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms} for _ in items]
|
|
out = []
|
|
for it in items:
|
|
p = j["answers"][it["decision"]]["probabilities"]
|
|
ids = [o["id"] for o in it["options"]]
|
|
out.append({"ok": True, "e2e_ms": ms, "probs": [float(p[i]) for i in ids],
|
|
"top": max(ids, key=lambda i: float(p[i]))})
|
|
return out
|
|
|
|
def score(self, item: dict, options: list[dict], averaged: bool) -> dict:
|
|
ids = [o["id"] for o in options]
|
|
if not averaged:
|
|
r = self._one(item, options)
|
|
if not r["ok"]:
|
|
return r
|
|
p = r.pop("p")
|
|
return {**r, "probs": [p[i] for i in ids], "top": max(ids, key=lambda i: p[i])}
|
|
per, ms, extra = [], 0.0, {}
|
|
for k in range(len(options)):
|
|
r = self._one(item, options[k:] + options[:k])
|
|
if not r["ok"]:
|
|
return r
|
|
per.append(r["p"])
|
|
ms += r["e2e_ms"]
|
|
if "abstained" in r:
|
|
extra.setdefault("abstained_orderings", 0)
|
|
extra["abstained_orderings"] += bool(r["abstained"])
|
|
c = combine(per, ids)
|
|
return {"ok": True, "e2e_ms": ms, "probs": c["probabilities"], "top": c["top"],
|
|
"agreement": c["agreement"], **extra}
|
|
|
|
|
|
def shuffled(options: list[dict], key: str) -> list[dict]:
|
|
rng = random.Random(f"jev-bench-2026-09-30/{key}")
|
|
out = list(options)
|
|
for _ in range(10):
|
|
rng.shuffle(out)
|
|
if [o["id"] for o in out] != [o["id"] for o in options]:
|
|
break
|
|
return out
|
|
|
|
|
|
def rotated_descriptions(options: list[dict]) -> list[dict]:
|
|
descs = [o["description"] for o in options]
|
|
descs = descs[1:] + descs[:1]
|
|
return [{**o, "description": d} for o, d in zip(options, descs)]
|
|
|
|
|
|
def gold_desc_id(item: dict) -> str | None:
|
|
"""After rotated_descriptions, position j carries the description that was at j+1, so the gold
|
|
description (position g) now sits at g-1."""
|
|
if not item["gold"]:
|
|
return None
|
|
ids = [o["id"] for o in item["options"]]
|
|
g = ids.index(item["gold"][0])
|
|
return ids[(g - 1) % len(ids)]
|
|
|
|
|
|
def main() -> int:
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--backend", choices=("semif", "systemone"), required=True)
|
|
ap.add_argument("--url", required=True)
|
|
ap.add_argument("--token-file")
|
|
ap.add_argument("--label", required=True)
|
|
ap.add_argument("--out", required=True)
|
|
ap.add_argument("--conditions", default="single,rotations,repeat,reversed,shuffled,negative,multifield")
|
|
args = ap.parse_args()
|
|
backend = (Semif(args.url, Path(args.token_file).read_text().strip()) if args.backend == "semif"
|
|
else SystemOne(args.url))
|
|
items = load_items()
|
|
a144 = [i for i in items if i["set"] == "authored144"]
|
|
plan = {
|
|
"single": [(i, i["options"], False) for i in items],
|
|
"rotations": [(i, i["options"], True) for i in items],
|
|
"repeat": [(i, i["options"], False) for i in a144],
|
|
"reversed": [(i, list(reversed(i["options"])), False) for i in a144],
|
|
"shuffled": [(i, shuffled(i["options"], i["id"]), False) for i in a144],
|
|
"negative": [(i, rotated_descriptions(i["options"]), False) for i in a144],
|
|
}
|
|
report = {"label": args.label, "backend": args.backend, "url": args.url,
|
|
"started_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), "conditions": {}}
|
|
for cond in args.conditions.split(","):
|
|
if cond == "multifield":
|
|
# Wyrd's real shape: the 4 decisions of one turn over one state, in one request
|
|
groups: dict[str, list[dict]] = {}
|
|
for it in items:
|
|
if it["set"] == "wyrd" and it["cond"] == "evidence":
|
|
groups.setdefault(it["group"], []).append(it)
|
|
rows, t0 = [], time.perf_counter()
|
|
for g, its in groups.items():
|
|
for it, r in zip(its, backend.score_many(its)):
|
|
rows.append({"id": it["id"], "set": it["set"], "cond": it["cond"], "decision": it["decision"],
|
|
"tag": it["tag"], "group": it["group"], "gold": it["gold"], "forbid": it["forbid"],
|
|
"option_ids": [o["id"] for o in it["options"]], "gold_desc_id": None, **r})
|
|
fails = sum(not r["ok"] for r in rows)
|
|
report["conditions"][cond] = {"rows": rows, "seconds": round(time.perf_counter() - t0, 1), "failures": fails}
|
|
ok = [x for x in rows if x["ok"] and x["gold"]]
|
|
print(f"[{args.label}] {cond}: {len(rows)} rows in {len(groups)} requests, {fails} failed, "
|
|
f"acc {sum(x['top'] in x['gold'] for x in ok) / max(1, len(ok)):.3f}", flush=True)
|
|
continue
|
|
rows, t0, fails = [], time.perf_counter(), 0
|
|
for item, options, averaged in plan[cond]:
|
|
r = backend.score(item, options, averaged)
|
|
fails += not r["ok"]
|
|
rows.append({"id": item["id"], "set": item["set"], "cond": item["cond"],
|
|
"decision": item["decision"], "tag": item["tag"], "group": item["group"],
|
|
"gold": item["gold"], "forbid": item["forbid"],
|
|
"option_ids": [o["id"] for o in options],
|
|
# negative control: the id that now carries the gold option's description
|
|
"gold_desc_id": gold_desc_id(item) if cond == "negative" else None,
|
|
**r})
|
|
if fails >= 5 and fails == len(rows):
|
|
print(f"[{args.label}] {cond}: first {fails} rows all failed: {r}", file=sys.stderr)
|
|
return 2
|
|
report["conditions"][cond] = {"rows": rows, "seconds": round(time.perf_counter() - t0, 1),
|
|
"failures": fails}
|
|
ok = [x for x in rows if x["ok"] and x["gold"] and x["cond"] == "evidence"]
|
|
acc = sum(x["top"] in x["gold"] for x in ok) / len(ok) if ok else float("nan")
|
|
print(f"[{args.label}] {cond}: {len(rows)} rows, {fails} failed, "
|
|
f"{report['conditions'][cond]['seconds']}s, labelled-evidence acc {acc:.3f}", flush=True)
|
|
report["finished_utc"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
Path(args.out).write_text(json.dumps(report))
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|