Files
esh-pfi-infrastructure/services/semif-serve/bench-jev-2026-09-30/code/bench_sets.py
T
vh 475d6d6bcb docs: Jev candidate bench vs SemIf (fv-ml1 GPU 3) — Intern-Decision-4B is the replacement if SemIf is displaced
Operator ask relayed by brokkr-smithy-dev. Positive control (SemIf 187/231,
hard 0.613) reproduced exactly; negative control and a 4-restart noise floor
(0 flips) measured. On our replaced-baseline sets no candidate beats SemIf-with-
rotations beyond the ~4-pt floor; Intern-Decision-4B native matches it at one
ordering, is better on Wyrd, fits 9.7/10.3 GB and is 1.5-2.3x faster. JevBench
rank does not transfer. Raw per-item data kept out of git.
2026-09-30 05:00:45 -07:00

304 lines
15 KiB
Python

"""Jev-candidate bench (2026-09-30): the replaced-baseline sets through one backend.
Stdlib only (urllib), so it runs on fv-ml1's host python against a loopback server.
Backends
semif semif-serve (SemIf's own scorer and prompt). /decide; order averaging uses the
service's own `"orderings": "rotations"` (what a switch through the contract gets).
systemone a TypeSafe-style POST /v1/systemone server (a candidate's NATIVE runtime and prompt).
Each row is one `choice` question, criteria {option id: description} in the row's
order. Order averaging is done here: the n cyclic rotations are n separate requests,
combined exactly as semif-serve does (per-ordering log p, mean per option id,
renormalised; agreement = share of orderings whose top equals the combined top).
Sets (all labelled rows; see ../README in the results doc)
authored144, perturbations108 SemIf's own labelled sets (evidence interpretation, 3 options)
cicada-w1, cicada-w2 the 2026-09-27 Cicada affect-gate spike, both wordings, 35 cases
wyrd the 2026-09-27 Wyrd scene-change spike, 21 cases x 4 decisions
Spike rows are also asked over a content-free state (the null control, cond=blind).
Conditions (one process lifetime = one repeat r)
single every row, caller's order
rotations every row, n cyclic rotations averaged
repeat authored144 again, caller's order (A-vs-A inside the process)
reversed authored144, options reversed (order sensitivity)
shuffled authored144, options shuffled, fixed seed (order sensitivity)
negative authored144, descriptions rotated one place, ids fixed (NEGATIVE CONTROL)
python3 bench_sets.py --backend semif --url http://127.0.0.1:18032 --token-file T --out out.json
python3 bench_sets.py --backend systemone --url http://127.0.0.1:18090 --out out.json
"""
from __future__ import annotations
import argparse
import json
import math
import random
import sys
import time
import urllib.error
import urllib.request
from pathlib import Path
DATA = Path(__file__).resolve().parent / "data"
BLIND = "(No evidence is available for this turn.)"
def load_items() -> list[dict]:
items = []
for name in ("authored144", "perturbations108"):
for line in (DATA / f"{name}.jsonl").read_text().splitlines():
if not line.strip():
continue
r = json.loads(line)
items.append({"set": name, "id": r["id"], "cond": "evidence", "state": r["state"],
"question": r["question"],
"options": [{"id": o["id"], "description": o["description"]} for o in r["options"]],
"gold": [r["options"][r["label"]]["id"]], "group": r["group_id"], "tag": "case",
"decision": "evidence", "forbid": []})
for f in sorted((DATA / "spike").glob("*.json")):
scen = json.loads(f.read_text())
name = scen["name"]
setname = ("cicada-w1" if "-w1-" in name else "cicada-w2") if name.startswith("cicada") else "wyrd"
for case in scen["cases"]:
for d in scen["decisions"]:
lab = case.get("labels", {}).get(d["id"])
gold = None if lab is None else ([lab] if isinstance(lab, str) else list(lab))
base = {"set": setname, "decision": d["id"], "question": d["question"],
"options": d["options"], "gold": gold, "group": f"{name}/{case['id']}",
"tag": case.get("tag", "case"), "forbid": case.get("forbid", {}).get(d["id"], [])}
items.append({**base, "id": f"{name}/{case['id']}/{d['id']}", "cond": "evidence",
"state": case["state"]})
items.append({**base, "id": f"{name}/{case['id']}/{d['id']}#blind", "cond": "blind",
"state": BLIND})
return items
def post(url: str, body: dict, headers: dict, timeout: float = 300) -> tuple[int, dict, float]:
data = json.dumps(body).encode()
req = urllib.request.Request(url, data=data, headers={"Content-Type": "application/json", **headers})
t = time.perf_counter()
try:
with urllib.request.urlopen(req, timeout=timeout) as resp:
status, raw = resp.status, resp.read()
except urllib.error.HTTPError as e:
status, raw = e.code, e.read()
ms = (time.perf_counter() - t) * 1000
try:
parsed = json.loads(raw)
except ValueError:
parsed = {"raw": raw[:500].decode(errors="replace")}
return status, parsed, ms
def combine(per_ordering: list[dict[str, float]], ids: list[str]) -> dict:
"""semif-serve's method: mean of log p per option id, renormalised."""
logs = {i: sum(math.log(max(p[i], 1e-300)) for p in per_ordering) / len(per_ordering) for i in ids}
m = max(logs.values())
w = {i: math.exp(v - m) for i, v in logs.items()}
z = sum(w.values())
probs = {i: w[i] / z for i in ids}
top = max(ids, key=lambda i: probs[i])
tops = [max(ids, key=lambda i: p[i]) for p in per_ordering]
return {"probabilities": [probs[i] for i in ids], "top": top,
"agreement": sum(t == top for t in tops) / len(tops), "orderings": len(per_ordering)}
class Semif:
def __init__(self, url: str, token: str):
self.url, self.h = url.rstrip("/"), {"Authorization": f"Bearer {token}"}
def score(self, item: dict, options: list[dict], averaged: bool) -> dict:
body = {"id": item["id"], "state": item["state"], "question": item["question"], "options": options}
if averaged:
body["orderings"] = "rotations"
status, j, ms = post(f"{self.url}/decide", body, self.h)
if status != 200:
return {"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms}
ids = [o["id"] for o in options]
if averaged:
c = j["combined"]
probs = dict(zip(j["option_ids"], c["probabilities"]))
return {"ok": True, "e2e_ms": ms, "probs": [probs[i] for i in ids], "top": c["top"],
"agreement": c["agreement"],
"input_tokens": j["orderings"][0]["input_tokens"]}
probs = dict(zip(j["option_ids"], j["probabilities"]))
return {"ok": True, "e2e_ms": ms, "probs": [probs[i] for i in ids],
"top": max(ids, key=lambda i: probs[i]), "option_logits": j.get("option_logits"),
"prompt_sha256": j.get("prompt_sha256"), "input_tokens": j.get("input_tokens"),
"server_ms": round(j.get("total_seconds", 0) * 1000, 2)}
def score_many(self, items: list[dict]) -> list[dict]:
"""All decisions over one state in ONE /decide/shared (SemIf's shared-prefix path)."""
body = {"state": items[0]["state"], "decisions": [
{"id": i["id"], "question": i["question"], "options": i["options"]} for i in items]}
status, j, ms = post(f"{self.url}/decide/shared", body, self.h)
if status != 200:
return [{"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms} for _ in items]
out = []
for item, res in zip(items, j["results"]):
p = dict(zip(res["option_ids"], res["probabilities"]))
ids = [o["id"] for o in item["options"]]
out.append({"ok": True, "e2e_ms": ms, "probs": [p[i] for i in ids], "top": max(ids, key=lambda i: p[i])})
return out
class SystemOne:
def __init__(self, url: str):
self.url = url.rstrip("/")
def _one(self, item: dict, options: list[dict]) -> dict:
body = {"state": item["state"], "questions": {"q": {
"type": "choice", "instructions": item["question"],
"criteria": {o["id"]: o["description"] for o in options}}}}
status, j, ms = post(f"{self.url}/v1/systemone", body, {})
if status != 200:
return {"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms}
a = j["answers"]["q"]
p = a["probabilities"]
missing = [o["id"] for o in options if o["id"] not in p]
if missing:
return {"ok": False, "status": status, "error": f"missing ids {missing}", "e2e_ms": ms}
out = {"ok": True, "e2e_ms": ms, "p": {o["id"]: float(p[o["id"]]) for o in options},
"input_tokens": (j.get("usage") or {}).get("input_tokens")}
for k in ("unknown_probability", "abstained"):
if k in a:
out[k] = a[k]
return out
def score_many(self, items: list[dict]) -> list[dict]:
"""All decisions over one state as ONE request with several questions (the runtime's own
multi-question mode; Intern-Decision scores them in one forward over one prompt)."""
body = {"state": items[0]["state"], "questions": {
it["decision"]: {"type": "choice", "instructions": it["question"],
"criteria": {o["id"]: o["description"] for o in it["options"]}} for it in items}}
status, j, ms = post(f"{self.url}/v1/systemone", body, {})
if status != 200:
return [{"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms} for _ in items]
out = []
for it in items:
p = j["answers"][it["decision"]]["probabilities"]
ids = [o["id"] for o in it["options"]]
out.append({"ok": True, "e2e_ms": ms, "probs": [float(p[i]) for i in ids],
"top": max(ids, key=lambda i: float(p[i]))})
return out
def score(self, item: dict, options: list[dict], averaged: bool) -> dict:
ids = [o["id"] for o in options]
if not averaged:
r = self._one(item, options)
if not r["ok"]:
return r
p = r.pop("p")
return {**r, "probs": [p[i] for i in ids], "top": max(ids, key=lambda i: p[i])}
per, ms, extra = [], 0.0, {}
for k in range(len(options)):
r = self._one(item, options[k:] + options[:k])
if not r["ok"]:
return r
per.append(r["p"])
ms += r["e2e_ms"]
if "abstained" in r:
extra.setdefault("abstained_orderings", 0)
extra["abstained_orderings"] += bool(r["abstained"])
c = combine(per, ids)
return {"ok": True, "e2e_ms": ms, "probs": c["probabilities"], "top": c["top"],
"agreement": c["agreement"], **extra}
def shuffled(options: list[dict], key: str) -> list[dict]:
rng = random.Random(f"jev-bench-2026-09-30/{key}")
out = list(options)
for _ in range(10):
rng.shuffle(out)
if [o["id"] for o in out] != [o["id"] for o in options]:
break
return out
def rotated_descriptions(options: list[dict]) -> list[dict]:
descs = [o["description"] for o in options]
descs = descs[1:] + descs[:1]
return [{**o, "description": d} for o, d in zip(options, descs)]
def gold_desc_id(item: dict) -> str | None:
"""After rotated_descriptions, position j carries the description that was at j+1, so the gold
description (position g) now sits at g-1."""
if not item["gold"]:
return None
ids = [o["id"] for o in item["options"]]
g = ids.index(item["gold"][0])
return ids[(g - 1) % len(ids)]
def main() -> int:
ap = argparse.ArgumentParser()
ap.add_argument("--backend", choices=("semif", "systemone"), required=True)
ap.add_argument("--url", required=True)
ap.add_argument("--token-file")
ap.add_argument("--label", required=True)
ap.add_argument("--out", required=True)
ap.add_argument("--conditions", default="single,rotations,repeat,reversed,shuffled,negative,multifield")
args = ap.parse_args()
backend = (Semif(args.url, Path(args.token_file).read_text().strip()) if args.backend == "semif"
else SystemOne(args.url))
items = load_items()
a144 = [i for i in items if i["set"] == "authored144"]
plan = {
"single": [(i, i["options"], False) for i in items],
"rotations": [(i, i["options"], True) for i in items],
"repeat": [(i, i["options"], False) for i in a144],
"reversed": [(i, list(reversed(i["options"])), False) for i in a144],
"shuffled": [(i, shuffled(i["options"], i["id"]), False) for i in a144],
"negative": [(i, rotated_descriptions(i["options"]), False) for i in a144],
}
report = {"label": args.label, "backend": args.backend, "url": args.url,
"started_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), "conditions": {}}
for cond in args.conditions.split(","):
if cond == "multifield":
# Wyrd's real shape: the 4 decisions of one turn over one state, in one request
groups: dict[str, list[dict]] = {}
for it in items:
if it["set"] == "wyrd" and it["cond"] == "evidence":
groups.setdefault(it["group"], []).append(it)
rows, t0 = [], time.perf_counter()
for g, its in groups.items():
for it, r in zip(its, backend.score_many(its)):
rows.append({"id": it["id"], "set": it["set"], "cond": it["cond"], "decision": it["decision"],
"tag": it["tag"], "group": it["group"], "gold": it["gold"], "forbid": it["forbid"],
"option_ids": [o["id"] for o in it["options"]], "gold_desc_id": None, **r})
fails = sum(not r["ok"] for r in rows)
report["conditions"][cond] = {"rows": rows, "seconds": round(time.perf_counter() - t0, 1), "failures": fails}
ok = [x for x in rows if x["ok"] and x["gold"]]
print(f"[{args.label}] {cond}: {len(rows)} rows in {len(groups)} requests, {fails} failed, "
f"acc {sum(x['top'] in x['gold'] for x in ok) / max(1, len(ok)):.3f}", flush=True)
continue
rows, t0, fails = [], time.perf_counter(), 0
for item, options, averaged in plan[cond]:
r = backend.score(item, options, averaged)
fails += not r["ok"]
rows.append({"id": item["id"], "set": item["set"], "cond": item["cond"],
"decision": item["decision"], "tag": item["tag"], "group": item["group"],
"gold": item["gold"], "forbid": item["forbid"],
"option_ids": [o["id"] for o in options],
# negative control: the id that now carries the gold option's description
"gold_desc_id": gold_desc_id(item) if cond == "negative" else None,
**r})
if fails >= 5 and fails == len(rows):
print(f"[{args.label}] {cond}: first {fails} rows all failed: {r}", file=sys.stderr)
return 2
report["conditions"][cond] = {"rows": rows, "seconds": round(time.perf_counter() - t0, 1),
"failures": fails}
ok = [x for x in rows if x["ok"] and x["gold"] and x["cond"] == "evidence"]
acc = sum(x["top"] in x["gold"] for x in ok) / len(ok) if ok else float("nan")
print(f"[{args.label}] {cond}: {len(rows)} rows, {fails} failed, "
f"{report['conditions'][cond]['seconds']}s, labelled-evidence acc {acc:.3f}", flush=True)
report["finished_utc"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
Path(args.out).write_text(json.dumps(report))
return 0
if __name__ == "__main__":
raise SystemExit(main())