"""Jev-candidate bench (2026-09-30): the replaced-baseline sets through one backend. Stdlib only (urllib), so it runs on fv-ml1's host python against a loopback server. Backends semif semif-serve (SemIf's own scorer and prompt). /decide; order averaging uses the service's own `"orderings": "rotations"` (what a switch through the contract gets). systemone a TypeSafe-style POST /v1/systemone server (a candidate's NATIVE runtime and prompt). Each row is one `choice` question, criteria {option id: description} in the row's order. Order averaging is done here: the n cyclic rotations are n separate requests, combined exactly as semif-serve does (per-ordering log p, mean per option id, renormalised; agreement = share of orderings whose top equals the combined top). Sets (all labelled rows; see ../README in the results doc) authored144, perturbations108 SemIf's own labelled sets (evidence interpretation, 3 options) cicada-w1, cicada-w2 the 2026-09-27 Cicada affect-gate spike, both wordings, 35 cases wyrd the 2026-09-27 Wyrd scene-change spike, 21 cases x 4 decisions Spike rows are also asked over a content-free state (the null control, cond=blind). Conditions (one process lifetime = one repeat r) single every row, caller's order rotations every row, n cyclic rotations averaged repeat authored144 again, caller's order (A-vs-A inside the process) reversed authored144, options reversed (order sensitivity) shuffled authored144, options shuffled, fixed seed (order sensitivity) negative authored144, descriptions rotated one place, ids fixed (NEGATIVE CONTROL) python3 bench_sets.py --backend semif --url http://127.0.0.1:18032 --token-file T --out out.json python3 bench_sets.py --backend systemone --url http://127.0.0.1:18090 --out out.json """ from __future__ import annotations import argparse import json import math import random import sys import time import urllib.error import urllib.request from pathlib import Path DATA = Path(__file__).resolve().parent / "data" BLIND = "(No evidence is available for this turn.)" def load_items() -> list[dict]: items = [] for name in ("authored144", "perturbations108"): for line in (DATA / f"{name}.jsonl").read_text().splitlines(): if not line.strip(): continue r = json.loads(line) items.append({"set": name, "id": r["id"], "cond": "evidence", "state": r["state"], "question": r["question"], "options": [{"id": o["id"], "description": o["description"]} for o in r["options"]], "gold": [r["options"][r["label"]]["id"]], "group": r["group_id"], "tag": "case", "decision": "evidence", "forbid": []}) for f in sorted((DATA / "spike").glob("*.json")): scen = json.loads(f.read_text()) name = scen["name"] setname = ("cicada-w1" if "-w1-" in name else "cicada-w2") if name.startswith("cicada") else "wyrd" for case in scen["cases"]: for d in scen["decisions"]: lab = case.get("labels", {}).get(d["id"]) gold = None if lab is None else ([lab] if isinstance(lab, str) else list(lab)) base = {"set": setname, "decision": d["id"], "question": d["question"], "options": d["options"], "gold": gold, "group": f"{name}/{case['id']}", "tag": case.get("tag", "case"), "forbid": case.get("forbid", {}).get(d["id"], [])} items.append({**base, "id": f"{name}/{case['id']}/{d['id']}", "cond": "evidence", "state": case["state"]}) items.append({**base, "id": f"{name}/{case['id']}/{d['id']}#blind", "cond": "blind", "state": BLIND}) return items def post(url: str, body: dict, headers: dict, timeout: float = 300) -> tuple[int, dict, float]: data = json.dumps(body).encode() req = urllib.request.Request(url, data=data, headers={"Content-Type": "application/json", **headers}) t = time.perf_counter() try: with urllib.request.urlopen(req, timeout=timeout) as resp: status, raw = resp.status, resp.read() except urllib.error.HTTPError as e: status, raw = e.code, e.read() ms = (time.perf_counter() - t) * 1000 try: parsed = json.loads(raw) except ValueError: parsed = {"raw": raw[:500].decode(errors="replace")} return status, parsed, ms def combine(per_ordering: list[dict[str, float]], ids: list[str]) -> dict: """semif-serve's method: mean of log p per option id, renormalised.""" logs = {i: sum(math.log(max(p[i], 1e-300)) for p in per_ordering) / len(per_ordering) for i in ids} m = max(logs.values()) w = {i: math.exp(v - m) for i, v in logs.items()} z = sum(w.values()) probs = {i: w[i] / z for i in ids} top = max(ids, key=lambda i: probs[i]) tops = [max(ids, key=lambda i: p[i]) for p in per_ordering] return {"probabilities": [probs[i] for i in ids], "top": top, "agreement": sum(t == top for t in tops) / len(tops), "orderings": len(per_ordering)} class Semif: def __init__(self, url: str, token: str): self.url, self.h = url.rstrip("/"), {"Authorization": f"Bearer {token}"} def score(self, item: dict, options: list[dict], averaged: bool) -> dict: body = {"id": item["id"], "state": item["state"], "question": item["question"], "options": options} if averaged: body["orderings"] = "rotations" status, j, ms = post(f"{self.url}/decide", body, self.h) if status != 200: return {"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms} ids = [o["id"] for o in options] if averaged: c = j["combined"] probs = dict(zip(j["option_ids"], c["probabilities"])) return {"ok": True, "e2e_ms": ms, "probs": [probs[i] for i in ids], "top": c["top"], "agreement": c["agreement"], "input_tokens": j["orderings"][0]["input_tokens"]} probs = dict(zip(j["option_ids"], j["probabilities"])) return {"ok": True, "e2e_ms": ms, "probs": [probs[i] for i in ids], "top": max(ids, key=lambda i: probs[i]), "option_logits": j.get("option_logits"), "prompt_sha256": j.get("prompt_sha256"), "input_tokens": j.get("input_tokens"), "server_ms": round(j.get("total_seconds", 0) * 1000, 2)} def score_many(self, items: list[dict]) -> list[dict]: """All decisions over one state in ONE /decide/shared (SemIf's shared-prefix path).""" body = {"state": items[0]["state"], "decisions": [ {"id": i["id"], "question": i["question"], "options": i["options"]} for i in items]} status, j, ms = post(f"{self.url}/decide/shared", body, self.h) if status != 200: return [{"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms} for _ in items] out = [] for item, res in zip(items, j["results"]): p = dict(zip(res["option_ids"], res["probabilities"])) ids = [o["id"] for o in item["options"]] out.append({"ok": True, "e2e_ms": ms, "probs": [p[i] for i in ids], "top": max(ids, key=lambda i: p[i])}) return out class SystemOne: def __init__(self, url: str): self.url = url.rstrip("/") def _one(self, item: dict, options: list[dict]) -> dict: body = {"state": item["state"], "questions": {"q": { "type": "choice", "instructions": item["question"], "criteria": {o["id"]: o["description"] for o in options}}}} status, j, ms = post(f"{self.url}/v1/systemone", body, {}) if status != 200: return {"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms} a = j["answers"]["q"] p = a["probabilities"] missing = [o["id"] for o in options if o["id"] not in p] if missing: return {"ok": False, "status": status, "error": f"missing ids {missing}", "e2e_ms": ms} out = {"ok": True, "e2e_ms": ms, "p": {o["id"]: float(p[o["id"]]) for o in options}, "input_tokens": (j.get("usage") or {}).get("input_tokens")} for k in ("unknown_probability", "abstained"): if k in a: out[k] = a[k] return out def score_many(self, items: list[dict]) -> list[dict]: """All decisions over one state as ONE request with several questions (the runtime's own multi-question mode; Intern-Decision scores them in one forward over one prompt).""" body = {"state": items[0]["state"], "questions": { it["decision"]: {"type": "choice", "instructions": it["question"], "criteria": {o["id"]: o["description"] for o in it["options"]}} for it in items}} status, j, ms = post(f"{self.url}/v1/systemone", body, {}) if status != 200: return [{"ok": False, "status": status, "error": str(j)[:300], "e2e_ms": ms} for _ in items] out = [] for it in items: p = j["answers"][it["decision"]]["probabilities"] ids = [o["id"] for o in it["options"]] out.append({"ok": True, "e2e_ms": ms, "probs": [float(p[i]) for i in ids], "top": max(ids, key=lambda i: float(p[i]))}) return out def score(self, item: dict, options: list[dict], averaged: bool) -> dict: ids = [o["id"] for o in options] if not averaged: r = self._one(item, options) if not r["ok"]: return r p = r.pop("p") return {**r, "probs": [p[i] for i in ids], "top": max(ids, key=lambda i: p[i])} per, ms, extra = [], 0.0, {} for k in range(len(options)): r = self._one(item, options[k:] + options[:k]) if not r["ok"]: return r per.append(r["p"]) ms += r["e2e_ms"] if "abstained" in r: extra.setdefault("abstained_orderings", 0) extra["abstained_orderings"] += bool(r["abstained"]) c = combine(per, ids) return {"ok": True, "e2e_ms": ms, "probs": c["probabilities"], "top": c["top"], "agreement": c["agreement"], **extra} def shuffled(options: list[dict], key: str) -> list[dict]: rng = random.Random(f"jev-bench-2026-09-30/{key}") out = list(options) for _ in range(10): rng.shuffle(out) if [o["id"] for o in out] != [o["id"] for o in options]: break return out def rotated_descriptions(options: list[dict]) -> list[dict]: descs = [o["description"] for o in options] descs = descs[1:] + descs[:1] return [{**o, "description": d} for o, d in zip(options, descs)] def gold_desc_id(item: dict) -> str | None: """After rotated_descriptions, position j carries the description that was at j+1, so the gold description (position g) now sits at g-1.""" if not item["gold"]: return None ids = [o["id"] for o in item["options"]] g = ids.index(item["gold"][0]) return ids[(g - 1) % len(ids)] def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("--backend", choices=("semif", "systemone"), required=True) ap.add_argument("--url", required=True) ap.add_argument("--token-file") ap.add_argument("--label", required=True) ap.add_argument("--out", required=True) ap.add_argument("--conditions", default="single,rotations,repeat,reversed,shuffled,negative,multifield") args = ap.parse_args() backend = (Semif(args.url, Path(args.token_file).read_text().strip()) if args.backend == "semif" else SystemOne(args.url)) items = load_items() a144 = [i for i in items if i["set"] == "authored144"] plan = { "single": [(i, i["options"], False) for i in items], "rotations": [(i, i["options"], True) for i in items], "repeat": [(i, i["options"], False) for i in a144], "reversed": [(i, list(reversed(i["options"])), False) for i in a144], "shuffled": [(i, shuffled(i["options"], i["id"]), False) for i in a144], "negative": [(i, rotated_descriptions(i["options"]), False) for i in a144], } report = {"label": args.label, "backend": args.backend, "url": args.url, "started_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), "conditions": {}} for cond in args.conditions.split(","): if cond == "multifield": # Wyrd's real shape: the 4 decisions of one turn over one state, in one request groups: dict[str, list[dict]] = {} for it in items: if it["set"] == "wyrd" and it["cond"] == "evidence": groups.setdefault(it["group"], []).append(it) rows, t0 = [], time.perf_counter() for g, its in groups.items(): for it, r in zip(its, backend.score_many(its)): rows.append({"id": it["id"], "set": it["set"], "cond": it["cond"], "decision": it["decision"], "tag": it["tag"], "group": it["group"], "gold": it["gold"], "forbid": it["forbid"], "option_ids": [o["id"] for o in it["options"]], "gold_desc_id": None, **r}) fails = sum(not r["ok"] for r in rows) report["conditions"][cond] = {"rows": rows, "seconds": round(time.perf_counter() - t0, 1), "failures": fails} ok = [x for x in rows if x["ok"] and x["gold"]] print(f"[{args.label}] {cond}: {len(rows)} rows in {len(groups)} requests, {fails} failed, " f"acc {sum(x['top'] in x['gold'] for x in ok) / max(1, len(ok)):.3f}", flush=True) continue rows, t0, fails = [], time.perf_counter(), 0 for item, options, averaged in plan[cond]: r = backend.score(item, options, averaged) fails += not r["ok"] rows.append({"id": item["id"], "set": item["set"], "cond": item["cond"], "decision": item["decision"], "tag": item["tag"], "group": item["group"], "gold": item["gold"], "forbid": item["forbid"], "option_ids": [o["id"] for o in options], # negative control: the id that now carries the gold option's description "gold_desc_id": gold_desc_id(item) if cond == "negative" else None, **r}) if fails >= 5 and fails == len(rows): print(f"[{args.label}] {cond}: first {fails} rows all failed: {r}", file=sys.stderr) return 2 report["conditions"][cond] = {"rows": rows, "seconds": round(time.perf_counter() - t0, 1), "failures": fails} ok = [x for x in rows if x["ok"] and x["gold"] and x["cond"] == "evidence"] acc = sum(x["top"] in x["gold"] for x in ok) / len(ok) if ok else float("nan") print(f"[{args.label}] {cond}: {len(rows)} rows, {fails} failed, " f"{report['conditions'][cond]['seconds']}s, labelled-evidence acc {acc:.3f}", flush=True) report["finished_utc"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()) Path(args.out).write_text(json.dumps(report)) return 0 if __name__ == "__main__": raise SystemExit(main())