"""SPIKE (2026-09-27): is semif-serve a clean fit for a consumer's per-turn decision? Throwaway measurement, no service change. A scenario file names the decisions a consumer makes per turn and a set of hand-labelled cases; every case goes out as ONE /decide/shared request, every decision with "orderings": "rotations". Conditions per case: evidence the case's real state Labels may be one option id or a list of acceptable ones; a case may also name "forbid" options per decision (an egregious pick, e.g. a joyful face on bad news), counted separately. blind NULL CONTROL: the same decisions over a content-free state. Whatever it still gets "right" is the prior the question and options carry by themselves, so the evidence condition is judged against it rather than against 0%. Positive controls are cases tagged "control" in the scenario file: unambiguous by construction, so a miss there means the instrument (question wording, option text) is broken, not the model. The service is deterministic (acceptance: same 144 rows twice, gap 0.0), so repeats measure latency only: RUNS full passes, median and min..max reported. SEMIF_URL=... SEMIF_TOKEN=... uv run --with httpx python consumer_fit.py scenario.json out.json """ import json, os, statistics as st, sys, time import httpx U, H = os.environ["SEMIF_URL"], {"Authorization": f"Bearer {os.environ['SEMIF_TOKEN']}"} RUNS = int(os.environ.get("RUNS", "3")) BLIND = "(No evidence is available for this turn.)" scen = json.load(open(sys.argv[1])) decisions = scen["decisions"] def ask(state, only=None): body = {"state": state, "decisions": [ {"id": d["id"], "question": d["question"], "options": d["options"], "orderings": "rotations"} for d in decisions if only is None or d["id"] in only]} t = time.perf_counter() r = httpx.post(f"{U}/decide/shared", headers=H, json=body, timeout=120) r.raise_for_status() e2e = (time.perf_counter() - t) * 1000 j = r.json() out = {} for res in j["results"]: c = res["combined"] p = dict(zip(res["option_ids"], c["probabilities"])) out[res["id"]] = {"top": c["top"], "p_top": round(p[c["top"]], 3), "agreement": c["agreement"], "probs": {k: round(v, 3) for k, v in p.items()}, "input_tokens": res["orderings"][0]["input_tokens"]} return out, e2e, j["timing"]["total_seconds"] * 1000 def ok(top, label): # a label is one option id or a list of acceptable ones return None if label is None else top in ([label] if isinstance(label, str) else label) rows, lat = [], {"e2e_ms": [], "srv_ms": []} for case in scen["cases"]: labels, forbid = case.get("labels", {}), case.get("forbid", {}) runs = [ask(case["state"]) for _ in range(RUNS)] ev = runs[0][0] assert all(r[0][k]["top"] == ev[k]["top"] for r in runs for k in ev), f"{case['id']}: nondeterministic top" for _, e2e, srv in runs: lat["e2e_ms"].append(e2e); lat["srv_ms"].append(srv) blind, _, _ = ask(BLIND) for d in decisions: k = d["id"] rows.append({"case": case["id"], "tag": case.get("tag", "case"), "decision": k, "label": labels.get(k), "top": ev[k]["top"], "p_top": ev[k]["p_top"], "agreement": ev[k]["agreement"], "probs": ev[k]["probs"], "blind_top": blind[k]["top"], "input_tokens": ev[k]["input_tokens"], "ok": ok(ev[k]["top"], labels.get(k)), "blind_ok": ok(blind[k]["top"], labels.get(k)), "egregious": ev[k]["top"] in forbid.get(k, []), "blind_egregious": blind[k]["top"] in forbid.get(k, [])}) def acc(sub, key="ok"): s = [r[key] for r in sub if r[key] is not None] return {"n": len(s), "correct": sum(s), "acc": round(sum(s) / len(s), 3) if s else None} def med(xs): return {"median": round(st.median(xs), 1), "min": round(min(xs), 1), "max": round(max(xs), 1), "n": len(xs)} report = {"scenario": scen["name"], "cases": len(scen["cases"]), "runs": RUNS, "latency_per_turn": {k: med(v) for k, v in lat.items()}, "by_decision": {}} for d in decisions: sub = [r for r in rows if r["decision"] == d["id"]] unan = [r for r in sub if r["agreement"] == 1.0] split = [r for r in sub if r["agreement"] < 1.0] report["by_decision"][d["id"]] = { "evidence": acc(sub), "blind_null": acc(sub, "blind_ok"), "controls": acc([r for r in sub if r["tag"] == "control"]), "non_controls": acc([r for r in sub if r["tag"] != "control"]), "by_label": {lab: {"evidence": acc([r for r in sub if r["label"] == lab]), "blind_null": acc([r for r in sub if r["label"] == lab], "blind_ok")} for lab in sorted({r["label"] for r in sub if isinstance(r["label"], str)})}, "unanimous": acc(unan), "split": acc(split), "egregious": sum(r["egregious"] for r in sub), "blind_egregious": sum(r["blind_egregious"] for r in sub), "misses": [f"{r['case']}: said {r['top']} (p={r['p_top']}, agree={r['agreement']}) want {r['label']}" for r in sub if r["ok"] is False]} json.dump({"report": report, "rows": rows}, open(sys.argv[2], "w"), indent=1) print(json.dumps(report, indent=1)) for r in rows: mark = "BAD!" if r["egregious"] else {True: "ok ", False: "MISS", None: " - "}[r["ok"]] print(f"{mark} {r['case']:<28} {r['decision']:<16} top={r['top']:<14} p={r['p_top']:<5} " f"agree={r['agreement']:<4} want={r['label']} blind={r['blind_top']}")