Harness consumer_fit.py runs a consumer's per-turn decisions over hand-labelled cases, with rotations, a content-free null control and tagged positive controls. Cicada: an input-only 'does this earn a visible reaction?' gate scored 30/31 with descriptive options and 19/31 with terse yes/no options. Scoped by Cicada's 2026-09-20 ruling (affect is emitted once, no mood-ring classifier). Wyrd: on 3 real seed graphs, the first place-change wording failed its positive controls (1/6 moves). A location-anchored rewording scored 21/21, and exit selection scored 18/21. Semif fits the choice, not writing the node.
105 lines
5.5 KiB
Python
105 lines
5.5 KiB
Python
"""SPIKE (2026-09-27): is semif-serve a clean fit for a consumer's per-turn decision?
|
|
Throwaway measurement, no service change. A scenario file names the decisions a consumer makes
|
|
per turn and a set of hand-labelled cases; every case goes out as ONE /decide/shared request,
|
|
every decision with "orderings": "rotations".
|
|
|
|
Conditions per case:
|
|
evidence the case's real state
|
|
Labels may be one option id or a list of acceptable ones; a case may also name "forbid"
|
|
options per decision (an egregious pick, e.g. a joyful face on bad news), counted separately.
|
|
blind NULL CONTROL: the same decisions over a content-free state. Whatever it still gets
|
|
"right" is the prior the question and options carry by themselves, so the evidence
|
|
condition is judged against it rather than against 0%.
|
|
Positive controls are cases tagged "control" in the scenario file: unambiguous by construction,
|
|
so a miss there means the instrument (question wording, option text) is broken, not the model.
|
|
The service is deterministic (acceptance: same 144 rows twice, gap 0.0), so repeats measure
|
|
latency only: RUNS full passes, median and min..max reported.
|
|
SEMIF_URL=... SEMIF_TOKEN=... uv run --with httpx python consumer_fit.py scenario.json out.json
|
|
"""
|
|
import json, os, statistics as st, sys, time
|
|
import httpx
|
|
|
|
U, H = os.environ["SEMIF_URL"], {"Authorization": f"Bearer {os.environ['SEMIF_TOKEN']}"}
|
|
RUNS = int(os.environ.get("RUNS", "3"))
|
|
BLIND = "(No evidence is available for this turn.)"
|
|
scen = json.load(open(sys.argv[1]))
|
|
decisions = scen["decisions"]
|
|
|
|
|
|
def ask(state, only=None):
|
|
body = {"state": state, "decisions": [
|
|
{"id": d["id"], "question": d["question"], "options": d["options"], "orderings": "rotations"}
|
|
for d in decisions if only is None or d["id"] in only]}
|
|
t = time.perf_counter()
|
|
r = httpx.post(f"{U}/decide/shared", headers=H, json=body, timeout=120)
|
|
r.raise_for_status()
|
|
e2e = (time.perf_counter() - t) * 1000
|
|
j = r.json()
|
|
out = {}
|
|
for res in j["results"]:
|
|
c = res["combined"]
|
|
p = dict(zip(res["option_ids"], c["probabilities"]))
|
|
out[res["id"]] = {"top": c["top"], "p_top": round(p[c["top"]], 3), "agreement": c["agreement"],
|
|
"probs": {k: round(v, 3) for k, v in p.items()},
|
|
"input_tokens": res["orderings"][0]["input_tokens"]}
|
|
return out, e2e, j["timing"]["total_seconds"] * 1000
|
|
|
|
|
|
def ok(top, label): # a label is one option id or a list of acceptable ones
|
|
return None if label is None else top in ([label] if isinstance(label, str) else label)
|
|
|
|
|
|
rows, lat = [], {"e2e_ms": [], "srv_ms": []}
|
|
for case in scen["cases"]:
|
|
labels, forbid = case.get("labels", {}), case.get("forbid", {})
|
|
runs = [ask(case["state"]) for _ in range(RUNS)]
|
|
ev = runs[0][0]
|
|
assert all(r[0][k]["top"] == ev[k]["top"] for r in runs for k in ev), f"{case['id']}: nondeterministic top"
|
|
for _, e2e, srv in runs:
|
|
lat["e2e_ms"].append(e2e); lat["srv_ms"].append(srv)
|
|
blind, _, _ = ask(BLIND)
|
|
for d in decisions:
|
|
k = d["id"]
|
|
rows.append({"case": case["id"], "tag": case.get("tag", "case"), "decision": k,
|
|
"label": labels.get(k), "top": ev[k]["top"], "p_top": ev[k]["p_top"],
|
|
"agreement": ev[k]["agreement"], "probs": ev[k]["probs"],
|
|
"blind_top": blind[k]["top"], "input_tokens": ev[k]["input_tokens"],
|
|
"ok": ok(ev[k]["top"], labels.get(k)), "blind_ok": ok(blind[k]["top"], labels.get(k)),
|
|
"egregious": ev[k]["top"] in forbid.get(k, []),
|
|
"blind_egregious": blind[k]["top"] in forbid.get(k, [])})
|
|
|
|
|
|
def acc(sub, key="ok"):
|
|
s = [r[key] for r in sub if r[key] is not None]
|
|
return {"n": len(s), "correct": sum(s), "acc": round(sum(s) / len(s), 3) if s else None}
|
|
|
|
|
|
def med(xs):
|
|
return {"median": round(st.median(xs), 1), "min": round(min(xs), 1), "max": round(max(xs), 1), "n": len(xs)}
|
|
|
|
|
|
report = {"scenario": scen["name"], "cases": len(scen["cases"]), "runs": RUNS,
|
|
"latency_per_turn": {k: med(v) for k, v in lat.items()},
|
|
"by_decision": {}}
|
|
for d in decisions:
|
|
sub = [r for r in rows if r["decision"] == d["id"]]
|
|
unan = [r for r in sub if r["agreement"] == 1.0]
|
|
split = [r for r in sub if r["agreement"] < 1.0]
|
|
report["by_decision"][d["id"]] = {
|
|
"evidence": acc(sub), "blind_null": acc(sub, "blind_ok"),
|
|
"controls": acc([r for r in sub if r["tag"] == "control"]),
|
|
"non_controls": acc([r for r in sub if r["tag"] != "control"]),
|
|
"by_label": {lab: {"evidence": acc([r for r in sub if r["label"] == lab]),
|
|
"blind_null": acc([r for r in sub if r["label"] == lab], "blind_ok")}
|
|
for lab in sorted({r["label"] for r in sub if isinstance(r["label"], str)})},
|
|
"unanimous": acc(unan), "split": acc(split),
|
|
"egregious": sum(r["egregious"] for r in sub), "blind_egregious": sum(r["blind_egregious"] for r in sub),
|
|
"misses": [f"{r['case']}: said {r['top']} (p={r['p_top']}, agree={r['agreement']}) want {r['label']}"
|
|
for r in sub if r["ok"] is False]}
|
|
json.dump({"report": report, "rows": rows}, open(sys.argv[2], "w"), indent=1)
|
|
print(json.dumps(report, indent=1))
|
|
for r in rows:
|
|
mark = "BAD!" if r["egregious"] else {True: "ok ", False: "MISS", None: " - "}[r["ok"]]
|
|
print(f"{mark} {r['case']:<28} {r['decision']:<16} top={r['top']:<14} p={r['p_top']:<5} "
|
|
f"agree={r['agreement']:<4} want={r['label']} blind={r['blind_top']}")
|