Files
esh-pfi-infrastructure/services/semif-serve/spike/consumer_fit.py
T
vh e268ff7c99 spike(semif): consumer fit for Wyrd scene change and Cicada affect gate (no service change)
Harness consumer_fit.py runs a consumer's per-turn decisions over hand-labelled
cases, with rotations, a content-free null control and tagged positive controls.

Cicada: an input-only 'does this earn a visible reaction?' gate scored 30/31
with descriptive options and 19/31 with terse yes/no options. Scoped by
Cicada's 2026-09-20 ruling (affect is emitted once, no mood-ring classifier).

Wyrd: on 3 real seed graphs, the first place-change wording failed its
positive controls (1/6 moves). A location-anchored rewording scored 21/21, and
exit selection scored 18/21. Semif fits the choice, not writing the node.
2026-09-27 09:20:35 -07:00

105 lines
5.5 KiB
Python

"""SPIKE (2026-09-27): is semif-serve a clean fit for a consumer's per-turn decision?
Throwaway measurement, no service change. A scenario file names the decisions a consumer makes
per turn and a set of hand-labelled cases; every case goes out as ONE /decide/shared request,
every decision with "orderings": "rotations".
Conditions per case:
evidence the case's real state
Labels may be one option id or a list of acceptable ones; a case may also name "forbid"
options per decision (an egregious pick, e.g. a joyful face on bad news), counted separately.
blind NULL CONTROL: the same decisions over a content-free state. Whatever it still gets
"right" is the prior the question and options carry by themselves, so the evidence
condition is judged against it rather than against 0%.
Positive controls are cases tagged "control" in the scenario file: unambiguous by construction,
so a miss there means the instrument (question wording, option text) is broken, not the model.
The service is deterministic (acceptance: same 144 rows twice, gap 0.0), so repeats measure
latency only: RUNS full passes, median and min..max reported.
SEMIF_URL=... SEMIF_TOKEN=... uv run --with httpx python consumer_fit.py scenario.json out.json
"""
import json, os, statistics as st, sys, time
import httpx
U, H = os.environ["SEMIF_URL"], {"Authorization": f"Bearer {os.environ['SEMIF_TOKEN']}"}
RUNS = int(os.environ.get("RUNS", "3"))
BLIND = "(No evidence is available for this turn.)"
scen = json.load(open(sys.argv[1]))
decisions = scen["decisions"]
def ask(state, only=None):
body = {"state": state, "decisions": [
{"id": d["id"], "question": d["question"], "options": d["options"], "orderings": "rotations"}
for d in decisions if only is None or d["id"] in only]}
t = time.perf_counter()
r = httpx.post(f"{U}/decide/shared", headers=H, json=body, timeout=120)
r.raise_for_status()
e2e = (time.perf_counter() - t) * 1000
j = r.json()
out = {}
for res in j["results"]:
c = res["combined"]
p = dict(zip(res["option_ids"], c["probabilities"]))
out[res["id"]] = {"top": c["top"], "p_top": round(p[c["top"]], 3), "agreement": c["agreement"],
"probs": {k: round(v, 3) for k, v in p.items()},
"input_tokens": res["orderings"][0]["input_tokens"]}
return out, e2e, j["timing"]["total_seconds"] * 1000
def ok(top, label): # a label is one option id or a list of acceptable ones
return None if label is None else top in ([label] if isinstance(label, str) else label)
rows, lat = [], {"e2e_ms": [], "srv_ms": []}
for case in scen["cases"]:
labels, forbid = case.get("labels", {}), case.get("forbid", {})
runs = [ask(case["state"]) for _ in range(RUNS)]
ev = runs[0][0]
assert all(r[0][k]["top"] == ev[k]["top"] for r in runs for k in ev), f"{case['id']}: nondeterministic top"
for _, e2e, srv in runs:
lat["e2e_ms"].append(e2e); lat["srv_ms"].append(srv)
blind, _, _ = ask(BLIND)
for d in decisions:
k = d["id"]
rows.append({"case": case["id"], "tag": case.get("tag", "case"), "decision": k,
"label": labels.get(k), "top": ev[k]["top"], "p_top": ev[k]["p_top"],
"agreement": ev[k]["agreement"], "probs": ev[k]["probs"],
"blind_top": blind[k]["top"], "input_tokens": ev[k]["input_tokens"],
"ok": ok(ev[k]["top"], labels.get(k)), "blind_ok": ok(blind[k]["top"], labels.get(k)),
"egregious": ev[k]["top"] in forbid.get(k, []),
"blind_egregious": blind[k]["top"] in forbid.get(k, [])})
def acc(sub, key="ok"):
s = [r[key] for r in sub if r[key] is not None]
return {"n": len(s), "correct": sum(s), "acc": round(sum(s) / len(s), 3) if s else None}
def med(xs):
return {"median": round(st.median(xs), 1), "min": round(min(xs), 1), "max": round(max(xs), 1), "n": len(xs)}
report = {"scenario": scen["name"], "cases": len(scen["cases"]), "runs": RUNS,
"latency_per_turn": {k: med(v) for k, v in lat.items()},
"by_decision": {}}
for d in decisions:
sub = [r for r in rows if r["decision"] == d["id"]]
unan = [r for r in sub if r["agreement"] == 1.0]
split = [r for r in sub if r["agreement"] < 1.0]
report["by_decision"][d["id"]] = {
"evidence": acc(sub), "blind_null": acc(sub, "blind_ok"),
"controls": acc([r for r in sub if r["tag"] == "control"]),
"non_controls": acc([r for r in sub if r["tag"] != "control"]),
"by_label": {lab: {"evidence": acc([r for r in sub if r["label"] == lab]),
"blind_null": acc([r for r in sub if r["label"] == lab], "blind_ok")}
for lab in sorted({r["label"] for r in sub if isinstance(r["label"], str)})},
"unanimous": acc(unan), "split": acc(split),
"egregious": sum(r["egregious"] for r in sub), "blind_egregious": sum(r["blind_egregious"] for r in sub),
"misses": [f"{r['case']}: said {r['top']} (p={r['p_top']}, agree={r['agreement']}) want {r['label']}"
for r in sub if r["ok"] is False]}
json.dump({"report": report, "rows": rows}, open(sys.argv[2], "w"), indent=1)
print(json.dumps(report, indent=1))
for r in rows:
mark = "BAD!" if r["egregious"] else {True: "ok ", False: "MISS", None: " - "}[r["ok"]]
print(f"{mark} {r['case']:<28} {r['decision']:<16} top={r['top']:<14} p={r['p_top']:<5} "
f"agree={r['agreement']:<4} want={r['label']} blind={r['blind_top']}")