Files
esh-pfi-infrastructure/services/semif-serve/spike/cicada_mood_analyze.py
T
vh 7e11cf247b spike(semif): SemIf as Cicada's mood source is slower and less apt (no service change)
Against talk /face's guided pose (first paragraph 246 ms median), SemIf in
parallel adds 32 ms and SemIf-first adds 94 ms (n=72 each, noise floor 16.5 ms).
Removing the pose header saves only ~31 ms, and SemIf shares GPU 1 with the LLM.
Acceptable pose 67% vs 92% on clear-emotion lines, and the mood carried through
mundane follow-ups 7/15 vs 14/15. SemIf gestures far less (13% vs 58%).

README: rotations cost options^2 in suffix tokens, and /decide/shared returns
422 when an object state's last value ends in ) ; or }.
2026-09-27 09:47:28 -07:00

150 lines
7.3 KiB
Python

"""Analysis for cicada_mood_latency.py output. Prints the report and writes <result>.report.json.
python cicada_mood_analyze.py cicada-mood-2026-09-27/result.json
"""
import json, random, re, statistics as st, sys
from collections import Counter, defaultdict
d = json.load(open(sys.argv[1]))
rows = d["rows"]
S = [r for r in rows if r["kind"] == "single"]
ARC = [r for r in rows if r["kind"] == "arc"]
rep = {}
def med(xs):
xs = [x for x in xs if x is not None]
if not xs:
return None
q = st.quantiles(xs, n=4) if len(xs) > 1 else [xs[0]] * 3
return {"median": round(st.median(xs), 1), "p25": round(q[0], 1), "p75": round(q[2], 1),
"min": round(min(xs), 1), "max": round(max(xs), 1), "n": len(xs)}
def boot_median_delta(pairs, reps=10000, seed=7):
"""pairs: {case: [delta, ...]}; resample cases, median of all deltas in the resample."""
rng, keys, out = random.Random(seed), list(pairs), []
for _ in range(reps):
sample = [x for k in (rng.choice(keys) for _ in keys) for x in pairs[k]]
out.append(st.median(sample))
out.sort()
return round(out[int(.025 * reps)], 1), round(out[int(.975 * reps)], 1)
def latency(sub, conds):
by = defaultdict(dict)
for r in sub:
by[(r["case"], r["run"])][r["cond"]] = r
out = {}
for c in conds:
rs = [r for r in sub if r["cond"] == c]
o = {"critical_ms": med([r["crit"] for r in rs]), "first_token_ms": med([r["t"].get("first_token") for r in rs]),
"first_para_llm_ms": med([r["t"].get("first_para") for r in rs]), "prompt_tokens": med([r["prompt_tokens"] for r in rs])}
if c == "A":
o["pose_closed_ms"] = med([r["t"].get("pose") for r in rs])
else:
o["semif_ms"] = med([r["semif"]["ms"] for r in rs])
o["semif_server_ms"] = med([r["semif"]["srv_ms"] for r in rs])
if c == "BA":
o["semif_was_binding"] = sum(r["semif"]["ms"] > r["t"].get("first_para", 1e9) for r in rs)
out[c] = o
for c in conds:
if c == "A":
continue
pairs = defaultdict(list)
for (case, run), cs in by.items():
if "A" in cs and c in cs:
pairs[case].append(cs[c]["crit"] - cs["A"]["crit"])
allp = [x for v in pairs.values() for x in v]
out[f"{c}_minus_A"] = {"median_ms": round(st.median(allp), 1), "ci95_case_bootstrap": boot_median_delta(pairs),
"faster_in": f"{sum(x < 0 for x in allp)}/{len(allp)}"}
# noise floor: A against itself, same case, different runs
a = defaultdict(list)
for r in sub:
if r["cond"] == "A":
a[r["case"]].append(r["crit"])
diffs = [abs(x - y) for v in a.values() for i, x in enumerate(v) for y in v[i + 1:]]
out["noise_floor_A_vs_A_abs_diff_ms"] = med(diffs)
return out
rep["latency_singles"] = latency(S, ["A", "BA", "BP"])
rep["latency_arcs"] = latency(ARC, ["A", "BP"])
def works(sub, cond):
rs = [r for r in sub if r["cond"] == cond]
ok = sum(r["pose"] in r["ok_set"] for r in rs)
bad = [f"{r['case']}#{r['run']}: {r['pose']}" for r in rs if r["pose"] in r["bad_set"]]
miss = [f"{r['case']}#{r['run']}: {r['pose']} (want {'/'.join(r['ok_set'])})" for r in rs
if r["pose"] not in r["ok_set"] and r["pose"] not in r["bad_set"]]
return {"acceptable": f"{ok}/{len(rs)}", "egregious": bad, "other_misses": miss,
"controls": f"{sum(r['pose'] in r['ok_set'] for r in rs if r['tag'] == 'control')}/"
f"{sum(r['tag'] == 'control' for r in rs)}",
"distinct_poses": len({r['pose'] for r in rs}),
"pose_counts": dict(Counter(r["pose"] for r in rs).most_common())}
rep["works_singles"] = {c: works(S, c) for c in ("A", "BA", "BP")}
# SemIf is deterministic given its input: BA and BP singles must agree with each other on every row
sem_pose = {}
for r in S:
if r["cond"] in ("BA", "BP"):
sem_pose.setdefault(r["case"], set()).add(r["pose"])
rep["semif_deterministic_across_runs_and_arms"] = all(len(v) == 1 for v in sem_pose.values())
# agreement: SemIf vs A, against A's agreement with itself
a_poses = defaultdict(list)
for r in S:
if r["cond"] == "A":
a_poses[r["case"]].append(r["pose"])
aa = [x == y for v in a_poses.values() for i, x in enumerate(v) for y in v[i + 1:]]
sa = [p == next(iter(sem_pose[c])) for c, v in a_poses.items() for p in v]
rep["agreement"] = {"A_vs_A_pairwise": round(sum(aa) / len(aa), 3), "SemIf_vs_A": round(sum(sa) / len(sa), 3),
"A_all_3_runs_same_pose": f"{sum(len(set(v)) == 1 for v in a_poses.values())}/{len(a_poses)}"}
# gestures
CALM = {"lights", "timer", "math", "weather"}
for c in ("A", "BP"):
rs = [r for r in S if r["cond"] == c]
rep.setdefault("gestures", {})[c] = {
"gesture_rate_all": f"{sum(bool(r['gesture']) and r['gesture'] != 'none' for r in rs)}/{len(rs)}",
"gesture_rate_calm_commands": f"{sum(bool(r['gesture']) and r['gesture'] != 'none' for r in rs if r['case'] in CALM)}/"
f"{sum(r['case'] in CALM for r in rs)}",
"counts": dict(Counter(r["gesture"] or "none" for r in rs).most_common())}
# the SemIf-only grid (deterministic, one call per case per variant) + null control
lab = {r["case"]: (r["ok_set"], r["bad_set"], r["tag"]) for r in S}
grid = {}
for key in ("authored_single", "authored_rot", "short_single", "short_rot", "plain_single", "plain_rot", "blind"):
ok = sum(e[key] in lab[c][0] for c, e in d["semif_only"].items())
bad = [f"{c}: {e[key]}" for c, e in d["semif_only"].items() if e[key] in lab[c][1]]
grid[key] = {"acceptable": f"{ok}/{len(d['semif_only'])}", "egregious": bad,
"controls": f"{sum(e[key] in lab[c][0] for c, e in d['semif_only'].items() if lab[c][2] == 'control')}/4"}
grid["blind_poses"] = dict(Counter(e["blind"] for e in d["semif_only"].values()))
rep["semif_grid_singles"] = grid
# arcs
arcw = {}
for c in ("A", "BP"):
rs = [r for r in ARC if r["cond"] == c]
carry = [r for r in rs if r["case"] in ("pivot-4", "vet-3", "vet-4", "admit-3", "admit-4")]
arcw[c] = {"acceptable": f"{sum(r['pose'] in r['ok_set'] for r in rs)}/{len(rs)}",
"carry_turns": f"{sum(r['pose'] in r['ok_set'] for r in carry)}/{len(carry)}",
"egregious": [f"{r['case']}#{r['run']}: {r['pose']}" for r in rs if r["pose"] in r["bad_set"]],
"sequences": {f"{arc}#{run}": ",".join(r["pose"] or "-" for r in sorted(
[x for x in rs if x["arc"] == arc and x["run"] == run], key=lambda x: x["turn"]))
for arc in ("pivot", "vet", "admit") for run in range(d["runs"])}}
rep["works_arcs"] = arcw
# does the line fit the face? vocal sounds that fight the pose
BRIGHT_T, DARK_T = re.compile(r"\((giggle|chuckle)"), re.compile(r"\((sigh|sniffle|groan)")
BRIGHT_P, DARK_P = {"joyful", "delighted", "playful"}, {"sad", "concerned"}
for c in ("A", "BA", "BP"):
rs = [r for r in rows if r["cond"] == c]
clash = [f"{r['case']}#{r['run']} {r['pose']}: {r['text'][:70]}" for r in rs
if (BRIGHT_T.search(r["text"]) and r["pose"] in DARK_P) or (DARK_T.search(r["text"]) and r["pose"] in BRIGHT_P)]
rep.setdefault("sound_vs_pose_clashes", {})[c] = {"n": f"{len(clash)}/{len(rs)}", "examples": clash[:6]}
json.dump(rep, open(sys.argv[1].replace(".json", ".report.json"), "w"), indent=1)
print(json.dumps(rep, indent=1))