Against talk /face's guided pose (first paragraph 246 ms median), SemIf in parallel adds 32 ms and SemIf-first adds 94 ms (n=72 each, noise floor 16.5 ms). Removing the pose header saves only ~31 ms, and SemIf shares GPU 1 with the LLM. Acceptable pose 67% vs 92% on clear-emotion lines, and the mood carried through mundane follow-ups 7/15 vs 14/15. SemIf gestures far less (13% vs 58%). README: rotations cost options^2 in suffix tokens, and /decide/shared returns 422 when an object state's last value ends in ) ; or }.
150 lines
7.3 KiB
Python
150 lines
7.3 KiB
Python
"""Analysis for cicada_mood_latency.py output. Prints the report and writes <result>.report.json.
|
|
python cicada_mood_analyze.py cicada-mood-2026-09-27/result.json
|
|
"""
|
|
import json, random, re, statistics as st, sys
|
|
from collections import Counter, defaultdict
|
|
|
|
d = json.load(open(sys.argv[1]))
|
|
rows = d["rows"]
|
|
S = [r for r in rows if r["kind"] == "single"]
|
|
ARC = [r for r in rows if r["kind"] == "arc"]
|
|
rep = {}
|
|
|
|
|
|
def med(xs):
|
|
xs = [x for x in xs if x is not None]
|
|
if not xs:
|
|
return None
|
|
q = st.quantiles(xs, n=4) if len(xs) > 1 else [xs[0]] * 3
|
|
return {"median": round(st.median(xs), 1), "p25": round(q[0], 1), "p75": round(q[2], 1),
|
|
"min": round(min(xs), 1), "max": round(max(xs), 1), "n": len(xs)}
|
|
|
|
|
|
def boot_median_delta(pairs, reps=10000, seed=7):
|
|
"""pairs: {case: [delta, ...]}; resample cases, median of all deltas in the resample."""
|
|
rng, keys, out = random.Random(seed), list(pairs), []
|
|
for _ in range(reps):
|
|
sample = [x for k in (rng.choice(keys) for _ in keys) for x in pairs[k]]
|
|
out.append(st.median(sample))
|
|
out.sort()
|
|
return round(out[int(.025 * reps)], 1), round(out[int(.975 * reps)], 1)
|
|
|
|
|
|
def latency(sub, conds):
|
|
by = defaultdict(dict)
|
|
for r in sub:
|
|
by[(r["case"], r["run"])][r["cond"]] = r
|
|
out = {}
|
|
for c in conds:
|
|
rs = [r for r in sub if r["cond"] == c]
|
|
o = {"critical_ms": med([r["crit"] for r in rs]), "first_token_ms": med([r["t"].get("first_token") for r in rs]),
|
|
"first_para_llm_ms": med([r["t"].get("first_para") for r in rs]), "prompt_tokens": med([r["prompt_tokens"] for r in rs])}
|
|
if c == "A":
|
|
o["pose_closed_ms"] = med([r["t"].get("pose") for r in rs])
|
|
else:
|
|
o["semif_ms"] = med([r["semif"]["ms"] for r in rs])
|
|
o["semif_server_ms"] = med([r["semif"]["srv_ms"] for r in rs])
|
|
if c == "BA":
|
|
o["semif_was_binding"] = sum(r["semif"]["ms"] > r["t"].get("first_para", 1e9) for r in rs)
|
|
out[c] = o
|
|
for c in conds:
|
|
if c == "A":
|
|
continue
|
|
pairs = defaultdict(list)
|
|
for (case, run), cs in by.items():
|
|
if "A" in cs and c in cs:
|
|
pairs[case].append(cs[c]["crit"] - cs["A"]["crit"])
|
|
allp = [x for v in pairs.values() for x in v]
|
|
out[f"{c}_minus_A"] = {"median_ms": round(st.median(allp), 1), "ci95_case_bootstrap": boot_median_delta(pairs),
|
|
"faster_in": f"{sum(x < 0 for x in allp)}/{len(allp)}"}
|
|
# noise floor: A against itself, same case, different runs
|
|
a = defaultdict(list)
|
|
for r in sub:
|
|
if r["cond"] == "A":
|
|
a[r["case"]].append(r["crit"])
|
|
diffs = [abs(x - y) for v in a.values() for i, x in enumerate(v) for y in v[i + 1:]]
|
|
out["noise_floor_A_vs_A_abs_diff_ms"] = med(diffs)
|
|
return out
|
|
|
|
|
|
rep["latency_singles"] = latency(S, ["A", "BA", "BP"])
|
|
rep["latency_arcs"] = latency(ARC, ["A", "BP"])
|
|
|
|
|
|
def works(sub, cond):
|
|
rs = [r for r in sub if r["cond"] == cond]
|
|
ok = sum(r["pose"] in r["ok_set"] for r in rs)
|
|
bad = [f"{r['case']}#{r['run']}: {r['pose']}" for r in rs if r["pose"] in r["bad_set"]]
|
|
miss = [f"{r['case']}#{r['run']}: {r['pose']} (want {'/'.join(r['ok_set'])})" for r in rs
|
|
if r["pose"] not in r["ok_set"] and r["pose"] not in r["bad_set"]]
|
|
return {"acceptable": f"{ok}/{len(rs)}", "egregious": bad, "other_misses": miss,
|
|
"controls": f"{sum(r['pose'] in r['ok_set'] for r in rs if r['tag'] == 'control')}/"
|
|
f"{sum(r['tag'] == 'control' for r in rs)}",
|
|
"distinct_poses": len({r['pose'] for r in rs}),
|
|
"pose_counts": dict(Counter(r["pose"] for r in rs).most_common())}
|
|
|
|
|
|
rep["works_singles"] = {c: works(S, c) for c in ("A", "BA", "BP")}
|
|
# SemIf is deterministic given its input: BA and BP singles must agree with each other on every row
|
|
sem_pose = {}
|
|
for r in S:
|
|
if r["cond"] in ("BA", "BP"):
|
|
sem_pose.setdefault(r["case"], set()).add(r["pose"])
|
|
rep["semif_deterministic_across_runs_and_arms"] = all(len(v) == 1 for v in sem_pose.values())
|
|
|
|
# agreement: SemIf vs A, against A's agreement with itself
|
|
a_poses = defaultdict(list)
|
|
for r in S:
|
|
if r["cond"] == "A":
|
|
a_poses[r["case"]].append(r["pose"])
|
|
aa = [x == y for v in a_poses.values() for i, x in enumerate(v) for y in v[i + 1:]]
|
|
sa = [p == next(iter(sem_pose[c])) for c, v in a_poses.items() for p in v]
|
|
rep["agreement"] = {"A_vs_A_pairwise": round(sum(aa) / len(aa), 3), "SemIf_vs_A": round(sum(sa) / len(sa), 3),
|
|
"A_all_3_runs_same_pose": f"{sum(len(set(v)) == 1 for v in a_poses.values())}/{len(a_poses)}"}
|
|
|
|
# gestures
|
|
CALM = {"lights", "timer", "math", "weather"}
|
|
for c in ("A", "BP"):
|
|
rs = [r for r in S if r["cond"] == c]
|
|
rep.setdefault("gestures", {})[c] = {
|
|
"gesture_rate_all": f"{sum(bool(r['gesture']) and r['gesture'] != 'none' for r in rs)}/{len(rs)}",
|
|
"gesture_rate_calm_commands": f"{sum(bool(r['gesture']) and r['gesture'] != 'none' for r in rs if r['case'] in CALM)}/"
|
|
f"{sum(r['case'] in CALM for r in rs)}",
|
|
"counts": dict(Counter(r["gesture"] or "none" for r in rs).most_common())}
|
|
|
|
# the SemIf-only grid (deterministic, one call per case per variant) + null control
|
|
lab = {r["case"]: (r["ok_set"], r["bad_set"], r["tag"]) for r in S}
|
|
grid = {}
|
|
for key in ("authored_single", "authored_rot", "short_single", "short_rot", "plain_single", "plain_rot", "blind"):
|
|
ok = sum(e[key] in lab[c][0] for c, e in d["semif_only"].items())
|
|
bad = [f"{c}: {e[key]}" for c, e in d["semif_only"].items() if e[key] in lab[c][1]]
|
|
grid[key] = {"acceptable": f"{ok}/{len(d['semif_only'])}", "egregious": bad,
|
|
"controls": f"{sum(e[key] in lab[c][0] for c, e in d['semif_only'].items() if lab[c][2] == 'control')}/4"}
|
|
grid["blind_poses"] = dict(Counter(e["blind"] for e in d["semif_only"].values()))
|
|
rep["semif_grid_singles"] = grid
|
|
|
|
# arcs
|
|
arcw = {}
|
|
for c in ("A", "BP"):
|
|
rs = [r for r in ARC if r["cond"] == c]
|
|
carry = [r for r in rs if r["case"] in ("pivot-4", "vet-3", "vet-4", "admit-3", "admit-4")]
|
|
arcw[c] = {"acceptable": f"{sum(r['pose'] in r['ok_set'] for r in rs)}/{len(rs)}",
|
|
"carry_turns": f"{sum(r['pose'] in r['ok_set'] for r in carry)}/{len(carry)}",
|
|
"egregious": [f"{r['case']}#{r['run']}: {r['pose']}" for r in rs if r["pose"] in r["bad_set"]],
|
|
"sequences": {f"{arc}#{run}": ",".join(r["pose"] or "-" for r in sorted(
|
|
[x for x in rs if x["arc"] == arc and x["run"] == run], key=lambda x: x["turn"]))
|
|
for arc in ("pivot", "vet", "admit") for run in range(d["runs"])}}
|
|
rep["works_arcs"] = arcw
|
|
|
|
# does the line fit the face? vocal sounds that fight the pose
|
|
BRIGHT_T, DARK_T = re.compile(r"\((giggle|chuckle)"), re.compile(r"\((sigh|sniffle|groan)")
|
|
BRIGHT_P, DARK_P = {"joyful", "delighted", "playful"}, {"sad", "concerned"}
|
|
for c in ("A", "BA", "BP"):
|
|
rs = [r for r in rows if r["cond"] == c]
|
|
clash = [f"{r['case']}#{r['run']} {r['pose']}: {r['text'][:70]}" for r in rs
|
|
if (BRIGHT_T.search(r["text"]) and r["pose"] in DARK_P) or (DARK_T.search(r["text"]) and r["pose"] in BRIGHT_P)]
|
|
rep.setdefault("sound_vs_pose_clashes", {})[c] = {"n": f"{len(clash)}/{len(rs)}", "examples": clash[:6]}
|
|
|
|
json.dump(rep, open(sys.argv[1].replace(".json", ".report.json"), "w"), indent=1)
|
|
print(json.dumps(rep, indent=1))
|