"""Analysis for cicada_mood_latency.py output. Prints the report and writes .report.json. python cicada_mood_analyze.py cicada-mood-2026-09-27/result.json """ import json, random, re, statistics as st, sys from collections import Counter, defaultdict d = json.load(open(sys.argv[1])) rows = d["rows"] S = [r for r in rows if r["kind"] == "single"] ARC = [r for r in rows if r["kind"] == "arc"] rep = {} def med(xs): xs = [x for x in xs if x is not None] if not xs: return None q = st.quantiles(xs, n=4) if len(xs) > 1 else [xs[0]] * 3 return {"median": round(st.median(xs), 1), "p25": round(q[0], 1), "p75": round(q[2], 1), "min": round(min(xs), 1), "max": round(max(xs), 1), "n": len(xs)} def boot_median_delta(pairs, reps=10000, seed=7): """pairs: {case: [delta, ...]}; resample cases, median of all deltas in the resample.""" rng, keys, out = random.Random(seed), list(pairs), [] for _ in range(reps): sample = [x for k in (rng.choice(keys) for _ in keys) for x in pairs[k]] out.append(st.median(sample)) out.sort() return round(out[int(.025 * reps)], 1), round(out[int(.975 * reps)], 1) def latency(sub, conds): by = defaultdict(dict) for r in sub: by[(r["case"], r["run"])][r["cond"]] = r out = {} for c in conds: rs = [r for r in sub if r["cond"] == c] o = {"critical_ms": med([r["crit"] for r in rs]), "first_token_ms": med([r["t"].get("first_token") for r in rs]), "first_para_llm_ms": med([r["t"].get("first_para") for r in rs]), "prompt_tokens": med([r["prompt_tokens"] for r in rs])} if c == "A": o["pose_closed_ms"] = med([r["t"].get("pose") for r in rs]) else: o["semif_ms"] = med([r["semif"]["ms"] for r in rs]) o["semif_server_ms"] = med([r["semif"]["srv_ms"] for r in rs]) if c == "BA": o["semif_was_binding"] = sum(r["semif"]["ms"] > r["t"].get("first_para", 1e9) for r in rs) out[c] = o for c in conds: if c == "A": continue pairs = defaultdict(list) for (case, run), cs in by.items(): if "A" in cs and c in cs: pairs[case].append(cs[c]["crit"] - cs["A"]["crit"]) allp = [x for v in pairs.values() for x in v] out[f"{c}_minus_A"] = {"median_ms": round(st.median(allp), 1), "ci95_case_bootstrap": boot_median_delta(pairs), "faster_in": f"{sum(x < 0 for x in allp)}/{len(allp)}"} # noise floor: A against itself, same case, different runs a = defaultdict(list) for r in sub: if r["cond"] == "A": a[r["case"]].append(r["crit"]) diffs = [abs(x - y) for v in a.values() for i, x in enumerate(v) for y in v[i + 1:]] out["noise_floor_A_vs_A_abs_diff_ms"] = med(diffs) return out rep["latency_singles"] = latency(S, ["A", "BA", "BP"]) rep["latency_arcs"] = latency(ARC, ["A", "BP"]) def works(sub, cond): rs = [r for r in sub if r["cond"] == cond] ok = sum(r["pose"] in r["ok_set"] for r in rs) bad = [f"{r['case']}#{r['run']}: {r['pose']}" for r in rs if r["pose"] in r["bad_set"]] miss = [f"{r['case']}#{r['run']}: {r['pose']} (want {'/'.join(r['ok_set'])})" for r in rs if r["pose"] not in r["ok_set"] and r["pose"] not in r["bad_set"]] return {"acceptable": f"{ok}/{len(rs)}", "egregious": bad, "other_misses": miss, "controls": f"{sum(r['pose'] in r['ok_set'] for r in rs if r['tag'] == 'control')}/" f"{sum(r['tag'] == 'control' for r in rs)}", "distinct_poses": len({r['pose'] for r in rs}), "pose_counts": dict(Counter(r["pose"] for r in rs).most_common())} rep["works_singles"] = {c: works(S, c) for c in ("A", "BA", "BP")} # SemIf is deterministic given its input: BA and BP singles must agree with each other on every row sem_pose = {} for r in S: if r["cond"] in ("BA", "BP"): sem_pose.setdefault(r["case"], set()).add(r["pose"]) rep["semif_deterministic_across_runs_and_arms"] = all(len(v) == 1 for v in sem_pose.values()) # agreement: SemIf vs A, against A's agreement with itself a_poses = defaultdict(list) for r in S: if r["cond"] == "A": a_poses[r["case"]].append(r["pose"]) aa = [x == y for v in a_poses.values() for i, x in enumerate(v) for y in v[i + 1:]] sa = [p == next(iter(sem_pose[c])) for c, v in a_poses.items() for p in v] rep["agreement"] = {"A_vs_A_pairwise": round(sum(aa) / len(aa), 3), "SemIf_vs_A": round(sum(sa) / len(sa), 3), "A_all_3_runs_same_pose": f"{sum(len(set(v)) == 1 for v in a_poses.values())}/{len(a_poses)}"} # gestures CALM = {"lights", "timer", "math", "weather"} for c in ("A", "BP"): rs = [r for r in S if r["cond"] == c] rep.setdefault("gestures", {})[c] = { "gesture_rate_all": f"{sum(bool(r['gesture']) and r['gesture'] != 'none' for r in rs)}/{len(rs)}", "gesture_rate_calm_commands": f"{sum(bool(r['gesture']) and r['gesture'] != 'none' for r in rs if r['case'] in CALM)}/" f"{sum(r['case'] in CALM for r in rs)}", "counts": dict(Counter(r["gesture"] or "none" for r in rs).most_common())} # the SemIf-only grid (deterministic, one call per case per variant) + null control lab = {r["case"]: (r["ok_set"], r["bad_set"], r["tag"]) for r in S} grid = {} for key in ("authored_single", "authored_rot", "short_single", "short_rot", "plain_single", "plain_rot", "blind"): ok = sum(e[key] in lab[c][0] for c, e in d["semif_only"].items()) bad = [f"{c}: {e[key]}" for c, e in d["semif_only"].items() if e[key] in lab[c][1]] grid[key] = {"acceptable": f"{ok}/{len(d['semif_only'])}", "egregious": bad, "controls": f"{sum(e[key] in lab[c][0] for c, e in d['semif_only'].items() if lab[c][2] == 'control')}/4"} grid["blind_poses"] = dict(Counter(e["blind"] for e in d["semif_only"].values())) rep["semif_grid_singles"] = grid # arcs arcw = {} for c in ("A", "BP"): rs = [r for r in ARC if r["cond"] == c] carry = [r for r in rs if r["case"] in ("pivot-4", "vet-3", "vet-4", "admit-3", "admit-4")] arcw[c] = {"acceptable": f"{sum(r['pose'] in r['ok_set'] for r in rs)}/{len(rs)}", "carry_turns": f"{sum(r['pose'] in r['ok_set'] for r in carry)}/{len(carry)}", "egregious": [f"{r['case']}#{r['run']}: {r['pose']}" for r in rs if r["pose"] in r["bad_set"]], "sequences": {f"{arc}#{run}": ",".join(r["pose"] or "-" for r in sorted( [x for x in rs if x["arc"] == arc and x["run"] == run], key=lambda x: x["turn"])) for arc in ("pivot", "vet", "admit") for run in range(d["runs"])}} rep["works_arcs"] = arcw # does the line fit the face? vocal sounds that fight the pose BRIGHT_T, DARK_T = re.compile(r"\((giggle|chuckle)"), re.compile(r"\((sigh|sniffle|groan)") BRIGHT_P, DARK_P = {"joyful", "delighted", "playful"}, {"sad", "concerned"} for c in ("A", "BA", "BP"): rs = [r for r in rows if r["cond"] == c] clash = [f"{r['case']}#{r['run']} {r['pose']}: {r['text'][:70]}" for r in rs if (BRIGHT_T.search(r["text"]) and r["pose"] in DARK_P) or (DARK_T.search(r["text"]) and r["pose"] in BRIGHT_P)] rep.setdefault("sound_vs_pose_clashes", {})[c] = {"n": f"{len(clash)}/{len(rs)}", "examples": clash[:6]} json.dump(rep, open(sys.argv[1].replace(".json", ".report.json"), "w"), indent=1) print(json.dumps(rep, indent=1))