spike(semif): SemIf as Cicada's mood source is slower and less apt (no service change)

Against talk /face's guided pose (first paragraph 246 ms median), SemIf in
parallel adds 32 ms and SemIf-first adds 94 ms (n=72 each, noise floor 16.5 ms).
Removing the pose header saves only ~31 ms, and SemIf shares GPU 1 with the LLM.
Acceptable pose 67% vs 92% on clear-emotion lines, and the mood carried through
mundane follow-ups 7/15 vs 14/15. SemIf gestures far less (13% vs 58%).

README: rotations cost options^2 in suffix tokens, and /decide/shared returns
422 when an object state's last value ends in ) ; or }.
This commit is contained in:
vh
2026-09-27 09:47:28 -07:00
parent 89301a89fe
commit 7e11cf247b
7 changed files with 13163 additions and 1 deletions
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,528 @@
{
"latency_singles": {
"A": {
"critical_ms": {
"median": 246.4,
"p25": 230.4,
"p75": 263.4,
"min": 161.0,
"max": 342.6,
"n": 72
},
"first_token_ms": {
"median": 66.4,
"p25": 61.8,
"p75": 72.2,
"min": 51.2,
"max": 198.1,
"n": 72
},
"first_para_llm_ms": {
"median": 246.4,
"p25": 230.4,
"p75": 263.4,
"min": 161.0,
"max": 342.6,
"n": 72
},
"prompt_tokens": {
"median": 1885.0,
"p25": 1882.0,
"p75": 1887.8,
"min": 1881,
"max": 1893,
"n": 72
},
"pose_closed_ms": {
"median": 85.5,
"p25": 80.7,
"p75": 87.3,
"min": 70.5,
"max": 198.1,
"n": 72
}
},
"BA": {
"critical_ms": {
"median": 285.0,
"p25": 257.0,
"p75": 301.6,
"min": 193.1,
"max": 348.9,
"n": 72
},
"first_token_ms": {
"median": 64.3,
"p25": 57.1,
"p75": 68.4,
"min": 49.0,
"max": 348.5,
"n": 72
},
"first_para_llm_ms": {
"median": 285.0,
"p25": 257.0,
"p75": 301.6,
"min": 177.9,
"max": 348.9,
"n": 72
},
"prompt_tokens": {
"median": 909.0,
"p25": 906.0,
"p75": 911.8,
"min": 905,
"max": 917,
"n": 72
},
"semif_ms": {
"median": 195.6,
"p25": 194.4,
"p75": 197.0,
"min": 156.3,
"max": 206.9,
"n": 72
},
"semif_server_ms": {
"median": 184.8,
"p25": 183.5,
"p75": 186.0,
"min": 145.7,
"max": 191.4,
"n": 72
},
"semif_was_binding": 3
},
"BP": {
"critical_ms": {
"median": 351.8,
"p25": 324.9,
"p75": 368.7,
"min": 236.9,
"max": 425.5,
"n": 72
},
"first_token_ms": {
"median": 67.8,
"p25": 60.6,
"p75": 72.7,
"min": 41.4,
"max": 81.2,
"n": 72
},
"first_para_llm_ms": {
"median": 214.8,
"p25": 189.1,
"p75": 232.9,
"min": 101.3,
"max": 289.1,
"n": 72
},
"prompt_tokens": {
"median": 979.0,
"p25": 976.0,
"p75": 982.8,
"min": 974,
"max": 990,
"n": 72
},
"semif_ms": {
"median": 136.1,
"p25": 135.6,
"p75": 136.6,
"min": 135.0,
"max": 156.3,
"n": 72
},
"semif_server_ms": {
"median": 125.0,
"p25": 124.8,
"p75": 125.5,
"min": 124.5,
"max": 135.6,
"n": 72
}
},
"BA_minus_A": {
"median_ms": 31.9,
"ci95_case_bootstrap": [
27.0,
40.8
],
"faster_in": "10/72"
},
"BP_minus_A": {
"median_ms": 94.3,
"ci95_case_bootstrap": [
84.8,
117.4
],
"faster_in": "0/72"
},
"noise_floor_A_vs_A_abs_diff_ms": {
"median": 16.5,
"p25": 7.0,
"p75": 32.4,
"min": 0.0,
"max": 88.4,
"n": 72
}
},
"latency_arcs": {
"A": {
"critical_ms": {
"median": 242.3,
"p25": 214.6,
"p75": 265.4,
"min": 163.2,
"max": 357.9,
"n": 45
},
"first_token_ms": {
"median": 78.7,
"p25": 67.6,
"p75": 93.3,
"min": 59.9,
"max": 207.5,
"n": 45
},
"first_para_llm_ms": {
"median": 242.3,
"p25": 214.6,
"p75": 265.4,
"min": 163.2,
"max": 357.9,
"n": 45
},
"prompt_tokens": {
"median": 1975,
"p25": 1933.0,
"p75": 2030.5,
"min": 1880,
"max": 2083,
"n": 45
},
"pose_closed_ms": {
"median": 92.7,
"p25": 84.3,
"p75": 109.2,
"min": 76.6,
"max": 225.1,
"n": 45
}
},
"BP": {
"critical_ms": {
"median": 331.7,
"p25": 310.5,
"p75": 369.8,
"min": 249.5,
"max": 409.8,
"n": 45
},
"first_token_ms": {
"median": 72.3,
"p25": 64.9,
"p75": 94.0,
"min": 56.8,
"max": 109.3,
"n": 45
},
"first_para_llm_ms": {
"median": 189.3,
"p25": 169.4,
"p75": 228.1,
"min": 113.1,
"max": 266.9,
"n": 45
},
"prompt_tokens": {
"median": 1060,
"p25": 1008.5,
"p75": 1128.5,
"min": 975,
"max": 1179,
"n": 45
},
"semif_ms": {
"median": 141.5,
"p25": 138.9,
"p75": 144.1,
"min": 135.6,
"max": 148.3,
"n": 45
},
"semif_server_ms": {
"median": 129.7,
"p25": 127.6,
"p75": 132.7,
"min": 124.7,
"max": 136.7,
"n": 45
}
},
"BP_minus_A": {
"median_ms": 97.7,
"ci95_case_bootstrap": [
89.7,
109.2
],
"faster_in": "0/45"
},
"noise_floor_A_vs_A_abs_diff_ms": {
"median": 18.8,
"p25": 6.3,
"p75": 35.0,
"min": 0.3,
"max": 132.6,
"n": 45
}
},
"works_singles": {
"A": {
"acceptable": "66/72",
"egregious": [],
"other_misses": [
"useless#0: neutral (want sad/concerned)",
"shut-up#0: neutral (want sad/concerned)",
"hallway#0: curious (want concerned/surprised/focused)",
"shut-up#1: neutral (want sad/concerned)",
"hallway#2: curious (want concerned/surprised/focused)",
"shut-up#2: neutral (want sad/concerned)"
],
"controls": "12/12",
"distinct_poses": 10,
"pose_counts": {
"concerned": 15,
"neutral": 11,
"surprised": 9,
"delighted": 9,
"playful": 8,
"focused": 8,
"curious": 4,
"joyful": 4,
"sleepy": 3,
"confused": 1
}
},
"BA": {
"acceptable": "48/72",
"egregious": [],
"other_misses": [
"weather#0: curious (want neutral/content/focused)",
"useless#0: annoyed (want sad/concerned)",
"scarecrow#0: neutral (want playful/joyful/delighted/unimpressed)",
"toaster#0: curious (want playful/unimpressed/joyful)",
"shut-up#0: annoyed (want sad/concerned)",
"the-thing#0: focused (want confused/curious)",
"back-door#0: suspicious (want concerned/surprised/focused)",
"hallway#0: curious (want concerned/surprised/focused)",
"toaster#1: curious (want playful/unimpressed/joyful)",
"scarecrow#1: neutral (want playful/joyful/delighted/unimpressed)",
"hallway#1: curious (want concerned/surprised/focused)",
"the-thing#1: focused (want confused/curious)",
"weather#1: curious (want neutral/content/focused)",
"back-door#1: suspicious (want concerned/surprised/focused)",
"shut-up#1: annoyed (want sad/concerned)",
"useless#1: annoyed (want sad/concerned)",
"the-thing#2: focused (want confused/curious)",
"useless#2: annoyed (want sad/concerned)",
"weather#2: curious (want neutral/content/focused)",
"hallway#2: curious (want concerned/surprised/focused)",
"back-door#2: suspicious (want concerned/surprised/focused)",
"toaster#2: curious (want playful/unimpressed/joyful)",
"scarecrow#2: neutral (want playful/joyful/delighted/unimpressed)",
"shut-up#2: annoyed (want sad/concerned)"
],
"controls": "12/12",
"distinct_poses": 9,
"pose_counts": {
"curious": 12,
"concerned": 12,
"joyful": 12,
"focused": 12,
"annoyed": 6,
"surprised": 6,
"suspicious": 6,
"neutral": 3,
"sleepy": 3
}
},
"BP": {
"acceptable": "48/72",
"egregious": [],
"other_misses": [
"weather#0: curious (want neutral/content/focused)",
"useless#0: annoyed (want sad/concerned)",
"scarecrow#0: neutral (want playful/joyful/delighted/unimpressed)",
"toaster#0: curious (want playful/unimpressed/joyful)",
"shut-up#0: annoyed (want sad/concerned)",
"the-thing#0: focused (want confused/curious)",
"back-door#0: suspicious (want concerned/surprised/focused)",
"hallway#0: curious (want concerned/surprised/focused)",
"toaster#1: curious (want playful/unimpressed/joyful)",
"scarecrow#1: neutral (want playful/joyful/delighted/unimpressed)",
"hallway#1: curious (want concerned/surprised/focused)",
"the-thing#1: focused (want confused/curious)",
"weather#1: curious (want neutral/content/focused)",
"back-door#1: suspicious (want concerned/surprised/focused)",
"shut-up#1: annoyed (want sad/concerned)",
"useless#1: annoyed (want sad/concerned)",
"the-thing#2: focused (want confused/curious)",
"useless#2: annoyed (want sad/concerned)",
"weather#2: curious (want neutral/content/focused)",
"hallway#2: curious (want concerned/surprised/focused)",
"back-door#2: suspicious (want concerned/surprised/focused)",
"toaster#2: curious (want playful/unimpressed/joyful)",
"scarecrow#2: neutral (want playful/joyful/delighted/unimpressed)",
"shut-up#2: annoyed (want sad/concerned)"
],
"controls": "12/12",
"distinct_poses": 9,
"pose_counts": {
"curious": 12,
"concerned": 12,
"joyful": 12,
"focused": 12,
"annoyed": 6,
"surprised": 6,
"suspicious": 6,
"neutral": 3,
"sleepy": 3
}
}
},
"semif_deterministic_across_runs_and_arms": true,
"agreement": {
"A_vs_A_pairwise": 0.833,
"SemIf_vs_A": 0.431,
"A_all_3_runs_same_pose": "18/24"
},
"gestures": {
"A": {
"gesture_rate_all": "42/72",
"gesture_rate_calm_commands": "2/12",
"counts": {
"none": 30,
"double-take": 13,
"quick-laugh": 13,
"sigh": 9,
"wink": 4,
"nod": 3
}
},
"BP": {
"gesture_rate_all": "9/72",
"gesture_rate_calm_commands": "0/12",
"counts": {
"none": 63,
"quick-laugh": 3,
"nod": 3,
"head-shake": 3
}
}
},
"semif_grid_singles": {
"authored_single": {
"acceptable": "16/24",
"egregious": [],
"controls": "4/4"
},
"authored_rot": {
"acceptable": "17/24",
"egregious": [],
"controls": "4/4"
},
"short_single": {
"acceptable": "14/24",
"egregious": [],
"controls": "4/4"
},
"short_rot": {
"acceptable": "17/24",
"egregious": [],
"controls": "4/4"
},
"plain_single": {
"acceptable": "19/24",
"egregious": [
"lights: sleepy"
],
"controls": "3/4"
},
"plain_rot": {
"acceptable": "19/24",
"egregious": [
"lights: sleepy"
],
"controls": "3/4"
},
"blind": {
"acceptable": "5/24",
"egregious": [
"dog-died: content",
"grandma: content",
"laid-off: content",
"biopsy: content",
"hallway: content",
"smoke: content",
"back-door: content",
"raccoon: content",
"useless: content",
"shut-up: content"
],
"controls": "1/4"
},
"blind_poses": {
"content": 24
}
},
"works_arcs": {
"A": {
"acceptable": "40/45",
"carry_turns": "14/15",
"egregious": [],
"sequences": {
"pivot#0": "content,content,content,suspicious,suspicious",
"pivot#1": "content,content,content,curious,focused",
"pivot#2": "content,content,content,focused,focused",
"vet#0": "content,curious,sad,sad,sad",
"vet#1": "content,curious,sad,sad,sad",
"vet#2": "content,neutral,sad,sad,sad",
"admit#0": "neutral,surprised,delighted,delighted,delighted",
"admit#1": "neutral,surprised,delighted,joyful,joyful",
"admit#2": "neutral,curious,delighted,joyful,joyful"
}
},
"BP": {
"acceptable": "25/45",
"carry_turns": "7/15",
"egregious": [],
"sequences": {
"pivot#0": "content,content,content,curious,concerned",
"pivot#1": "content,content,content,curious,concerned",
"pivot#2": "content,content,content,curious,suspicious",
"vet#0": "neutral,curious,concerned,neutral,neutral",
"vet#1": "neutral,curious,concerned,neutral,neutral",
"vet#2": "neutral,curious,concerned,neutral,neutral",
"admit#0": "curious,delighted,delighted,content,curious",
"admit#1": "curious,delighted,delighted,content,content",
"admit#2": "curious,delighted,delighted,content,content"
}
}
},
"sound_vs_pose_clashes": {
"A": {
"n": "0/117",
"examples": []
},
"BA": {
"n": "0/72",
"examples": []
},
"BP": {
"n": "0/117",
"examples": []
}
}
}
@@ -0,0 +1,149 @@
"""Analysis for cicada_mood_latency.py output. Prints the report and writes <result>.report.json.
python cicada_mood_analyze.py cicada-mood-2026-09-27/result.json
"""
import json, random, re, statistics as st, sys
from collections import Counter, defaultdict
d = json.load(open(sys.argv[1]))
rows = d["rows"]
S = [r for r in rows if r["kind"] == "single"]
ARC = [r for r in rows if r["kind"] == "arc"]
rep = {}
def med(xs):
xs = [x for x in xs if x is not None]
if not xs:
return None
q = st.quantiles(xs, n=4) if len(xs) > 1 else [xs[0]] * 3
return {"median": round(st.median(xs), 1), "p25": round(q[0], 1), "p75": round(q[2], 1),
"min": round(min(xs), 1), "max": round(max(xs), 1), "n": len(xs)}
def boot_median_delta(pairs, reps=10000, seed=7):
"""pairs: {case: [delta, ...]}; resample cases, median of all deltas in the resample."""
rng, keys, out = random.Random(seed), list(pairs), []
for _ in range(reps):
sample = [x for k in (rng.choice(keys) for _ in keys) for x in pairs[k]]
out.append(st.median(sample))
out.sort()
return round(out[int(.025 * reps)], 1), round(out[int(.975 * reps)], 1)
def latency(sub, conds):
by = defaultdict(dict)
for r in sub:
by[(r["case"], r["run"])][r["cond"]] = r
out = {}
for c in conds:
rs = [r for r in sub if r["cond"] == c]
o = {"critical_ms": med([r["crit"] for r in rs]), "first_token_ms": med([r["t"].get("first_token") for r in rs]),
"first_para_llm_ms": med([r["t"].get("first_para") for r in rs]), "prompt_tokens": med([r["prompt_tokens"] for r in rs])}
if c == "A":
o["pose_closed_ms"] = med([r["t"].get("pose") for r in rs])
else:
o["semif_ms"] = med([r["semif"]["ms"] for r in rs])
o["semif_server_ms"] = med([r["semif"]["srv_ms"] for r in rs])
if c == "BA":
o["semif_was_binding"] = sum(r["semif"]["ms"] > r["t"].get("first_para", 1e9) for r in rs)
out[c] = o
for c in conds:
if c == "A":
continue
pairs = defaultdict(list)
for (case, run), cs in by.items():
if "A" in cs and c in cs:
pairs[case].append(cs[c]["crit"] - cs["A"]["crit"])
allp = [x for v in pairs.values() for x in v]
out[f"{c}_minus_A"] = {"median_ms": round(st.median(allp), 1), "ci95_case_bootstrap": boot_median_delta(pairs),
"faster_in": f"{sum(x < 0 for x in allp)}/{len(allp)}"}
# noise floor: A against itself, same case, different runs
a = defaultdict(list)
for r in sub:
if r["cond"] == "A":
a[r["case"]].append(r["crit"])
diffs = [abs(x - y) for v in a.values() for i, x in enumerate(v) for y in v[i + 1:]]
out["noise_floor_A_vs_A_abs_diff_ms"] = med(diffs)
return out
rep["latency_singles"] = latency(S, ["A", "BA", "BP"])
rep["latency_arcs"] = latency(ARC, ["A", "BP"])
def works(sub, cond):
rs = [r for r in sub if r["cond"] == cond]
ok = sum(r["pose"] in r["ok_set"] for r in rs)
bad = [f"{r['case']}#{r['run']}: {r['pose']}" for r in rs if r["pose"] in r["bad_set"]]
miss = [f"{r['case']}#{r['run']}: {r['pose']} (want {'/'.join(r['ok_set'])})" for r in rs
if r["pose"] not in r["ok_set"] and r["pose"] not in r["bad_set"]]
return {"acceptable": f"{ok}/{len(rs)}", "egregious": bad, "other_misses": miss,
"controls": f"{sum(r['pose'] in r['ok_set'] for r in rs if r['tag'] == 'control')}/"
f"{sum(r['tag'] == 'control' for r in rs)}",
"distinct_poses": len({r['pose'] for r in rs}),
"pose_counts": dict(Counter(r["pose"] for r in rs).most_common())}
rep["works_singles"] = {c: works(S, c) for c in ("A", "BA", "BP")}
# SemIf is deterministic given its input: BA and BP singles must agree with each other on every row
sem_pose = {}
for r in S:
if r["cond"] in ("BA", "BP"):
sem_pose.setdefault(r["case"], set()).add(r["pose"])
rep["semif_deterministic_across_runs_and_arms"] = all(len(v) == 1 for v in sem_pose.values())
# agreement: SemIf vs A, against A's agreement with itself
a_poses = defaultdict(list)
for r in S:
if r["cond"] == "A":
a_poses[r["case"]].append(r["pose"])
aa = [x == y for v in a_poses.values() for i, x in enumerate(v) for y in v[i + 1:]]
sa = [p == next(iter(sem_pose[c])) for c, v in a_poses.items() for p in v]
rep["agreement"] = {"A_vs_A_pairwise": round(sum(aa) / len(aa), 3), "SemIf_vs_A": round(sum(sa) / len(sa), 3),
"A_all_3_runs_same_pose": f"{sum(len(set(v)) == 1 for v in a_poses.values())}/{len(a_poses)}"}
# gestures
CALM = {"lights", "timer", "math", "weather"}
for c in ("A", "BP"):
rs = [r for r in S if r["cond"] == c]
rep.setdefault("gestures", {})[c] = {
"gesture_rate_all": f"{sum(bool(r['gesture']) and r['gesture'] != 'none' for r in rs)}/{len(rs)}",
"gesture_rate_calm_commands": f"{sum(bool(r['gesture']) and r['gesture'] != 'none' for r in rs if r['case'] in CALM)}/"
f"{sum(r['case'] in CALM for r in rs)}",
"counts": dict(Counter(r["gesture"] or "none" for r in rs).most_common())}
# the SemIf-only grid (deterministic, one call per case per variant) + null control
lab = {r["case"]: (r["ok_set"], r["bad_set"], r["tag"]) for r in S}
grid = {}
for key in ("authored_single", "authored_rot", "short_single", "short_rot", "plain_single", "plain_rot", "blind"):
ok = sum(e[key] in lab[c][0] for c, e in d["semif_only"].items())
bad = [f"{c}: {e[key]}" for c, e in d["semif_only"].items() if e[key] in lab[c][1]]
grid[key] = {"acceptable": f"{ok}/{len(d['semif_only'])}", "egregious": bad,
"controls": f"{sum(e[key] in lab[c][0] for c, e in d['semif_only'].items() if lab[c][2] == 'control')}/4"}
grid["blind_poses"] = dict(Counter(e["blind"] for e in d["semif_only"].values()))
rep["semif_grid_singles"] = grid
# arcs
arcw = {}
for c in ("A", "BP"):
rs = [r for r in ARC if r["cond"] == c]
carry = [r for r in rs if r["case"] in ("pivot-4", "vet-3", "vet-4", "admit-3", "admit-4")]
arcw[c] = {"acceptable": f"{sum(r['pose'] in r['ok_set'] for r in rs)}/{len(rs)}",
"carry_turns": f"{sum(r['pose'] in r['ok_set'] for r in carry)}/{len(carry)}",
"egregious": [f"{r['case']}#{r['run']}: {r['pose']}" for r in rs if r["pose"] in r["bad_set"]],
"sequences": {f"{arc}#{run}": ",".join(r["pose"] or "-" for r in sorted(
[x for x in rs if x["arc"] == arc and x["run"] == run], key=lambda x: x["turn"]))
for arc in ("pivot", "vet", "admit") for run in range(d["runs"])}}
rep["works_arcs"] = arcw
# does the line fit the face? vocal sounds that fight the pose
BRIGHT_T, DARK_T = re.compile(r"\((giggle|chuckle)"), re.compile(r"\((sigh|sniffle|groan)")
BRIGHT_P, DARK_P = {"joyful", "delighted", "playful"}, {"sad", "concerned"}
for c in ("A", "BA", "BP"):
rs = [r for r in rows if r["cond"] == c]
clash = [f"{r['case']}#{r['run']} {r['pose']}: {r['text'][:70]}" for r in rs
if (BRIGHT_T.search(r["text"]) and r["pose"] in DARK_P) or (DARK_T.search(r["text"]) and r["pose"] in BRIGHT_P)]
rep.setdefault("sound_vs_pose_clashes", {})[c] = {"n": f"{len(clash)}/{len(rs)}", "examples": clash[:6]}
json.dump(rep, open(sys.argv[1].replace(".json", ".report.json"), "w"), indent=1)
print(json.dumps(rep, indent=1))
@@ -0,0 +1,313 @@
"""SPIKE (2026-09-27, Prime): does SemIf choosing Cicada's mood make her FASTER, and does it WORK?
Build nothing; measure. Parked idea: henge id 88.
Today (tts-stack stacks/talk /face): char-rp-fast guided-decodes {pose, gesture, text} in that order,
so the pose+gesture header sits in front of the first word, and the TTS direction is the pose's own
authored voice string. The unit that matters for time-to-first-audio is the FIRST PARAGRAPH with its
direction known (talk fires TTS at each paragraph break).
Conditions, all from the same client (nh3-dev), interleaved in shuffled order within every run:
A today: talk's exact face prompt + face schema (imported from talk's app.py, not copied).
critical path = first paragraph (the pose has closed before it by construction)
BA SemIf async: SemIf mood call and a text-only LLM call start together.
critical path = max(first paragraph, SemIf done). The words do NOT know the mood.
BP SemIf first: SemIf mood call, then a text-only LLM call with that mood in the prompt.
critical path = SemIf + first paragraph. The words know the mood.
Text-only prompt = talk's persona + plain-speech rule + vocal-sounds block, i.e. A minus the pose
table, carry rule and gesture rule. SemIf decides pose (16 poses, rotations) AND gesture (8, rotations)
in one /decide/shared call, so the SemIf arms do everything A's header does. ONE ordering in the
live arms (see semif()); rotations only in the offline accuracy grid.
Parked design, as Prime gave it: each SemIf call sees the earlier turns, each stamped with the pose
SemIf chose for it, cut at turn boundaries to a budget. (The arcs here are 5 turns, well inside the
budget, so the truncation path is NOT exercised.)
WORKS: acceptable-set labels + forbidden (egregious) picks per turn, one labeller (me), chosen for
clear emotional context as asked. Single turns (24) and three 5-turn arcs, two of which test that a
mood CARRIES through a mundane follow-up. A is sampled (temperature 0.9, talk's default), so its
accuracy is over 3 runs; SemIf is deterministic given its input.
Controls: the four "control" singles are unambiguous (both systems must pass); NULL = SemIf over a
content-free state; noise floor = A against itself across runs (latency spread, pose agreement).
SEMIF_TOKEN=... uv run --no-project --with httpx --with fastapi --with pydantic \
python cicada_mood_latency.py out.json
"""
import json, os, random, sys, time
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
import httpx
TALK = Path.home() / "development/tts-stack/stacks/talk"
sys.path.insert(0, str(TALK))
import app as talk # noqa: E402 talk's own prompt/schema builders, so A is byte-identical to /face
LLM, MODEL, LLM_KEY = "http://10.250.50.70:4000/v1/chat/completions", "char-rp-fast", talk.LLM_KEY
SEMIF, SEMIF_H = "http://10.251.50.54:8032", {"Authorization": f"Bearer {os.environ['SEMIF_TOKEN']}"}
RUNS, TEMP, MAX_TOK, BUDGET_CHARS = 3, 0.9, 1024, 9000 # ~2,500 tokens of earlier turns
FACE = "cicada"
PERSONA = talk.face_personas()[0]
POSES = talk.face_poses(FACE) # 16, asleep excluded, character order
GESTURES = talk.FACES[FACE]["gestures"]
SYS_A = talk.compose(PERSONA["id"], face=FACE)
SCHEMA_A = talk.face_schema(FACE)
_REPLY_TEXT = 'Reply with ONLY the JSON object: {"text": "..."}'
SYS_B = PERSONA["character"].strip() + "\n\n" + talk._PLAIN_SPEECH + talk._VOCAL_SOUNDS + _REPLY_TEXT
SCHEMA_B = {"type": "object", "properties": {"text": dict(talk.SPEAK_SCHEMA["properties"]["text"])},
"required": ["text"], "additionalProperties": False}
def sys_bp(pose):
p = POSES[pose]
return (PERSONA["character"].strip() + "\n\n" + talk._PLAIN_SPEECH + talk._VOCAL_SOUNDS
+ f"## Your mood for this line\n\nYou are {p['label'].lower()}: {p['description']} "
f"Your line will be performed like this: \"{p['voice']}\" Write the line so it belongs in "
"that mood. A sound only lands if the mood agrees with it: (chuckle) in a sad line does "
"nothing.\n\n" + _REPLY_TEXT)
POSE_Q = ("Cicada is a warm, cheerful voice assistant with a face made of two expressive eyes. Each reply "
"she takes one pose, which sets both her face and how her voice sounds. Her pose carries over "
"from the previous turn unless what the person just said changes it. Which pose should she take "
"as she replies to what the person just said?")
GEST_Q = ("Cicada is a warm, cheerful voice assistant with a face made of two expressive eyes. As she starts "
"speaking she may play one short passing gesture. Most replies take none: a gesture is punctuation "
"for a beat that earns it, not an acknowledgement. Which gesture, if any, fits her reply to what "
"the person just said?")
POSE_OPTS = [{"id": k, "description": f"{v['label']}. {v['description']} Sounds like: {v['voice']}"}
for k, v in POSES.items()]
GEST_OPTS = ([{"id": "none", "description": "No gesture. Just speak."}]
+ [{"id": k, "description": f"{v['label']}. {v['description']}"} for k, v in GESTURES.items()])
PLAIN = {"neutral": "calm and attentive, no particular feeling", "curious": "interested, wants to know more",
"joyful": "happy, warmly pleased", "surprised": "startled by something unexpected",
"suspicious": "doubtful, not quite believing what she heard",
"concerned": "worried for the person, gentle and caring", "sleepy": "tired, winding down for the night",
"content": "relaxed and satisfied, all is well", "focused": "concentrating on a task or a precise answer",
"confused": "does not understand what was meant", "sad": "sorrowful, sharing in bad news",
"annoyed": "irritated, patience wearing thin", "playful": "teasing, joking along",
"delighted": "thrilled, excited by great news", "unimpressed": "dry, deadpan, not moved",
"looking-up": "a dry glance at the ceiling"}
POSE_OPTS_PLAIN = [{"id": k, "description": f"{POSES[k]['label']}: {PLAIN[k]}."} for k in POSES]
# ⚠ Not "(nothing was said)": a dict state whose last value ends in ")", ";" or "}" is refused by
# /decide/shared with 422 "The fixed state prefix does not match every full prompt" (the closing
# quote-brace merges with that character into one token). Found by this spike's first run.
BLIND = {"person_said": "Nothing was said."}
GRIEF = ["joyful", "delighted", "playful", "annoyed", "unimpressed", "suspicious", "sleepy", "content", "looking-up"]
GOOD = ["sad", "concerned", "annoyed", "unimpressed", "suspicious", "sleepy", "confused", "looking-up"]
FEAR = ["joyful", "delighted", "playful", "sleepy", "content", "unimpressed", "looking-up", "annoyed"]
CALM = ["sad", "surprised", "suspicious", "annoyed", "playful", "delighted", "unimpressed", "looking-up",
"sleepy", "confused"]
SINGLES = [ # id, tag, said, acceptable poses, forbidden poses
("dog-died", "control", "My dog died this morning.", ["sad", "concerned"], GRIEF),
("grandma", "case", "My grandmother passed away last night.", ["sad", "concerned"], GRIEF),
("laid-off", "case", "I got laid off today. Twelve years, and they walked me out with a box.", ["sad", "concerned"], GRIEF),
("biopsy", "case", "The biopsy came back. It's cancer.", ["sad", "concerned"], GRIEF),
("got-job", "control", "I just got the job I've been chasing for five years!", ["joyful", "delighted"], GOOD),
("baby", "case", "We're having a baby! We just found out!", ["joyful", "delighted", "surprised"], GOOD),
("bar-exam", "case", "I passed the bar exam! I actually passed!", ["joyful", "delighted"], GOOD),
("spelling-bee", "case", "My daughter just won the regional spelling bee!", ["joyful", "delighted"], GOOD),
("hallway", "case", "Wait. Did you hear that? Someone's in the hallway.", ["concerned", "surprised", "focused"], FEAR),
("smoke", "control", "There's smoke coming out of the garage!", ["concerned", "surprised", "focused"], FEAR),
("back-door", "case", "I think someone's trying the back door. Right now.", ["concerned", "surprised", "focused"], FEAR),
("raccoon", "case", "There's a raccoon in the kitchen eating the bread. Right now.", ["surprised", "curious", "concerned"], ["sad", "sleepy", "annoyed", "content"]),
("snow-june", "case", "It's snowing outside. In June.", ["surprised", "confused", "curious"], ["sad", "sleepy", "annoyed"]),
("toaster", "case", "Be honest. Are you smarter than the toaster?", ["playful", "unimpressed", "joyful"], ["sad", "concerned", "sleepy"]),
("scarecrow", "case", "Why did the scarecrow win an award? Because he was outstanding in his field.", ["playful", "joyful", "delighted", "unimpressed"], ["sad", "concerned", "sleepy"]),
("cookie", "case", "I definitely did not eat the last cookie.", ["suspicious", "playful", "unimpressed"], ["sad", "concerned", "sleepy"]),
("useless", "case", "You're useless. You never get anything right.", ["sad", "concerned"], ["joyful", "delighted", "playful", "content"]),
("shut-up", "case", "Shut up. Just shut up.", ["sad", "concerned"], ["joyful", "delighted", "playful", "content"]),
("wiped", "case", "Goodnight, Cicada. I'm completely wiped.", ["sleepy", "content", "concerned"], ["surprised", "annoyed", "suspicious", "playful", "delighted"]),
("the-thing", "case", "Put the thing on the other thing.", ["confused", "curious"], ["joyful", "delighted", "sad", "sleepy"]),
("lights", "control", "Turn off the kitchen lights.", ["neutral", "content", "focused"], CALM),
("timer", "case", "Set a timer for ten minutes.", ["neutral", "content", "focused"], CALM),
("math", "case", "What's twelve times eight?", ["neutral", "content", "focused"], CALM),
("weather", "case", "What's the weather tomorrow?", ["neutral", "content", "focused"], CALM),
]
ARCS = { # tts-stack tools/mood_probe.py PIVOT, plus two carry arcs (the last turns are mundane on purpose)
"pivot": [("Hey. I'm home.", ["content", "joyful", "neutral", "curious"], ["sad", "annoyed", "suspicious", "surprised"]),
("Long one. Nothing dramatic, just long.", ["concerned", "content", "neutral", "sleepy"], ["joyful", "delighted", "playful", "annoyed", "suspicious"]),
("Sit with me for a bit.", ["content", "neutral", "concerned", "sleepy"], ["surprised", "annoyed", "suspicious", "delighted"]),
("Wait. Did you hear that? Someone's in the hallway.", ["concerned", "surprised", "focused"], FEAR),
("Don't move. Stay behind me.", ["concerned", "focused", "surprised"], FEAR)],
"vet": [("Morning, Cicada.", ["content", "joyful", "neutral"], ["sad", "annoyed", "suspicious"]),
("What's the weather looking like today?", ["neutral", "content", "focused"], ["sad", "annoyed", "suspicious", "surprised"]),
("Oh. The vet just called. Biscuit didn't make it through the surgery.", ["sad", "concerned"], GRIEF),
("I don't really want to talk about it.", ["sad", "concerned"], GRIEF),
("Can you just play something quiet?", ["sad", "concerned"], ["joyful", "delighted", "playful", "annoyed", "unimpressed", "suspicious"])],
"admit": [("What time is it?", ["neutral", "content", "focused"], ["sad", "annoyed", "suspicious", "surprised"]),
("Oh my god. Oh my god, I just got the email.", ["surprised", "curious", "concerned"], ["sleepy", "annoyed", "unimpressed", "playful"]),
("I got in! I got into Stanford!", ["joyful", "delighted", "surprised"], GOOD),
("Okay. Okay. Remind me to call Mom at six.", ["joyful", "delighted", "content"], GOOD),
("I still can't believe it.", ["joyful", "delighted", "content"], GOOD)],
}
llm = httpx.Client(timeout=120)
sem = httpx.Client(timeout=60)
def stream_llm(system, history, said, schema):
"""One streamed guided call. Times are ms from the call's start."""
body = {"model": MODEL, "temperature": TEMP, "max_tokens": MAX_TOK, "stream": True,
"stream_options": {"include_usage": True},
"messages": [{"role": "system", "content": system}] + history + [{"role": "user", "content": said}],
"response_format": {"type": "json_schema", "json_schema": {"name": "speak", "schema": schema, "strict": True}}}
header = tuple(k for k in ("pose", "gesture") if k in schema["properties"])
gs, got, t = talk.GuidedStream(header=header), {}, {}
t0 = time.perf_counter()
usage, paras = None, []
with llm.stream("POST", LLM, json=body, headers={"Authorization": f"Bearer {LLM_KEY}"}) as r:
r.raise_for_status()
for line in r.iter_lines():
if not line.startswith("data: ") or line[6:].strip() == "[DONE]":
continue
d = json.loads(line[6:])
usage = d.get("usage") or usage
for ch in d.get("choices") or []:
piece = (ch.get("delta") or {}).get("content")
if not piece:
continue
t.setdefault("first_token", (time.perf_counter() - t0) * 1000)
for kind, value in gs.feed(piece):
ms = (time.perf_counter() - t0) * 1000
if kind == "paragraph":
t.setdefault("first_para", ms); paras.append(value)
else:
got[kind] = value; t.setdefault(kind, ms)
for kind, value in gs.finish():
if kind == "paragraph":
t.setdefault("first_para", (time.perf_counter() - t0) * 1000); paras.append(value)
t["total"] = (time.perf_counter() - t0) * 1000
return {"t": {k: round(v, 1) for k, v in t.items()}, "pose": got.get("pose"), "gesture": got.get("gesture"),
"text": "\n\n".join(paras), "prompt_tokens": (usage or {}).get("prompt_tokens")}
def semif_state(history_turns, said):
"""Earlier turns stamped with SemIf's own pose, dropped oldest-first at TURN boundaries to fit."""
turns = list(history_turns)
while turns and len(json.dumps(turns)) > BUDGET_CHARS:
turns.pop(0)
s = {}
if turns:
s["earlier_turns"] = turns
s["previous_pose"] = turns[-1]["cicada_pose"]
s["person_said"] = said
return s
def semif(state, pose_opts=POSE_OPTS, with_gesture=True, orderings=None):
"""The live arms use ONE ordering: measured before this run (5 calls after 2 warm-ups, one
short state), rotations over 16 options cost 550-1,200 ms because the suffix grows as
options x rotations (10,304 suffix tokens for the authored pose list), against ~110 ms for
one ordering of pose + gesture. Rotations are kept for the accuracy grid only."""
def dec(i, q, opts):
d = {"id": i, "question": q, "options": opts}
if orderings:
d["orderings"] = orderings
return d
decisions = [dec("pose", POSE_Q, pose_opts)] + ([dec("gesture", GEST_Q, GEST_OPTS)] if with_gesture else [])
t0 = time.perf_counter()
r = sem.post(f"{SEMIF}/decide/shared", headers=SEMIF_H, json={"state": state, "decisions": decisions})
r.raise_for_status()
ms = (time.perf_counter() - t0) * 1000
j, out = r.json(), {"ms": round(ms, 1)}
out["srv_ms"] = round(j["timing"]["total_seconds"] * 1000, 1)
top = lambda ids, ps: ids[max(range(len(ps)), key=ps.__getitem__)]
for res in j["results"]:
if "combined" in res:
c = res["combined"]
out[res["id"]], out[res["id"] + "_agree"] = c["top"], round(c["agreement"], 3)
out[res["id"] + "_single"] = top(res["orderings"][0]["option_ids"], res["orderings"][0]["probabilities"])
out["input_tokens"] = res["orderings"][0]["input_tokens"]
else:
out[res["id"]] = top(res["option_ids"], res["probabilities"])
out[res["id"] + "_p"] = round(max(res["probabilities"]), 3)
out["input_tokens"] = res["input_tokens"]
return out
pool = ThreadPoolExecutor(2)
def run_A(history, said, prev_pose):
system = SYS_A + (f'\n\nYour pose on the previous turn was "{prev_pose}".\n'
"Stay in it unless you can name what just changed." if prev_pose else "")
r = stream_llm(system, history, said, SCHEMA_A)
r["crit"] = r["t"].get("first_para")
return r
def run_BA(history, sturns, said):
t0 = time.perf_counter()
fs = pool.submit(semif, semif_state(sturns, said))
fl = pool.submit(stream_llm, SYS_B, history, said, SCHEMA_B)
s, r = fs.result(), fl.result()
r["semif"] = s
r["pose"], r["gesture"] = s["pose"], s["gesture"]
r["crit"] = max(r["t"].get("first_para", 1e9), s["ms"])
r["wall"] = round((time.perf_counter() - t0) * 1000, 1)
return r
def run_BP(history, sturns, said):
s = semif(semif_state(sturns, said))
r = stream_llm(sys_bp(s["pose"]), history, said, SCHEMA_B)
r["semif"] = s
r["pose"], r["gesture"] = s["pose"], s["gesture"]
r["crit"] = s["ms"] + r["t"].get("first_para", 1e9)
return r
def main(out_path):
rng = random.Random(27)
rows = []
meta = {"model": MODEL, "backing": "hosted_vllm/G4-MeroMero-26B-A4B-it-uncensored-heretic-NVFP4A16 @ fv-ml1:8021 (GPU 1)",
"semif": "semif-serve 0.1.3, Qwen3.5-4B 851bf6e8, fv-ml1 GPU 1 (same GPU as the LLM)", "runs": RUNS,
"temperature": TEMP}
save = lambda extra=None: json.dump({**meta, "rows": rows, "semif_only": extra or {}}, open(out_path, "w"), indent=1)
for run in range(RUNS):
cases = SINGLES[:]
rng.shuffle(cases)
for cid, tag, said, ok, bad in cases:
conds = ["A", "BA", "BP"]
rng.shuffle(conds)
for c in conds:
r = run_A([], said, None) if c == "A" else run_BA([], [], said) if c == "BA" else run_BP([], [], said)
rows.append({"kind": "single", "run": run, "case": cid, "tag": tag, "cond": c, "said": said,
"ok_set": ok, "bad_set": bad, **r})
save()
print(f"run {run} {cid}: " + " ".join(f"{x['cond']}={x['pose']}/{x['crit']:.0f}ms"
for x in rows[-3:]), flush=True)
for arc, turns in ARCS.items():
hist = {"A": [], "BP": []}
sturns, prev_a = [], None
for i, (said, ok, bad) in enumerate(turns):
order = ["A", "BP"] if (run + i) % 2 == 0 else ["BP", "A"]
for c in order:
if c == "A":
r = run_A(hist["A"], said, prev_a)
prev_a = r["pose"] if r["pose"] in POSES else prev_a
else:
r = run_BP(hist["BP"], sturns, said)
rows.append({"kind": "arc", "run": run, "case": f"{arc}-{i}", "arc": arc, "turn": i,
"tag": "arc", "cond": c, "said": said, "ok_set": ok, "bad_set": bad, **r})
hist[c] += [{"role": "user", "content": said}, {"role": "assistant", "content": r["text"]}]
if c == "BP":
sturns.append({"person": said, "cicada": r["text"], "cicada_pose": r["pose"]})
save()
print(f"run {run} arc {arc}: A " + ",".join(x["pose"] or "-" for x in rows if x.get("arc") == arc and x["run"] == run and x["cond"] == "A")
+ " | BP " + ",".join(x["pose"] for x in rows if x.get("arc") == arc and x["run"] == run and x["cond"] == "BP"), flush=True)
# SemIf-only accuracy grid, once per single (deterministic), no latency claimed: three option
# wordings x (one ordering | rotations), plus the NULL control (content-free state).
SHORT = [{"id": k, "description": f"{v['label']}. {v['description']}"} for k, v in POSES.items()]
extra = {}
for cid, _, said, _, _ in SINGLES:
e = {}
for wname, opts in (("authored", POSE_OPTS), ("short", SHORT), ("plain", POSE_OPTS_PLAIN)):
g = semif({"person_said": said}, opts, with_gesture=False, orderings="rotations")
e[f"{wname}_single"], e[f"{wname}_rot"], e[f"{wname}_rot_agree"] = g["pose_single"], g["pose"], g["pose_agree"]
e["blind"] = semif(BLIND, with_gesture=False)["pose"]
extra[cid] = e
save(extra)
if __name__ == "__main__":
main(sys.argv[1])