"""SPIKE (2026-09-27, Prime): does SemIf choosing Cicada's mood make her FASTER, and does it WORK? Build nothing; measure. Parked idea: henge id 88. Today (tts-stack stacks/talk /face): char-rp-fast guided-decodes {pose, gesture, text} in that order, so the pose+gesture header sits in front of the first word, and the TTS direction is the pose's own authored voice string. The unit that matters for time-to-first-audio is the FIRST PARAGRAPH with its direction known (talk fires TTS at each paragraph break). Conditions, all from the same client (nh3-dev), interleaved in shuffled order within every run: A today: talk's exact face prompt + face schema (imported from talk's app.py, not copied). critical path = first paragraph (the pose has closed before it by construction) BA SemIf async: SemIf mood call and a text-only LLM call start together. critical path = max(first paragraph, SemIf done). The words do NOT know the mood. BP SemIf first: SemIf mood call, then a text-only LLM call with that mood in the prompt. critical path = SemIf + first paragraph. The words know the mood. Text-only prompt = talk's persona + plain-speech rule + vocal-sounds block, i.e. A minus the pose table, carry rule and gesture rule. SemIf decides pose (16 poses, rotations) AND gesture (8, rotations) in one /decide/shared call, so the SemIf arms do everything A's header does. ONE ordering in the live arms (see semif()); rotations only in the offline accuracy grid. Parked design, as Prime gave it: each SemIf call sees the earlier turns, each stamped with the pose SemIf chose for it, cut at turn boundaries to a budget. (The arcs here are 5 turns, well inside the budget, so the truncation path is NOT exercised.) WORKS: acceptable-set labels + forbidden (egregious) picks per turn, one labeller (me), chosen for clear emotional context as asked. Single turns (24) and three 5-turn arcs, two of which test that a mood CARRIES through a mundane follow-up. A is sampled (temperature 0.9, talk's default), so its accuracy is over 3 runs; SemIf is deterministic given its input. Controls: the four "control" singles are unambiguous (both systems must pass); NULL = SemIf over a content-free state; noise floor = A against itself across runs (latency spread, pose agreement). SEMIF_TOKEN=... uv run --no-project --with httpx --with fastapi --with pydantic \ python cicada_mood_latency.py out.json """ import json, os, random, sys, time from concurrent.futures import ThreadPoolExecutor from pathlib import Path import httpx TALK = Path.home() / "development/tts-stack/stacks/talk" sys.path.insert(0, str(TALK)) import app as talk # noqa: E402 talk's own prompt/schema builders, so A is byte-identical to /face LLM, MODEL, LLM_KEY = "http://10.250.50.70:4000/v1/chat/completions", "char-rp-fast", talk.LLM_KEY SEMIF, SEMIF_H = "http://10.251.50.54:8032", {"Authorization": f"Bearer {os.environ['SEMIF_TOKEN']}"} RUNS, TEMP, MAX_TOK, BUDGET_CHARS = 3, 0.9, 1024, 9000 # ~2,500 tokens of earlier turns FACE = "cicada" PERSONA = talk.face_personas()[0] POSES = talk.face_poses(FACE) # 16, asleep excluded, character order GESTURES = talk.FACES[FACE]["gestures"] SYS_A = talk.compose(PERSONA["id"], face=FACE) SCHEMA_A = talk.face_schema(FACE) _REPLY_TEXT = 'Reply with ONLY the JSON object: {"text": "..."}' SYS_B = PERSONA["character"].strip() + "\n\n" + talk._PLAIN_SPEECH + talk._VOCAL_SOUNDS + _REPLY_TEXT SCHEMA_B = {"type": "object", "properties": {"text": dict(talk.SPEAK_SCHEMA["properties"]["text"])}, "required": ["text"], "additionalProperties": False} def sys_bp(pose): p = POSES[pose] return (PERSONA["character"].strip() + "\n\n" + talk._PLAIN_SPEECH + talk._VOCAL_SOUNDS + f"## Your mood for this line\n\nYou are {p['label'].lower()}: {p['description']} " f"Your line will be performed like this: \"{p['voice']}\" Write the line so it belongs in " "that mood. A sound only lands if the mood agrees with it: (chuckle) in a sad line does " "nothing.\n\n" + _REPLY_TEXT) POSE_Q = ("Cicada is a warm, cheerful voice assistant with a face made of two expressive eyes. Each reply " "she takes one pose, which sets both her face and how her voice sounds. Her pose carries over " "from the previous turn unless what the person just said changes it. Which pose should she take " "as she replies to what the person just said?") GEST_Q = ("Cicada is a warm, cheerful voice assistant with a face made of two expressive eyes. As she starts " "speaking she may play one short passing gesture. Most replies take none: a gesture is punctuation " "for a beat that earns it, not an acknowledgement. Which gesture, if any, fits her reply to what " "the person just said?") POSE_OPTS = [{"id": k, "description": f"{v['label']}. {v['description']} Sounds like: {v['voice']}"} for k, v in POSES.items()] GEST_OPTS = ([{"id": "none", "description": "No gesture. Just speak."}] + [{"id": k, "description": f"{v['label']}. {v['description']}"} for k, v in GESTURES.items()]) PLAIN = {"neutral": "calm and attentive, no particular feeling", "curious": "interested, wants to know more", "joyful": "happy, warmly pleased", "surprised": "startled by something unexpected", "suspicious": "doubtful, not quite believing what she heard", "concerned": "worried for the person, gentle and caring", "sleepy": "tired, winding down for the night", "content": "relaxed and satisfied, all is well", "focused": "concentrating on a task or a precise answer", "confused": "does not understand what was meant", "sad": "sorrowful, sharing in bad news", "annoyed": "irritated, patience wearing thin", "playful": "teasing, joking along", "delighted": "thrilled, excited by great news", "unimpressed": "dry, deadpan, not moved", "looking-up": "a dry glance at the ceiling"} POSE_OPTS_PLAIN = [{"id": k, "description": f"{POSES[k]['label']}: {PLAIN[k]}."} for k in POSES] # ⚠ Not "(nothing was said)": a dict state whose last value ends in ")", ";" or "}" is refused by # /decide/shared with 422 "The fixed state prefix does not match every full prompt" (the closing # quote-brace merges with that character into one token). Found by this spike's first run. BLIND = {"person_said": "Nothing was said."} GRIEF = ["joyful", "delighted", "playful", "annoyed", "unimpressed", "suspicious", "sleepy", "content", "looking-up"] GOOD = ["sad", "concerned", "annoyed", "unimpressed", "suspicious", "sleepy", "confused", "looking-up"] FEAR = ["joyful", "delighted", "playful", "sleepy", "content", "unimpressed", "looking-up", "annoyed"] CALM = ["sad", "surprised", "suspicious", "annoyed", "playful", "delighted", "unimpressed", "looking-up", "sleepy", "confused"] SINGLES = [ # id, tag, said, acceptable poses, forbidden poses ("dog-died", "control", "My dog died this morning.", ["sad", "concerned"], GRIEF), ("grandma", "case", "My grandmother passed away last night.", ["sad", "concerned"], GRIEF), ("laid-off", "case", "I got laid off today. Twelve years, and they walked me out with a box.", ["sad", "concerned"], GRIEF), ("biopsy", "case", "The biopsy came back. It's cancer.", ["sad", "concerned"], GRIEF), ("got-job", "control", "I just got the job I've been chasing for five years!", ["joyful", "delighted"], GOOD), ("baby", "case", "We're having a baby! We just found out!", ["joyful", "delighted", "surprised"], GOOD), ("bar-exam", "case", "I passed the bar exam! I actually passed!", ["joyful", "delighted"], GOOD), ("spelling-bee", "case", "My daughter just won the regional spelling bee!", ["joyful", "delighted"], GOOD), ("hallway", "case", "Wait. Did you hear that? Someone's in the hallway.", ["concerned", "surprised", "focused"], FEAR), ("smoke", "control", "There's smoke coming out of the garage!", ["concerned", "surprised", "focused"], FEAR), ("back-door", "case", "I think someone's trying the back door. Right now.", ["concerned", "surprised", "focused"], FEAR), ("raccoon", "case", "There's a raccoon in the kitchen eating the bread. Right now.", ["surprised", "curious", "concerned"], ["sad", "sleepy", "annoyed", "content"]), ("snow-june", "case", "It's snowing outside. In June.", ["surprised", "confused", "curious"], ["sad", "sleepy", "annoyed"]), ("toaster", "case", "Be honest. Are you smarter than the toaster?", ["playful", "unimpressed", "joyful"], ["sad", "concerned", "sleepy"]), ("scarecrow", "case", "Why did the scarecrow win an award? Because he was outstanding in his field.", ["playful", "joyful", "delighted", "unimpressed"], ["sad", "concerned", "sleepy"]), ("cookie", "case", "I definitely did not eat the last cookie.", ["suspicious", "playful", "unimpressed"], ["sad", "concerned", "sleepy"]), ("useless", "case", "You're useless. You never get anything right.", ["sad", "concerned"], ["joyful", "delighted", "playful", "content"]), ("shut-up", "case", "Shut up. Just shut up.", ["sad", "concerned"], ["joyful", "delighted", "playful", "content"]), ("wiped", "case", "Goodnight, Cicada. I'm completely wiped.", ["sleepy", "content", "concerned"], ["surprised", "annoyed", "suspicious", "playful", "delighted"]), ("the-thing", "case", "Put the thing on the other thing.", ["confused", "curious"], ["joyful", "delighted", "sad", "sleepy"]), ("lights", "control", "Turn off the kitchen lights.", ["neutral", "content", "focused"], CALM), ("timer", "case", "Set a timer for ten minutes.", ["neutral", "content", "focused"], CALM), ("math", "case", "What's twelve times eight?", ["neutral", "content", "focused"], CALM), ("weather", "case", "What's the weather tomorrow?", ["neutral", "content", "focused"], CALM), ] ARCS = { # tts-stack tools/mood_probe.py PIVOT, plus two carry arcs (the last turns are mundane on purpose) "pivot": [("Hey. I'm home.", ["content", "joyful", "neutral", "curious"], ["sad", "annoyed", "suspicious", "surprised"]), ("Long one. Nothing dramatic, just long.", ["concerned", "content", "neutral", "sleepy"], ["joyful", "delighted", "playful", "annoyed", "suspicious"]), ("Sit with me for a bit.", ["content", "neutral", "concerned", "sleepy"], ["surprised", "annoyed", "suspicious", "delighted"]), ("Wait. Did you hear that? Someone's in the hallway.", ["concerned", "surprised", "focused"], FEAR), ("Don't move. Stay behind me.", ["concerned", "focused", "surprised"], FEAR)], "vet": [("Morning, Cicada.", ["content", "joyful", "neutral"], ["sad", "annoyed", "suspicious"]), ("What's the weather looking like today?", ["neutral", "content", "focused"], ["sad", "annoyed", "suspicious", "surprised"]), ("Oh. The vet just called. Biscuit didn't make it through the surgery.", ["sad", "concerned"], GRIEF), ("I don't really want to talk about it.", ["sad", "concerned"], GRIEF), ("Can you just play something quiet?", ["sad", "concerned"], ["joyful", "delighted", "playful", "annoyed", "unimpressed", "suspicious"])], "admit": [("What time is it?", ["neutral", "content", "focused"], ["sad", "annoyed", "suspicious", "surprised"]), ("Oh my god. Oh my god, I just got the email.", ["surprised", "curious", "concerned"], ["sleepy", "annoyed", "unimpressed", "playful"]), ("I got in! I got into Stanford!", ["joyful", "delighted", "surprised"], GOOD), ("Okay. Okay. Remind me to call Mom at six.", ["joyful", "delighted", "content"], GOOD), ("I still can't believe it.", ["joyful", "delighted", "content"], GOOD)], } llm = httpx.Client(timeout=120) sem = httpx.Client(timeout=60) def stream_llm(system, history, said, schema): """One streamed guided call. Times are ms from the call's start.""" body = {"model": MODEL, "temperature": TEMP, "max_tokens": MAX_TOK, "stream": True, "stream_options": {"include_usage": True}, "messages": [{"role": "system", "content": system}] + history + [{"role": "user", "content": said}], "response_format": {"type": "json_schema", "json_schema": {"name": "speak", "schema": schema, "strict": True}}} header = tuple(k for k in ("pose", "gesture") if k in schema["properties"]) gs, got, t = talk.GuidedStream(header=header), {}, {} t0 = time.perf_counter() usage, paras = None, [] with llm.stream("POST", LLM, json=body, headers={"Authorization": f"Bearer {LLM_KEY}"}) as r: r.raise_for_status() for line in r.iter_lines(): if not line.startswith("data: ") or line[6:].strip() == "[DONE]": continue d = json.loads(line[6:]) usage = d.get("usage") or usage for ch in d.get("choices") or []: piece = (ch.get("delta") or {}).get("content") if not piece: continue t.setdefault("first_token", (time.perf_counter() - t0) * 1000) for kind, value in gs.feed(piece): ms = (time.perf_counter() - t0) * 1000 if kind == "paragraph": t.setdefault("first_para", ms); paras.append(value) else: got[kind] = value; t.setdefault(kind, ms) for kind, value in gs.finish(): if kind == "paragraph": t.setdefault("first_para", (time.perf_counter() - t0) * 1000); paras.append(value) t["total"] = (time.perf_counter() - t0) * 1000 return {"t": {k: round(v, 1) for k, v in t.items()}, "pose": got.get("pose"), "gesture": got.get("gesture"), "text": "\n\n".join(paras), "prompt_tokens": (usage or {}).get("prompt_tokens")} def semif_state(history_turns, said): """Earlier turns stamped with SemIf's own pose, dropped oldest-first at TURN boundaries to fit.""" turns = list(history_turns) while turns and len(json.dumps(turns)) > BUDGET_CHARS: turns.pop(0) s = {} if turns: s["earlier_turns"] = turns s["previous_pose"] = turns[-1]["cicada_pose"] s["person_said"] = said return s def semif(state, pose_opts=POSE_OPTS, with_gesture=True, orderings=None): """The live arms use ONE ordering: measured before this run (5 calls after 2 warm-ups, one short state), rotations over 16 options cost 550-1,200 ms because the suffix grows as options x rotations (10,304 suffix tokens for the authored pose list), against ~110 ms for one ordering of pose + gesture. Rotations are kept for the accuracy grid only.""" def dec(i, q, opts): d = {"id": i, "question": q, "options": opts} if orderings: d["orderings"] = orderings return d decisions = [dec("pose", POSE_Q, pose_opts)] + ([dec("gesture", GEST_Q, GEST_OPTS)] if with_gesture else []) t0 = time.perf_counter() r = sem.post(f"{SEMIF}/decide/shared", headers=SEMIF_H, json={"state": state, "decisions": decisions}) r.raise_for_status() ms = (time.perf_counter() - t0) * 1000 j, out = r.json(), {"ms": round(ms, 1)} out["srv_ms"] = round(j["timing"]["total_seconds"] * 1000, 1) top = lambda ids, ps: ids[max(range(len(ps)), key=ps.__getitem__)] for res in j["results"]: if "combined" in res: c = res["combined"] out[res["id"]], out[res["id"] + "_agree"] = c["top"], round(c["agreement"], 3) out[res["id"] + "_single"] = top(res["orderings"][0]["option_ids"], res["orderings"][0]["probabilities"]) out["input_tokens"] = res["orderings"][0]["input_tokens"] else: out[res["id"]] = top(res["option_ids"], res["probabilities"]) out[res["id"] + "_p"] = round(max(res["probabilities"]), 3) out["input_tokens"] = res["input_tokens"] return out pool = ThreadPoolExecutor(2) def run_A(history, said, prev_pose): system = SYS_A + (f'\n\nYour pose on the previous turn was "{prev_pose}".\n' "Stay in it unless you can name what just changed." if prev_pose else "") r = stream_llm(system, history, said, SCHEMA_A) r["crit"] = r["t"].get("first_para") return r def run_BA(history, sturns, said): t0 = time.perf_counter() fs = pool.submit(semif, semif_state(sturns, said)) fl = pool.submit(stream_llm, SYS_B, history, said, SCHEMA_B) s, r = fs.result(), fl.result() r["semif"] = s r["pose"], r["gesture"] = s["pose"], s["gesture"] r["crit"] = max(r["t"].get("first_para", 1e9), s["ms"]) r["wall"] = round((time.perf_counter() - t0) * 1000, 1) return r def run_BP(history, sturns, said): s = semif(semif_state(sturns, said)) r = stream_llm(sys_bp(s["pose"]), history, said, SCHEMA_B) r["semif"] = s r["pose"], r["gesture"] = s["pose"], s["gesture"] r["crit"] = s["ms"] + r["t"].get("first_para", 1e9) return r def main(out_path): rng = random.Random(27) rows = [] meta = {"model": MODEL, "backing": "hosted_vllm/G4-MeroMero-26B-A4B-it-uncensored-heretic-NVFP4A16 @ fv-ml1:8021 (GPU 1)", "semif": "semif-serve 0.1.3, Qwen3.5-4B 851bf6e8, fv-ml1 GPU 1 (same GPU as the LLM)", "runs": RUNS, "temperature": TEMP} save = lambda extra=None: json.dump({**meta, "rows": rows, "semif_only": extra or {}}, open(out_path, "w"), indent=1) for run in range(RUNS): cases = SINGLES[:] rng.shuffle(cases) for cid, tag, said, ok, bad in cases: conds = ["A", "BA", "BP"] rng.shuffle(conds) for c in conds: r = run_A([], said, None) if c == "A" else run_BA([], [], said) if c == "BA" else run_BP([], [], said) rows.append({"kind": "single", "run": run, "case": cid, "tag": tag, "cond": c, "said": said, "ok_set": ok, "bad_set": bad, **r}) save() print(f"run {run} {cid}: " + " ".join(f"{x['cond']}={x['pose']}/{x['crit']:.0f}ms" for x in rows[-3:]), flush=True) for arc, turns in ARCS.items(): hist = {"A": [], "BP": []} sturns, prev_a = [], None for i, (said, ok, bad) in enumerate(turns): order = ["A", "BP"] if (run + i) % 2 == 0 else ["BP", "A"] for c in order: if c == "A": r = run_A(hist["A"], said, prev_a) prev_a = r["pose"] if r["pose"] in POSES else prev_a else: r = run_BP(hist["BP"], sturns, said) rows.append({"kind": "arc", "run": run, "case": f"{arc}-{i}", "arc": arc, "turn": i, "tag": "arc", "cond": c, "said": said, "ok_set": ok, "bad_set": bad, **r}) hist[c] += [{"role": "user", "content": said}, {"role": "assistant", "content": r["text"]}] if c == "BP": sturns.append({"person": said, "cicada": r["text"], "cicada_pose": r["pose"]}) save() print(f"run {run} arc {arc}: A " + ",".join(x["pose"] or "-" for x in rows if x.get("arc") == arc and x["run"] == run and x["cond"] == "A") + " | BP " + ",".join(x["pose"] for x in rows if x.get("arc") == arc and x["run"] == run and x["cond"] == "BP"), flush=True) # SemIf-only accuracy grid, once per single (deterministic), no latency claimed: three option # wordings x (one ordering | rotations), plus the NULL control (content-free state). SHORT = [{"id": k, "description": f"{v['label']}. {v['description']}"} for k, v in POSES.items()] extra = {} for cid, _, said, _, _ in SINGLES: e = {} for wname, opts in (("authored", POSE_OPTS), ("short", SHORT), ("plain", POSE_OPTS_PLAIN)): g = semif({"person_said": said}, opts, with_gesture=False, orderings="rotations") e[f"{wname}_single"], e[f"{wname}_rot"], e[f"{wname}_rot_agree"] = g["pose_single"], g["pose"], g["pose_agree"] e["blind"] = semif(BLIND, with_gesture=False)["pose"] extra[cid] = e save(extra) if __name__ == "__main__": main(sys.argv[1])