diff --git a/docs/diagnostics/fiction_wing_probe.py b/docs/diagnostics/fiction_wing_probe.py index f2782a1..584e0f5 100644 --- a/docs/diagnostics/fiction_wing_probe.py +++ b/docs/diagnostics/fiction_wing_probe.py @@ -24,6 +24,10 @@ search_library is deterministic (no CI needed); reference_knowledge variance is formulation, not floor noise. Residual = bucket-boundary sensitivity at 0.016/0.030 — so probe N>=3-5 times per term and report the bucket DISTRIBUTION, never a single-run point label. +Session protocol: ONE fresh session per query. A reused mimir session returns EMPTY +search_library results after its first turn (Worldtree #391), silently scoring later terms +false-MISS; the retrieval helpers self-session to enforce it. Never hoist the session out. + Config from env (source ratatoskr's env.sh): WORLDTREE_API_URL, WORLDTREE_API_KEY, RATATOSKR_END_USER_ID. No secrets are stored here. @@ -38,6 +42,7 @@ from __future__ import annotations import argparse import json import os +import unicodedata from collections import Counter import httpx @@ -60,9 +65,24 @@ def _bucket(score: float | None) -> str: return "high" if score >= HIGH else "medium" if score >= MEDIUM else "low" +# Curly punctuation the b170 extraction emits (U+2019 etc.) folded to ASCII so a +# possessive-named subject ("Darcy's letter") matches regardless of quote style. +_QUOTE_FOLD = str.maketrans({ + "‘": "'", "’": "'", # noqa: RUF001 - single curly quotes / apostrophe + "“": '"', "”": '"', # double curly quotes + "′": "'", "″": '"', # noqa: RUF001 - primes +}) + + +def _fold(s: str) -> str: + """NFKC-normalize, fold curly quotes/apostrophes to ASCII, lowercase. + NFKC alone does NOT fold U+2019, so the explicit quote-fold is load-bearing.""" + return unicodedata.normalize("NFKC", s or "").translate(_QUOTE_FOLD).lower() + + def _on_target(excerpt: str, keywords: list[str]) -> bool: - ex = (excerpt or "").lower() - return any(k.lower() in ex for k in keywords) + ex = _fold(excerpt) + return any(_fold(k) in ex for k in keywords) def _session(base: str, headers: dict, agent_id: str, end_user: str) -> str: @@ -90,8 +110,14 @@ def _drive(base: str, headers: dict, sid: str, content: str) -> tuple[str | None return tool_query, result -def search_library(base, headers, msid, term, keywords): - """RANKING-clean path: fixed-string search over the mimir librarian. Deterministic.""" +def search_library(base, headers, end_user, term, keywords): + """RANKING-clean path: fixed-string search over the mimir librarian. Deterministic. + + Opens a FRESH mimir session per call — REQUIRED. mimir stops returning + search_library results after the first turn on a reused session (Worldtree #391), + silently scoring every later term a false-MISS; do not hoist the session to the caller. + """ + msid = _session(base, headers, "mimir", end_user) _, res = _drive(base, headers, msid, f"Use search_library to find: {term}") rows = res.get("results", []) if isinstance(res, dict) else [] on = [h for h in rows if isinstance(h, dict) and _on_target(h.get("excerpt", ""), keywords)] @@ -102,8 +128,13 @@ def search_library(base, headers, msid, term, keywords): "excerpt": (top.get("excerpt", "")[:140] if top else None)} -def reference_knowledge(base, headers, dsid, question, keywords): - """Consumer path: Donut reformulates -> capture her tool_query. NOT for ranking numbers.""" +def reference_knowledge(base, headers, end_user, question, keywords): + """Consumer path: Donut reformulates -> capture her tool_query. NOT for ranking numbers. + + Fresh donut session per call (same reuse-degradation guard as search_library, and + it keeps each run an independent first-turn sample rather than a growing conversation). + """ + dsid = _session(base, headers, "ratatoskr:donut", end_user) q, res = _drive(base, headers, dsid, question) hits = res.get("hits", []) if isinstance(res, dict) else [] on = [h for h in hits if isinstance(h, dict) and _on_target(h.get("excerpt", ""), keywords)] @@ -123,13 +154,12 @@ YARDSTICK = [ def run_yardstick(runs: int) -> None: base, headers, end_user = _cfg() - msid = _session(base, headers, "mimir", end_user) print(f"# Fiction-wing ranking yardstick (search_library, {runs} run(s) per name)\n") for full, partial, kw in YARDSTICK: for label, term in (("full ", full), ("part ", partial)): buckets, hits = Counter(), 0 for _ in range(runs): - r = search_library(base, headers, msid, term, kw) + r = search_library(base, headers, end_user, term, kw) buckets[r["bucket"]] += 1 hits += r["hit"] dist = " ".join(f"{b}:{c}" for b, c in buckets.most_common()) @@ -139,17 +169,15 @@ def run_yardstick(runs: int) -> None: def run_term(term: str, keywords: list[str], runs: int) -> None: base, headers, end_user = _cfg() - msid = _session(base, headers, "mimir", end_user) - dsid = _session(base, headers, "ratatoskr:donut", end_user) print(f"# Probe: {term!r} ({runs} run(s))\n") sl_buckets, sl_hits = Counter(), 0 for _ in range(runs): - r = search_library(base, headers, msid, term, keywords) + r = search_library(base, headers, end_user, term, keywords) sl_buckets[r["bucket"]] += 1 sl_hits += r["hit"] print(f" search_library : hit {sl_hits}/{runs} buckets({dict(sl_buckets)})") for _ in range(runs): - rk = reference_knowledge(base, headers, dsid, f"Tell me about the {term}.", keywords) + rk = reference_knowledge(base, headers, end_user, f"Tell me about the {term}.", keywords) print(f" reference_knowledge: conf={rk['confidence']} on_target={rk['on_target']}" f" (donut query: {rk['tool_query']!r})") diff --git a/tests/test_fiction_wing_probe.py b/tests/test_fiction_wing_probe.py new file mode 100644 index 0000000..8915378 --- /dev/null +++ b/tests/test_fiction_wing_probe.py @@ -0,0 +1,33 @@ +"""Unit tests for the fiction-wing probe harness's pure helpers. + +Only the network-free helpers are covered here; the live retrieval paths +(search_library / reference_knowledge) are exercised by running the harness +against a live Worldtree index, not by pytest. +""" +import sys +from pathlib import Path + +_DIAG = Path(__file__).resolve().parent.parent / "docs" / "diagnostics" +sys.path.insert(0, str(_DIAG)) + +from fiction_wing_probe import _on_target # noqa: E402 + +CURLY = "’" # noqa: RUF001 - the b170 extraction's default apostrophe + + +def test_on_target_matches_curly_apostrophe_excerpt(): + # Excerpt stores the curly apostrophe; keyword is ASCII. Must still match. + excerpt = f"Mr. Darcy{CURLY}s letter to Elizabeth explains his conduct." + assert _on_target(excerpt, ["darcy's letter"]) + + +def test_on_target_matches_ascii_excerpt_against_curly_keyword(): + # Symmetric: ASCII excerpt, curly-quoted keyword. Fold both sides. + excerpt = "Mrs. Gardiner's letter arrived the next morning." + assert _on_target(excerpt, [f"gardiner{CURLY}s letter"]) + + +def test_on_target_still_rejects_absent_subject(): + # Negative control: folding must not make unrelated excerpts match. + excerpt = "A passage about dungeons, crawlers, and monsters." + assert not _on_target(excerpt, ["darcy's letter"])