"""Did the adapter move the voice TOWARD the held-out author? A seat-free relative measure. Written for Yarros, since used on Brontë and Hemingway. The author is now a REQUIRED argument rather than a hardcoded string — see the note on `--author` in main(). NOT the frozen adjudication. That needs a control-author panel (to place an absolute band and a hard-negative sister), a seed-to-seed spread, and — for BEAT INCUMBENT — the gen seat, none of which are available here. This answers the smaller, honest question the operator can act on: of the arms generated on ONE harness, which sits closest to the real held-out author, and does the adapter beat the base control? Instrument: Burrows's Delta over CHARACTER BIGRAMS (hence delta_cb). Char bigrams are dominated by function-word morphology and rhythm, not proper nouns, so the rename does not move them. Reference profile is the HELD-OUT (val) split — text no arm was trained on. Each arm's pooled generations are scored against it; lower = closer to the author. Discipline: this is a RELATIVE reading (arms vs each other, same harness), never an absolute-band claim. The A-vs-A floor below is the only thing that makes a between-arm gap meaningful — half-vs-half of the held-out reference gives the distance the metric returns for two samples of the SAME author, so a between-arm gap smaller than that floor is not a finding. ⭐ TWO OPT-IN READS ADDED FOR lv-mccarthy (2026-09-21), both pre-registered in `scripts/mccarthy-corpus/GATE-PREREG.md` before any McCarthy generation existed. Neither runs by default and neither changes a byte of the default output, because the Brontë, Yarros and Hemingway records were written by the default path and must stay reproducible. --secondary-normalised Re-runs the WHOLE analysis with punctuation stripped from the reference and from every arm. McCarthy's corpus measures 0.0 quote marks per 10k words against Hemingway's 838, so "emit no quotation marks" is the single cheapest way to move a char-bigram distance without having learned a sentence. This read is a deliberately CONSERVATIVE LOWER BOUND: it also strips terminal punctuation, and therefore strips sentence-length signal the adapter legitimately learned. It is reported, it is never the verdict. --punct-report Per-arm punctuation density against the reference. This is the check that says whether the PRIMARY read is confounded at all: the eval harness hands the base control the same register prompt, tics included, so if base COMPLIES its quote density sits near the corpus's and the adapter earns no delta for the cheap win. If base IGNORES the instruction, the primary gap is partly punctuation and the caller must say so. The trigger is pre-registered, not chosen here: see PUNCT_CONFOUND_PER_10K. """ from __future__ import annotations import argparse, json, re, sys, statistics as st from collections import Counter from pathlib import Path # ⚠ PRE-REGISTERED, in GATE-PREREG.md, before any McCarthy arm was generated. If the base # control's quote-mark density exceeds this, the control did NOT take the punctuation win # it was instructed to take, the primary delta_cb gap is partly that win, and the # normalised secondary read is promoted to load-bearing. The corpus measures 0.0 per 10k # and Hemingway's measures 838; 100 is the order-of-magnitude line between them. PUNCT_CONFOUND_PER_10K = 100.0 # ⚠⚠ QUOTE MARKS ONLY -- NO APOSTROPHE CHARACTERS. This class shipped 2026-09-21 with # `'` and `’` in it, which made it an APOSTROPHE counter wearing a quote-mark label. On # lv-mccarthy that reported the held-out reference at 121.1 "quote marks" per 10k for a # corpus whose builder ASSERTS 0.0, and it fired the confound trigger on a base arm whose # true quote density is 19.9. The pre-registered trigger names quote marks and anchors its # line to this corpus's 0.0 against Hemingway's 838 -- both quotation-mark counts -- so an # apostrophe-inclusive class does not measure the quantity the rule names. # Verified after the fix: McCarthy held-out ref 0.0 (matches the builder's assertion), # Hemingway held-out ref 694.7 (the documented ~838 scale). See GATE-PREREG.md AMENDMENT 2. _QUOTE_RE = re.compile('["“”«»‹›„]') _PUNCT_RE = re.compile(r"[^\w\s]|_", re.UNICODE) # apostrophes are counted SEPARATELY and never folded into the quote column again. _APOS_RE = re.compile(r"['’‘`]") # a contraction apostrophe is one sitting BETWEEN letters -- `dont` vs `don't` is the tic # the register names, and a possessive or a quote mark is not the same measurement. _CONTRACTION_APOS_RE = re.compile(r"(?<=[A-Za-z])['’](?=[A-Za-z])") _DASH_RE = re.compile("[—–]|--") def identity(text: str) -> str: return text def strip_punct(text: str) -> str: """Remove every punctuation mark, keeping letters, digits and word boundaries. Deliberately blunt. The point is not to isolate one tic but to remove the entire punctuation channel, so that whatever gap survives is carried by words and their morphology alone. Underscore is stripped explicitly because `\\w` keeps it. """ return re.sub(r"\s+", " ", _PUNCT_RE.sub(" ", text)).strip() def bigrams(text: str) -> Counter: t = re.sub(r"\s+", " ", text.lower()) return Counter(t[i:i+2] for i in range(len(t) - 1)) def profile(text: str, keys: list[str]) -> dict: c = bigrams(text); n = sum(c.values()) or 1 return {k: c.get(k, 0) / n for k in keys} def delta(arm_text: str, ref_prof: dict, mu: dict, sd: dict, keys: list[str]) -> float: ap = profile(arm_text, keys) # Burrows's Delta = mean |z(arm) - z(ref)| over the shared feature set return st.mean(abs((ap[k] - mu[k]) / sd[k] - (ref_prof[k] - mu[k]) / sd[k]) for k in keys) def load_arms(evaldir: Path) -> list[tuple[str, list[dict]]]: out = [] for f in sorted(evaldir.glob("voice.*.jsonl")): recs = [json.loads(l) for l in f.read_text(encoding="utf-8").splitlines() if l.strip()] out.append((f.stem.replace("voice.", ""), recs)) return out def density(text: str, pattern: re.Pattern) -> float: w = len(text.split()) or 1 return len(pattern.findall(text)) * 10000.0 / w def punct_report(ref_text: str, arms: list[tuple[str, list[dict]]]) -> None: """Did the base control take the punctuation win the register prompt handed it?""" print("\n PUNCTUATION DENSITY per 10k words -- the confound check, not an axis") print(" (the eval harness drives EVERY arm with the same register prompt, tics included;") print(" a compliant base control earns the adapter no delta_cb for them)") print(f" {'arm':22s} {'quote-marks':>12s} {'all-apos':>10s} " f"{'contraction-apos':>18s} {'dashes':>9s}") rows = [("held-out reference", ref_text)] rows += [(a, "\n".join(r["continuation"] for r in recs)) for a, recs in arms] base_q = None for name, txt in rows: q = density(txt, _QUOTE_RE) print(f" {name:22s} {q:12.1f} {density(txt, _APOS_RE):10.1f} " f"{density(txt, _CONTRACTION_APOS_RE):18.1f} " f"{density(txt, _DASH_RE):9.1f}") if "unadapted" in name: base_q = q if base_q is None: print(" ⚠ no arm name contains `unadapted` -- the control was not identified, so the") print(" pre-registered confound trigger CANNOT be evaluated. This is not a pass.") return if base_q > PUNCT_CONFOUND_PER_10K: print(f"\n ⚠⚠ CONFOUND TRIGGERED: base control quote density {base_q:.1f} > " f"{PUNCT_CONFOUND_PER_10K:.0f} per 10k.") print(" The control's output is punctuation-rich. Where the register prompt NAMES the") print(" punctuation (lv-mccarthy does; lv-hemingway and lv-bronte do not), that means") print(" the control did not take the win it was handed, so part of the primary") print(" delta_cb gap is that win rather than sentence structure, and per the") print(" pre-registration the NORMALISED secondary read becomes load-bearing. Where the") print(" register does NOT name it, this is a description of the corpus, not a defect.") else: print(f"\n [PASS] base control quote density {base_q:.1f} <= " f"{PUNCT_CONFOUND_PER_10K:.0f} per 10k: the control complied with the register,") print(" so the punctuation win is handed to both sides and the primary read stands.") def analyse(ref_text_raw: str, arms: list[tuple[str, list[dict]]], author: str, transform=identity) -> None: ref_text = transform(ref_text_raw) # feature set: the most frequent bigrams in the reference (stable, high-signal) keys = [k for k, _ in bigrams(ref_text).most_common(400)] # mu/sd across the val text split into chunks, for z-scoring words = ref_text.split() chunks = [" ".join(words[i:i+800]) for i in range(0, len(words), 800) if len(words[i:i+800]) > 200] profs = [profile(c, keys) for c in chunks] mu = {k: st.mean(p[k] for p in profs) for k in keys} sd = {k: (st.pstdev(p[k] for p in profs) or 1e-9) for k in keys} ref_prof = profile(ref_text, keys) # SAME-AUTHOR REFERENCE (the target, not a significance threshold): two halves # of the held-out author. A perfect mimic scores about this; you cannot get closer # to the author than the author gets to itself at this sample size. half = len(words) // 2 same_author = delta(" ".join(words[:half]), profile(" ".join(words[half:]), keys), mu, sd, keys) print(f"reference: held-out {author}, {len(words):,} words, {len(chunks)} chunks, {len(keys)} char-bigram features") print(f"same-author target (held-out {author} vs itself): delta_cb = {same_author:.3f}") print(f" -> the floor of what any arm could reach; lower is more {author}-like, this is the best possible\n") rows = [] for arm, recs in arms: allt = transform("\n".join(r["continuation"] for r in recs)) d = delta(allt, ref_prof, mu, sd, keys) # within-arm sampling spread = the REAL noise floor for a between-arm gap: # split by seed and score each subset; the range is this metric's variance # at this sample size, measured rather than assumed. by_seed = {} for r in recs: by_seed.setdefault(r["seed"], []).append(r["continuation"]) seed_ds = [delta(transform("\n".join(v)), ref_prof, mu, sd, keys) for v in by_seed.values() if len(v) > 2] spread = (max(seed_ds) - min(seed_ds)) if len(seed_ds) > 1 else float("nan") rows.append((arm, d, len(allt.split()), seed_ds, spread)) print(" arm delta_cb per-seed [words]") for arm, d, w, sd_, spread in sorted(rows, key=lambda x: x[1]): seeds = " ".join(f"{x:.3f}" for x in sd_) print(f" {arm:20s} {d:.3f} ({seeds}) [{w}] spread {spread:.3f}") # ⚠⚠ THE FLOOR IS PAIRWISE, and that is a RULE CHANGE made because the all-arms # rule decided lv-bronte. Measured there: # base-unadapted spread 0.062 # ckpt475 spread 0.092 <- the candidate that shipped # ckpt925 spread 0.251 <- set the floor, on ONE outlier seed # ckpt475's +0.193 was failed by a floor contributed entirely by a THIRD arm nobody # was shipping. Run as base-vs-ckpt475 the floor is 0.092 and the same gap clears at # 2.1x. A candidate's verdict must not depend on which other arms you happened to # generate, so the comparison's floor is the larger of the TWO arms being compared. # The all-arms number is still printed, because lv-bronte's record used it and a # reader comparing the two runs needs both. floors = [r[4] for r in rows if r[4] == r[4]] noise_all = max(floors) if floors else float("nan") spread_of = {r[0]: r[4] for r in rows} base_row = next(((a, d) for a, d, _, _, _ in rows if "unadapted" in a), None) if base_row is None: print(f"\n all-arms noise floor (largest within-arm seed spread): {noise_all:.3f}") print(" ⚠ no arm name contains `unadapted` -- no control identified, no verdict\n") return base_arm, base = base_row print(f"\n all-arms noise floor (largest within-arm seed spread, lv-bronte's rule): {noise_all:.3f}") print(f" PAIRWISE floor is the verdict: max(spread(candidate), spread({base_arm}) = " f"{spread_of[base_arm]:.3f})\n") print(f" vs {base_arm} control (positive gap = moved toward {author}):") for arm, d, _, _, _ in sorted(rows, key=lambda x: x[1]): if arm == base_arm: continue gap = base - d pair_floor = max(spread_of[arm], spread_of[base_arm]) verdict = (f"MOVED toward {author} ({gap / pair_floor:.1f}x the pairwise floor " f"{pair_floor:.3f})" if gap > pair_floor else f"within the pairwise floor {pair_floor:.3f} -- NOT a finding") flag = "" if (gap > noise_all) == (gap > pair_floor) else " <- the two rules DISAGREE" print(f" {arm:20s} {gap:+.3f} ({verdict}){flag}") ordered = [a for a, *_ in sorted(rows, key=lambda x: x[1])] print(f"\n ordering: {' < '.join(ordered)} (lower = more {author}-like)") # ⚠ This used to assert "one seed-pair per arm" and claim corroboration from a # "Base < Instruct" held-out loss ordering. Both were Yarros-run facts hardcoded # as if they were properties of the instrument: by lv-bronte every arm carried # four seeds, and no Base-vs-Instruct comparison was in the run at all. Report # what this run actually has instead of a remembered one. nseeds = sorted({len(r[3]) for r in rows}) print(f" ⚠ RELATIVE reading on one harness: {nseeds if len(nseeds) > 1 else nseeds[0]} " f"seed group(s) per arm, scored against this corpus's own held-out split. " f"It is not an absolute-band claim and corroborates nothing on its own.") def main() -> int: # ⚠ --author IS REQUIRED, and that is the fix for a defect this script shipped with. # The reference label was hardcoded "Yarros". Run against Brontë it printed # "reference: held-out Yarros" over Brontë's numbers, and that output is now sitting # in a committed artifact saying the wrong author. A default would have kept the # silent-wrong-label failure and only moved it; naming the author is one word at the # call site and the label can no longer disagree with the data. ap = argparse.ArgumentParser() ap.add_argument("corpus", help="renamed corpus dir containing copies/ with split=val records") ap.add_argument("evaldir", help="dir of voice..jsonl; the control arm's name must " "contain the substring `unadapted`") ap.add_argument("--author", required=True, help="reference author label, e.g. Hemingway. Required: see above.") ap.add_argument("--secondary-normalised", action="store_true", help="ALSO run the whole analysis with punctuation stripped from the " "reference and every arm. A conservative LOWER BOUND on the voice " "gain, reported alongside; it never overturns the primary verdict.") ap.add_argument("--punct-report", action="store_true", help="ALSO print per-arm punctuation density vs the reference, and " "evaluate the pre-registered base-control confound trigger.") a = ap.parse_args() corpus = Path(a.corpus) evaldir = Path(a.evaldir) author = a.author # reference = held-out val text val = [] for p in sorted((corpus / "copies").glob("*.jsonl")): for l in p.read_text(encoding="utf-8").splitlines(): r = json.loads(l) if r.get("split") == "val": val.append(r["text"]) # dedup identical val chapters across copies (renaming aside, the same chapter recurs) ref_text = "\n".join(dict.fromkeys(val)) arms = load_arms(evaldir) analyse(ref_text, arms, author, identity) if a.punct_report: punct_report(ref_text, arms) if a.secondary_normalised: print("\n" + "=" * 78) print("SECONDARY READ -- PUNCTUATION STRIPPED. Pre-registered, REPORTED, NOT THE VERDICT.") print("Every punctuation mark is removed from the reference and from every arm, so a") print("gap that survives here is carried by words rather than by marks. It is a LOWER") print("BOUND and not a better measurement: stripping terminal punctuation also strips") print("sentence-length signal the adapter legitimately learned. Read it as `at least") print("this much of the primary gap is not the punctuation trick`.") print("=" * 78 + "\n") analyse(ref_text, arms, author, strip_punct) return 0 if __name__ == "__main__": sys.exit(main())