Files
esh-pfi-infrastructure/scripts/r49-corpus/voice_distance.py
T
vh 5558d9c7d3 R49/BabyYarros: voice + beat evaluation tooling and the seat-free delta_cb
Adds the Yarros-side evaluation the training exists to justify: does the adapter
move arbitrary prose toward Yarros, and can the instruct arm still expand a beat
to a paragraph on direction. Yarros-flavoured voice prompts (modern/neutral/
romantasy tiers so any Yarros voice in the modern tier is adapter-attributable,
not prompt-supplied) and a Yarros-register beat SYS on the chat generator.

voice_distance.py is the honest slice of adjudication that needs no seat: Burrows's
Delta over character bigrams against held-out Yarros. Its first cut mis-framed the
noise floor — it used the same-author distance (held-out vs itself) as the
between-arm significance threshold, which is the target, not the threshold. Fixed
to the measured floor: the within-arm seed spread, which is this metric's sampling
variance at this sample size, computed from the two seeds already generated rather
than assumed.

Result on the built corpus, ordering base-125-tuned < instruct-tuned <
base-unadapted, both adapters clearing the 0.046 measured floor (base +0.157,
instruct +0.076), and the ordering corroborating the independent held-out loss
ordering (Base below Instruct). One seed-pair per arm, so it corroborates rather
than settles; the full frozen adjudication still needs a romantasy control panel,
a second seed, and the gen seat for the beat-incumbent leg.
2026-09-11 15:59:41 -07:00

122 lines
6.1 KiB
Python

"""Did the adapter move the voice TOWARD held-out Yarros? A seat-free relative measure.
NOT the frozen adjudication. That needs a romantasy control-author panel (to place an
absolute band and a hard-negative sister), a seed-to-seed spread, and — for BEAT
INCUMBENT — the gen seat, none of which are available here. This answers the smaller,
honest question the operator can act on: of the three arms generated on ONE harness,
which sits closest to real held-out Yarros, and does the adapter beat the base control?
Instrument: Burrows's Delta over CHARACTER BIGRAMS (hence delta_cb). Char bigrams are
dominated by function-word morphology and rhythm, not proper nouns, so the rename does
not move them. Reference profile is the HELD-OUT (val) split — text no arm was trained
on. Each arm's pooled generations are scored against it; lower = closer to Yarros.
Discipline: this is a RELATIVE reading (arms vs each other, same harness), never an
absolute-band claim. The A-vs-A floor below is the only thing that makes a between-arm
gap meaningful — half-vs-half of the held-out reference gives the distance the metric
returns for two samples of the SAME author, so a between-arm gap smaller than that floor
is not a finding.
"""
from __future__ import annotations
import json, re, sys, statistics as st
from collections import Counter
from pathlib import Path
def bigrams(text: str) -> Counter:
t = re.sub(r"\s+", " ", text.lower())
return Counter(t[i:i+2] for i in range(len(t) - 1))
def profile(text: str, keys: list[str]) -> dict:
c = bigrams(text); n = sum(c.values()) or 1
return {k: c.get(k, 0) / n for k in keys}
def delta(arm_text: str, ref_prof: dict, mu: dict, sd: dict, keys: list[str]) -> float:
ap = profile(arm_text, keys)
# Burrows's Delta = mean |z(arm) - z(ref)| over the shared feature set
return st.mean(abs((ap[k] - mu[k]) / sd[k] - (ref_prof[k] - mu[k]) / sd[k]) for k in keys)
def main() -> int:
corpus = Path(sys.argv[1]) # yarros-corpus-renamed (has split=val)
evaldir = Path(sys.argv[2]) # dir of voice.*.jsonl
# reference = held-out val text
val = []
for p in sorted((corpus / "copies").glob("*.jsonl")):
for l in p.read_text(encoding="utf-8").splitlines():
r = json.loads(l)
if r.get("split") == "val":
val.append(r["text"])
# dedup identical val chapters across copies (renaming aside, the same chapter recurs)
ref_text = "\n".join(dict.fromkeys(val))
# feature set: the most frequent bigrams in the reference (stable, high-signal)
keys = [k for k, _ in bigrams(ref_text).most_common(400)]
# mu/sd across the val text split into chunks, for z-scoring
words = ref_text.split()
chunks = [" ".join(words[i:i+800]) for i in range(0, len(words), 800) if len(words[i:i+800]) > 200]
profs = [profile(c, keys) for c in chunks]
mu = {k: st.mean(p[k] for p in profs) for k in keys}
sd = {k: (st.pstdev(p[k] for p in profs) or 1e-9) for k in keys}
ref_prof = profile(ref_text, keys)
# SAME-AUTHOR REFERENCE (the target, not a significance threshold): two halves
# of held-out Yarros. A perfect mimic scores about this; you cannot get closer
# to Yarros than Yarros gets to itself at this sample size.
half = len(words) // 2
same_author = delta(" ".join(words[:half]), profile(" ".join(words[half:]), keys), mu, sd, keys)
print(f"reference: held-out Yarros, {len(words):,} words, {len(chunks)} chunks, {len(keys)} char-bigram features")
print(f"same-author target (held-out Yarros vs itself): delta_cb = {same_author:.3f}")
print(f" -> the floor of what any arm could reach; lower is more Yarros-like, this is the best possible\n")
def arm_texts(f):
return [json.loads(l) for l in f.read_text(encoding="utf-8").splitlines()]
rows = []
for f in sorted(evaldir.glob("voice.*.jsonl")):
arm = f.stem.replace("voice.", "")
recs = arm_texts(f)
allt = "\n".join(r["continuation"] for r in recs)
d = delta(allt, ref_prof, mu, sd, keys)
# within-arm sampling spread = the REAL noise floor for a between-arm gap:
# split by seed and score each subset; the range is this metric's variance
# at this sample size, measured rather than assumed.
by_seed = {}
for r in recs:
by_seed.setdefault(r["seed"], []).append(r["continuation"])
seed_ds = [delta("\n".join(v), ref_prof, mu, sd, keys) for v in by_seed.values() if len(v) > 2]
spread = (max(seed_ds) - min(seed_ds)) if len(seed_ds) > 1 else float("nan")
rows.append((arm, d, len(allt.split()), seed_ds, spread))
print(" arm delta_cb per-seed [words]")
for arm, d, w, sd_, spread in sorted(rows, key=lambda x: x[1]):
seeds = " ".join(f"{x:.3f}" for x in sd_)
print(f" {arm:20s} {d:.3f} ({seeds}) [{w}]")
# the noise floor is the LARGEST within-arm spread across arms
floors = [r[4] for r in rows if r[4] == r[4]]
noise = max(floors) if floors else float("nan")
print(f"\n measured noise floor (largest within-arm seed spread): {noise:.3f}")
print(f" -> a between-arm gap must exceed ~{noise:.3f} to be a real difference\n")
base = next((d for a, d, _, _, _ in rows if "unadapted" in a), None)
if base is not None:
print(" vs base-unadapted control (positive gap = moved toward Yarros):")
for arm, d, _, _, _ in sorted(rows, key=lambda x: x[1]):
if "unadapted" in arm:
continue
gap = base - d
verdict = ("MOVED toward Yarros (exceeds noise floor)" if gap > noise
else "moved toward Yarros, but within the measured noise floor")
print(f" {arm:20s} {gap:+.3f} ({verdict})")
ordered = [a for a, *_ in sorted(rows, key=lambda x: x[1])]
print(f"\n ordering: {' < '.join(ordered)} (lower = more Yarros-like)")
print(" ⚠ one seed-pair per arm; this ordering CORROBORATES the independent held-out")
print(" loss ordering (Base < Instruct) but is not itself a multi-seed result.")
return 0
if __name__ == "__main__":
sys.exit(main())