195 lines
10 KiB
Python
195 lines
10 KiB
Python
"""Score beat arms against a decision rule fixed BEFORE any arm's output was read.
|
|
|
|
⚠ THE RULE BELOW IS PRE-REGISTERED. It was written and committed while the generation
|
|
chain was still running, for the reason brokkr's frozen adjudication exists: a threshold
|
|
chosen after seeing the numbers is a threshold chosen to produce a verdict. If it needs to
|
|
change, change it in a commit that says so and re-run every arm -- do not edit it in place
|
|
between a read and a conclusion.
|
|
|
|
THE QUESTION: does pair-SFT fix the length/close discipline that raw-text training cost,
|
|
without giving back the direction-following it bought? The raw-text instruct arm's known
|
|
profile is on-beat strong, in-band weak, ran-on frequent. In-band is therefore the axis the
|
|
pilot exists to move, and it is the axis the rule keys on.
|
|
|
|
⚠⚠ AMENDED 2026-09-15 BY THE OPERATOR, AFTER THE v1 RULE HAD BEEN RUN AND REPORTED.
|
|
The amendment is recorded here rather than applied silently, because a frozen rule edited
|
|
in place between a read and a conclusion is indistinguishable from a rule chosen to produce
|
|
a verdict. What changed, and why:
|
|
|
|
v1 (commit 713e83d, pre-registered before any arm was read) gated on IN-BAND, ON-BEAT and
|
|
RAN-ON. It returned DO-NOT-SCALE at n=120: in-band +0.08 against a 0.233 floor (the
|
|
+0.45 seen at n=20 was noise), on-beat -0.27, ran-on -0.35.
|
|
|
|
The defect in v1: it gated on metrics the UNADAPTED carrier already maxes. Measured,
|
|
n=120 -- base-unadapted in-band 0.96. Instruction-following is a property Qwen3-4B-Instruct
|
|
ships with, so those three axes can only detect DAMAGE that training does. They cannot
|
|
detect the thing an adapter exists to buy, which is VOICE, and v1 contained no voice term
|
|
at all. It was a well-formed rule measuring the wrong question.
|
|
|
|
v2 therefore gates on the two axes that distinguish the arms, and keeps one v1 term as a
|
|
guard rather than a target:
|
|
A. VOICE -- delta_cb vs held-out Yarros must beat the base-unadapted control by MORE
|
|
than the measured noise floor. (pairs +0.230 vs floor 0.153; rawtext
|
|
+0.141, which does NOT clear -- so this term discriminates.)
|
|
B. NOT COPIED -- verbatim 8-gram overlap with the training corpus must not exceed the
|
|
base-unadapted negative control by a meaningful margin. delta_cb is blind
|
|
to regurgitation and a memorising arm scores near the same-author target,
|
|
so A without B is a trap. (pairs 0.10 vs control 0.07; rawtext 0.14.)
|
|
C. NO DAMAGE -- ran-on must not get worse than raw-text by more than the floor.
|
|
Retained because overshoot is the one behavioural axis where the adapters
|
|
actually differ from the base carrier.
|
|
|
|
NOT carried into v2: in-band (unresolvable -- base maxes it) and on-beat. ⚠ on-beat's
|
|
-0.27 was OUTSIDE the floor and is a REAL signal by the keyword proxy; it is dropped from
|
|
the gate, not explained away. The open question is whether the proxy punishes
|
|
dramatisation -- a generation rendering "she mocks him" as actual mockery scores zero for
|
|
"mocks" -- and three read samples is an anecdote, not an answer. It stays an open
|
|
follow-up against the full run.
|
|
|
|
Authorised by the operator 2026-09-15 ("amend the rule and run the full corpus"). A and B
|
|
are evaluated by voice_distance.py and memorization_check.py, which this script does not
|
|
recompute; it reports C and prints the v1 table for continuity.
|
|
|
|
v1 DECISION RULE (superseded, retained verbatim) — pairs replace raw-text for Skaldsong iff
|
|
ALL THREE hold, same harness, same n, same box, every arm re-measured in one session:
|
|
|
|
1. in-band rate is HIGHER than the raw-text arm by MORE than the pooled within-arm
|
|
seed spread. A gain inside the spread is noise, not a fix.
|
|
2. on-beat coverage is not WORSE than the raw-text arm by more than that same spread --
|
|
it must not have bought length by losing direction.
|
|
3. ran-on rate is not HIGHER than the raw-text arm by more than that spread.
|
|
|
|
(1) fails -> the pilot did not do its job. Do not scale.
|
|
(1) holds, 2 or 3 no -> mixed result. Surface to the operator; do not auto-scale.
|
|
all three -> scale to the full corpus.
|
|
|
|
on-beat is scored as COVERAGE (fraction of the beat's content keywords that surface in the
|
|
paragraph), thresholded at >= 0.5 for the binary. Both the coverage mean and the binary are
|
|
reported: a rule that reads only the binary hides a large move inside a threshold, and a
|
|
rule that reads only the mean hides a bimodal arm.
|
|
|
|
NOISE FLOOR: the spread is the max-minus-min of a metric across the four SEEDS within an
|
|
arm, pooled (max) over arms. It is measured per run and printed with the verdict, because a
|
|
between-arm gap smaller than it is not a finding. This is the A-vs-A floor; it is not the
|
|
same thing as the distance to real Yarros and must not be reported as if it were.
|
|
"""
|
|
from __future__ import annotations
|
|
import argparse, json, math, statistics as st
|
|
from collections import defaultdict
|
|
from pathlib import Path
|
|
|
|
ON_BEAT_COVERAGE = 0.5
|
|
|
|
|
|
def load(path: Path) -> list[dict]:
|
|
return [json.loads(l) for l in path.read_text(encoding="utf-8").splitlines() if l.strip()]
|
|
|
|
|
|
def view(r: dict, source: str) -> tuple[str, int, bool]:
|
|
"""Return (text, words, ran_on) for the chosen measurement surface.
|
|
|
|
The harness has always scored `paragraph` -- the generation truncated at its first
|
|
blank line -- because every arm before this one was expected to emit ONE paragraph.
|
|
The pair arm emits a multi-paragraph passage by construction, so that surface reported
|
|
it as a 19-word off-beat fragment while the untruncated output was 90-132 words with
|
|
the beat rendered in a later block. Both views are kept and the verdict names which one
|
|
it used: truncated answers "does it emit one paragraph", raw answers "did it render the
|
|
beat at the requested length", and either alone answers a different question than the
|
|
reader assumes it does.
|
|
"""
|
|
if source == "raw":
|
|
txt = r.get("raw", r["paragraph"])
|
|
w = r.get("raw_words", len(txt.split()))
|
|
return txt, w, w > 140 # ran-on := overshot the requested band
|
|
return r["paragraph"], r["words"], r["ran_on"]
|
|
|
|
|
|
def per_seed(rows: list[dict], source: str) -> dict[int, dict]:
|
|
by = defaultdict(list)
|
|
for r in rows:
|
|
by[r["seed"]].append(r)
|
|
out = {}
|
|
for seed, rs in by.items():
|
|
views = [view(r, source) for r in rs]
|
|
cov = []
|
|
for r, (txt, _w, _ro) in zip(rs, views):
|
|
kws = r["beat_keywords"]
|
|
low = txt.lower()
|
|
cov.append((sum(1 for k in kws if k[:5] in low) / len(kws)) if kws else 0.0)
|
|
out[seed] = {
|
|
"n": len(rs),
|
|
"in_band": sum(1 for _t, w, _ro in views if 90 <= w <= 140) / len(rs),
|
|
"ran_on": sum(1 for _t, _w, ro in views if ro) / len(rs),
|
|
"coverage": st.mean(cov),
|
|
"on_beat": sum(1 for c in cov if c >= ON_BEAT_COVERAGE) / len(cov),
|
|
"words_median": st.median(w for _t, w, _ro in views),
|
|
}
|
|
return out
|
|
|
|
|
|
def summarise(name: str, rows: list[dict], source: str) -> dict:
|
|
seeds = per_seed(rows, source)
|
|
agg = {"arm": name, "n": len(rows), "seeds": len(seeds)}
|
|
for k in ("in_band", "ran_on", "coverage", "on_beat", "words_median"):
|
|
vals = [s[k] for s in seeds.values()]
|
|
agg[k] = st.mean(vals)
|
|
agg[k + "_spread"] = max(vals) - min(vals)
|
|
return agg
|
|
|
|
|
|
def main() -> int:
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--arm", action="append", required=True, metavar="NAME=PATH")
|
|
ap.add_argument("--baseline", required=True, help="arm name the rule compares against")
|
|
ap.add_argument("--candidate", required=True)
|
|
ap.add_argument("--out", default=None)
|
|
ap.add_argument("--metric-source", choices=["paragraph", "raw"], default="paragraph")
|
|
a = ap.parse_args()
|
|
|
|
arms = {}
|
|
for spec in a.arm:
|
|
name, path = spec.split("=", 1)
|
|
arms[name] = summarise(name, load(Path(path)), a.metric_source)
|
|
|
|
floor = max(max(v[k + "_spread"] for k in ("in_band", "ran_on", "on_beat", "coverage"))
|
|
for v in arms.values())
|
|
|
|
print(f"measurement surface: {a.metric_source}")
|
|
hdr = f"{'arm':<28} {'n':>3} {'in-band':>8} {'on-beat':>8} {'cover':>7} {'ran-on':>7} {'words':>6}"
|
|
print(hdr); print("-" * len(hdr))
|
|
for v in arms.values():
|
|
print(f"{v['arm']:<28} {v['n']:>3} {v['in_band']:>8.2f} {v['on_beat']:>8.2f} "
|
|
f"{v['coverage']:>7.2f} {v['ran_on']:>7.2f} {v['words_median']:>6.0f}")
|
|
print(f"\nnoise floor (max within-arm spread across seeds): {floor:.3f}")
|
|
print("⚠ any between-arm gap at or under that is NOT a finding\n")
|
|
|
|
b, c = arms[a.baseline], arms[a.candidate]
|
|
d_band = c["in_band"] - b["in_band"]
|
|
d_beat = c["on_beat"] - b["on_beat"]
|
|
d_ran = c["ran_on"] - b["ran_on"]
|
|
c1 = d_band > floor
|
|
c2 = d_beat >= -floor
|
|
c3 = d_ran <= floor
|
|
print(f"1. in-band {c['in_band']:.2f} vs {b['in_band']:.2f} delta {d_band:+.2f} "
|
|
f"{'PASS' if c1 else 'FAIL'} (needs > +{floor:.3f})")
|
|
print(f"2. on-beat {c['on_beat']:.2f} vs {b['on_beat']:.2f} delta {d_beat:+.2f} "
|
|
f"{'PASS' if c2 else 'FAIL'} (needs >= -{floor:.3f})")
|
|
print(f"3. ran-on {c['ran_on']:.2f} vs {b['ran_on']:.2f} delta {d_ran:+.2f} "
|
|
f"{'PASS' if c3 else 'FAIL'} (needs <= +{floor:.3f})")
|
|
|
|
verdict = ("SCALE" if (c1 and c2 and c3) else
|
|
"DO-NOT-SCALE" if not c1 else "MIXED-SURFACE-TO-OPERATOR")
|
|
print(f"\nVERDICT: {verdict}")
|
|
if a.out:
|
|
Path(a.out).write_text(json.dumps(
|
|
{"arms": arms, "noise_floor": floor, "metric_source": a.metric_source, "on_beat_coverage_threshold": ON_BEAT_COVERAGE,
|
|
"baseline": a.baseline, "candidate": a.candidate,
|
|
"deltas": {"in_band": d_band, "on_beat": d_beat, "ran_on": d_ran},
|
|
"criteria": {"in_band_gt_floor": c1, "on_beat_not_worse": c2, "ran_on_not_worse": c3},
|
|
"verdict": verdict}, indent=2), encoding="utf-8")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|