BabyYarros: raw-surface scoring and a memorization check with both controls
This commit is contained in:
@@ -46,27 +46,50 @@ def load(path: Path) -> list[dict]:
|
||||
return [json.loads(l) for l in path.read_text(encoding="utf-8").splitlines() if l.strip()]
|
||||
|
||||
|
||||
def per_seed(rows: list[dict]) -> dict[int, dict]:
|
||||
def view(r: dict, source: str) -> tuple[str, int, bool]:
|
||||
"""Return (text, words, ran_on) for the chosen measurement surface.
|
||||
|
||||
The harness has always scored `paragraph` -- the generation truncated at its first
|
||||
blank line -- because every arm before this one was expected to emit ONE paragraph.
|
||||
The pair arm emits a multi-paragraph passage by construction, so that surface reported
|
||||
it as a 19-word off-beat fragment while the untruncated output was 90-132 words with
|
||||
the beat rendered in a later block. Both views are kept and the verdict names which one
|
||||
it used: truncated answers "does it emit one paragraph", raw answers "did it render the
|
||||
beat at the requested length", and either alone answers a different question than the
|
||||
reader assumes it does.
|
||||
"""
|
||||
if source == "raw":
|
||||
txt = r.get("raw", r["paragraph"])
|
||||
w = r.get("raw_words", len(txt.split()))
|
||||
return txt, w, w > 140 # ran-on := overshot the requested band
|
||||
return r["paragraph"], r["words"], r["ran_on"]
|
||||
|
||||
|
||||
def per_seed(rows: list[dict], source: str) -> dict[int, dict]:
|
||||
by = defaultdict(list)
|
||||
for r in rows:
|
||||
by[r["seed"]].append(r)
|
||||
out = {}
|
||||
for seed, rs in by.items():
|
||||
cov = [(r["keyword_hits"] / len(r["beat_keywords"])) if r["beat_keywords"] else 0.0
|
||||
for r in rs]
|
||||
views = [view(r, source) for r in rs]
|
||||
cov = []
|
||||
for r, (txt, _w, _ro) in zip(rs, views):
|
||||
kws = r["beat_keywords"]
|
||||
low = txt.lower()
|
||||
cov.append((sum(1 for k in kws if k[:5] in low) / len(kws)) if kws else 0.0)
|
||||
out[seed] = {
|
||||
"n": len(rs),
|
||||
"in_band": sum(1 for r in rs if r["in_band"]) / len(rs),
|
||||
"ran_on": sum(1 for r in rs if r["ran_on"]) / len(rs),
|
||||
"in_band": sum(1 for _t, w, _ro in views if 90 <= w <= 140) / len(rs),
|
||||
"ran_on": sum(1 for _t, _w, ro in views if ro) / len(rs),
|
||||
"coverage": st.mean(cov),
|
||||
"on_beat": sum(1 for c in cov if c >= ON_BEAT_COVERAGE) / len(cov),
|
||||
"words_median": st.median(r["words"] for r in rs),
|
||||
"words_median": st.median(w for _t, w, _ro in views),
|
||||
}
|
||||
return out
|
||||
|
||||
|
||||
def summarise(name: str, rows: list[dict]) -> dict:
|
||||
seeds = per_seed(rows)
|
||||
def summarise(name: str, rows: list[dict], source: str) -> dict:
|
||||
seeds = per_seed(rows, source)
|
||||
agg = {"arm": name, "n": len(rows), "seeds": len(seeds)}
|
||||
for k in ("in_band", "ran_on", "coverage", "on_beat", "words_median"):
|
||||
vals = [s[k] for s in seeds.values()]
|
||||
@@ -81,16 +104,18 @@ def main() -> int:
|
||||
ap.add_argument("--baseline", required=True, help="arm name the rule compares against")
|
||||
ap.add_argument("--candidate", required=True)
|
||||
ap.add_argument("--out", default=None)
|
||||
ap.add_argument("--metric-source", choices=["paragraph", "raw"], default="paragraph")
|
||||
a = ap.parse_args()
|
||||
|
||||
arms = {}
|
||||
for spec in a.arm:
|
||||
name, path = spec.split("=", 1)
|
||||
arms[name] = summarise(name, load(Path(path)))
|
||||
arms[name] = summarise(name, load(Path(path)), a.metric_source)
|
||||
|
||||
floor = max(max(v[k + "_spread"] for k in ("in_band", "ran_on", "on_beat", "coverage"))
|
||||
for v in arms.values())
|
||||
|
||||
print(f"measurement surface: {a.metric_source}")
|
||||
hdr = f"{'arm':<28} {'n':>3} {'in-band':>8} {'on-beat':>8} {'cover':>7} {'ran-on':>7} {'words':>6}"
|
||||
print(hdr); print("-" * len(hdr))
|
||||
for v in arms.values():
|
||||
@@ -118,7 +143,7 @@ def main() -> int:
|
||||
print(f"\nVERDICT: {verdict}")
|
||||
if a.out:
|
||||
Path(a.out).write_text(json.dumps(
|
||||
{"arms": arms, "noise_floor": floor, "on_beat_coverage_threshold": ON_BEAT_COVERAGE,
|
||||
{"arms": arms, "noise_floor": floor, "metric_source": a.metric_source, "on_beat_coverage_threshold": ON_BEAT_COVERAGE,
|
||||
"baseline": a.baseline, "candidate": a.candidate,
|
||||
"deltas": {"in_band": d_band, "on_beat": d_beat, "ran_on": d_ran},
|
||||
"criteria": {"in_band_gt_floor": c1, "on_beat_not_worse": c2, "ran_on_not_worse": c3},
|
||||
|
||||
Reference in New Issue
Block a user