#!/bin/bash # lv-mccarthy v2 gate. Design is FROZEN in scripts/mccarthy-corpus/GATE-PREREG.md # and was written before this script ever ran. Do not edit the arms, the fixture # size or the seeds to chase a result -- re-run, do not re-tune. # # THREE ARMS. ckpt900 is the eval-loss minimum (2.38706, epoch 1.958). ckpt450 # (2.4063, epoch 0.980) is +0.0193 against a 0.00393 median neighbour jitter = # 4.9x, i.e. NOT tied -- and that is the difference from the Hemingway gate, where # the second arm was a coin-flip. It is here to test a stated prior (on Bronte the # earlier epoch-1 checkpoint won on the axes that resolve) and to price the # memorisation headroom an earlier checkpoint would buy on an IN-COPYRIGHT author. # `base` is the negative control for memorisation and the voice baseline. # adapter/ (epoch 3.0, +18.4x jitter) is NOT gated: the loss curve settles that one. set -o pipefail cd ~/lv-mccarthy || exit 1 PY=/home/infra-ops/ml/.venv/bin/python RUN=~/r49-runs/mccarthy-4b-pairs-3ep OUT=~/r49-runs/mccarthy-eval BEATS=beats-mccarthy-60.json SIDE=beats-mccarthy-60.sidecar.json PROV=pairs/pairs-full.jsonl.provenance.json SEEDS="1234 5678 9012 3456" mkdir -p "$OUT" log(){ echo "[eval $(date +%H:%M:%S)] $*"; } # --system-from is MANDATORY and it is load-bearing for the VOICE axis here, not # merely hygienic. The mccarthy register NAMES the punctuation (no quote marks, # `dont`/`aint`/`wont`), so driving the base control with the same prompt hands the # cheap char-bigram win to BOTH sides and the adapter earns no delta for it. See # GATE-PREREG.md section 5. The harness's built-in SYS is Yarros's; using it would # confound the adapter change with a prompt change AND hand the adapter the win. [ -f "$PROV" ] || { log "MISSING $PROV"; exit 1; } # The run must have trained under the same system prompt the eval drives. Asserted, # not assumed -- the Hemingway script claimed this was "verified" in a comment, which # is a claim no reader can check. Here it fails the run. "$PY" - "$PROV" "$RUN/provenance.json" <<'PYCHK' || exit 1 import json, sys pairs = json.load(open(sys.argv[1])) run = json.load(open(sys.argv[2])) ps = pairs.get("system_prompt") or (pairs.get("register") or {}).get("system_prompt") rs = run.get("system_prompt") if ps is None: print(f"== CANNOT VERIFY: no system_prompt in {sys.argv[1]}"); sys.exit(1) if ps != rs: print("== REFUSING: the run trained under a DIFFERENT system prompt than the eval would drive.") print(f" pairs provenance: {ps[:120]!r}") print(f" run provenance : {(rs or '')[:120]!r}") sys.exit(1) print(f" [PASS] run system_prompt == pairs system_prompt ({len(ps)} chars)") PYCHK # 60 beats, max-words 140: the mccarthy register asks for 90-140 and score_beats.py # scores the in-band rate at 90-140. Building the fixture at the script's default 150 # would put reference passages outside the band the product asks for. if [ ! -s "$BEATS" ]; then log "building fixture" "$PY" scripts/r49-corpus/build_beat_fixture.py \ --pairs pairs/pairs-val.jsonl --out "$BEATS" --sidecar "$SIDE" \ -n 60 --seed 4919 --min-words 90 --max-words 140 || exit 1 else log "fixture $BEATS already exists -- reusing" fi # AMENDMENT 3 adds ckpt300 (ep 0.652) and ckpt225 (ep 0.489). Same fixture, same seeds, # same rule -- adding candidates does not move a bar. Arms with generations already on # disk are skipped, so re-running this script only fills in the new ones. for arm in base:NONE ckpt900:$RUN/checkpoints/checkpoint-900 ckpt450:$RUN/checkpoints/checkpoint-450 \ ckpt300:$RUN/checkpoints/checkpoint-300 ckpt225:$RUN/checkpoints/checkpoint-225; do name=${arm%%:*}; path=${arm#*:} if [ "$name" != "base" ] && [ ! -d "$path" ]; then log "MISSING $path"; exit 1; fi if [ -s "$OUT/beats5.$name.jsonl" ]; then log "arm $name already has $(wc -l < "$OUT/beats5.$name.jsonl") generations -- skipping" continue fi log "arm $name" if [ "$name" = "base" ]; then "$PY" scripts/r49-corpus/gen_beats_chat_yarros.py \ --base ~/carriers/Qwen3-4B-Instruct --beats "$BEATS" \ --out "$OUT/beats5.$name.jsonl" --arm "$name" --seeds $SEEDS \ --system-from "$PROV" || exit 1 else "$PY" scripts/r49-corpus/gen_beats_chat_yarros.py \ --base ~/carriers/Qwen3-4B-Instruct --adapter "$path" --beats "$BEATS" \ --out "$OUT/beats5.$name.jsonl" --arm "$name" --seeds $SEEDS \ --system-from "$PROV" || exit 1 fi log " $(wc -l < "$OUT/beats5.$name.jsonl") generations" done log "AXIS B -- MEMORISATION (corpus = the renamed copies the adapter trained on)" # --corpus and --eval-dir are passed explicitly: the script's Yarros defaults would # compare a McCarthy arm against the YARROS corpus and report a clean zero that # means "different book", not "did not memorise". "$PY" scripts/yarros-corpus/memorization_check.py \ --eval-dir "$OUT" --corpus corpus-renamed/copies --glob 'beats5.*.jsonl' --strip 'beats5.' -n 8 \ 2>&1 | tee "$OUT/memorization.txt" for cand in ckpt900 ckpt450 ckpt300 ckpt225; do log "AXIS C -- DAMAGE (ran-on / out-of-band), $cand vs base" "$PY" scripts/yarros-corpus/score_beats.py \ --arm base="$OUT/beats5.base.jsonl" \ --arm ckpt900="$OUT/beats5.ckpt900.jsonl" \ --arm ckpt450="$OUT/beats5.ckpt450.jsonl" \ --arm ckpt300="$OUT/beats5.ckpt300.jsonl" \ --arm ckpt225="$OUT/beats5.ckpt225.jsonl" \ --baseline base --candidate "$cand" --metric-source raw \ --out "$OUT/score.$cand.json" 2>&1 | tee "$OUT/score.$cand.txt" done log "AXIS A -- VOICE (delta_cb vs held-out McCarthy), + the pre-registered punct reads" "$PY" scripts/mccarthy-corpus/voice-prep.py || exit 1 "$PY" scripts/r49-corpus/voice_distance.py corpus-renamed "$OUT" --author McCarthy \ --punct-report --secondary-normalised \ 2>&1 | tee "$OUT/voice_distance.txt" echo "rc=0" > ~/lv-mccarthy/.eval-complete log "done"