"""Long-form scoring, identical to the dropout investigation's public-file method: gtscore.py (copied verbatim from /tank/spikes/scriberr-slicer/code/dropout/) against the timed ground truth, dropouts split into clean speech vs crosstalk with boot.py's rule (diarized overlap >= 10 % of the stretch = crosstalk). Our hypotheses are text-only (the seat returns no timestamps), so they are tokenised with gtscore.tokens_from_text; gap times come from the ground-truth token times. usage: longscore.py RAW_LONG_DIR -> json lines per (file, arm, k) """ import glob import json import os import sys sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) from gtscore import score, tokens_from_text # noqa: E402 ROOT = "/tank/spikes/scriberr-slicer" def keeper(f): tg = json.load(open(f"{ROOT}/gt/{f}.timed.json")) gtt = tg["times"] segs = json.load(open(f"{ROOT}/public/{f}.diar.json"))["segments"] def clean(g): i0, i1 = g["ref_start"], g["ref_start"] + g["ref_words"] t0, t1 = gtt[i0], (gtt[i1] if i1 < len(gtt) else gtt[-1] + 0.5) n = max(1, int((max(t1, t0 + 0.5) - t0) / 0.1)) ov = sum(sum(1 for s in segs if s["start"] <= t0 + k * 0.1 < s["end"]) >= 2 for k in range(n)) / n return ov < 0.1, round(t0, 1), round(t1, 1) return tg["tokens"], clean for p in sorted(glob.glob(f"{sys.argv[1]}/*.json")): d = json.load(open(p)) ref, clean = keeper(d["file"]) if d.get("status") != 200 or d.get("text") is None: print(json.dumps(dict(file=d["file"], arm=d["arm"], k=d["k"], status=d.get("status"), err=d.get("err")))) continue r = score(ref, tokens_from_text(d["text"])) drops = [(g, *clean(g)) for g in r["drop_gaps"]] cl = [g for g, c, _, _ in drops if c] print(json.dumps(dict(file=d["file"], arm=d["arm"], k=d["k"], e2e_s=round(d["e2e_ms"] / 1000, 1), wer=round(100 * r["wer"], 2), S=r["S"], D=r["D"], I=r["I"], ref_words=r["ref_words"], hyp_words=r["hyp_words"], dropouts=len(drops), dropout_words=sum(g["ref_words"] for g, *_ in drops), clean_dropouts=len(cl), clean_dropout_words=sum(g["ref_words"] for g in cl), insertion_runs=r["insertion_runs"], insertion_words=r["insertion_words"], drop_spans_s=[[t0, t1, g["ref_words"], "clean" if c else "crosstalk"] for g, c, t0, t1 in drops])))