"""Accuracy summary over out/raw/acc/--.jsonl. Run with envs/score. - WER per arm per set (score.py: whole-string Whisper normaliser, exact Levenshtein). - Paired bootstrap deltas vs a reference arm (default ab-a1 = the seat's runtime + weights). - Positive control (pc: 1.5 s of digital silence inside 40 utterances, reference unchanged): the arm must register extra DELETIONS vs the same 40 utterances in ls-clean. - Null control (null: ls-clean at -0.5 dB): the WER change vs ls-clean must sit inside the paired CI. - Determinism: identical text across two instances / thread counts. - Output features: share of outputs with sentence punctuation, with an upper-case letter, all-lower. usage: acc_summary.py ACC_DIR DATA_DIR [REF_ARM] """ import glob import json import os import re import sys sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import score # noqa: E402 ACC, DATA = sys.argv[1], sys.argv[2] REF = sys.argv[3] if len(sys.argv) > 3 else "ab-a1" SETS = ["ls-clean", "ls-other", "ami", "pc", "null"] def load(set_name): ref = {json.loads(l)["id"]: json.loads(l) for l in open(f"{DATA}/{set_name}.jsonl")} by = {} for p in glob.glob(f"{ACC}/{set_name}--*.jsonl"): arm = p.rsplit("--", 1)[1][:-6] by[arm] = {json.loads(l)["id"]: json.loads(l) for l in open(p)} return ref, by out = dict(wer=[], delta=[], pc=[], null=[], same=[], features=[]) pu = {} for s in SETS: ref, by = load(s) for arm, rows in sorted(by.items()): pu[(s, arm)] = score.per_utt(ref, rows) out["wer"].append(dict(set=s, arm=arm, **score.summary(pu[(s, arm)]))) if REF in by: for arm in sorted(by): if arm != REF: out["delta"].append(dict(set=s, x=arm, y=REF, **score.boot(pu[(s, arm)], pu[(s, REF)]))) if "ab-b32" in by: for arm in sorted(by): if arm not in ("ab-b32", REF): out["delta"].append(dict(set=s, x=arm, y="ab-b32", **score.boot(pu[(s, arm)], pu[(s, "ab-b32")]))) # positive control: deletions on the silenced copies vs the same utterances unsilenced pc_ref = {json.loads(l)["id"]: json.loads(l) for l in open(f"{DATA}/pc.jsonl")} for (s, arm), v in pu.items(): if s != "pc" or ("ls-clean", arm) not in pu: continue base = pu[("ls-clean", arm)] ids = [k for k in pc_ref if v.get(k) and base.get(k)] d_pc = sum(v[k][1] for k in ids); d_base = sum(base[k][1] for k in ids) e_pc = sum(sum(v[k][:3]) for k in ids); e_base = sum(sum(base[k][:3]) for k in ids) n = sum(base[k][3] for k in ids) hit = sum(1 for k in ids if v[k][1] > base[k][1]) out["pc"].append(dict(arm=arm, utts=len(ids), deletions_clean=d_base, deletions_silenced=d_pc, errors_clean=e_base, errors_silenced=e_pc, wer_clean=round(100 * e_base / n, 2), wer_silenced=round(100 * e_pc / n, 2), utts_with_more_deletions=hit)) # null control for (s, arm), v in pu.items(): if s == "null" and ("ls-clean", arm) in pu: b = score.boot(v, pu[("ls-clean", arm)]) out["null"].append(dict(arm=arm, **b)) # determinism: identical text between pairs of arms on ls-clean ref, by = load("ls-clean") for a, b in (("ab-a1", "ab-a2"), ("ab-a1", "ab-at16"), ("ab-b32", "ab-b32t"), ("ab-a1", "A-live")): if a in by and b in by: ids = [k for k in ref if k in by[a] and k in by[b]] out["same"].append(dict(a=a, b=b, utts=len(ids), identical_text=sum(by[a][k]["text"] == by[b][k]["text"] for k in ids))) # features for arm, rows in sorted(by.items()): txt = [r["text"] or "" for r in rows.values() if r.get("status") == 200] out["features"].append(dict(arm=arm, outputs=len(txt), with_sentence_punct=round(100 * sum(bool(re.search(r"[.?!]", t)) for t in txt) / len(txt), 1), with_comma=round(100 * sum("," in t for t in txt) / len(txt), 1), with_upper=round(100 * sum(any(c.isupper() for c in t) for t in txt) / len(txt), 1), all_lower=round(100 * sum(t == t.lower() and t.strip() != "" for t in txt) / len(txt), 1), empty=sum(t.strip() == "" for t in txt), non_ascii=sum(any(ord(c) > 127 for c in t) for t in txt))) for k, v in out.items(): print(f"## {k}") for r in v: print(json.dumps(r)) json.dump(out, open(f"{ACC}/../../accuracy-summary.json", "w"), indent=1)