"""Score transcripts against ground truth (or any reference) and find dropped stretches. Both sides are normalised word by word with Whisper's English normaliser (transformers 4.53.3, no spelling map), so a hypothesis token keeps the time of the word it came from. Alignment: difflib opcodes, then exact Levenshtein inside each non-matching block. A GAP is a maximal stretch between matching blocks of >= ISLAND tokens (shorter matching islands inside a mismatch are treated as part of it, so one spurious "the" cannot split a skipped paragraph in two). A gap is a DROPOUT if it holds >= RUN reference words and the hypothesis emitted fewer than half as many there (the hypothesis skipped speech), and an INSERTION run if the mirror holds (the hypothesis added >= RUN words the reference does not have). Prints counts only. """ import difflib import json import sys from transformers.models.whisper.english_normalizer import EnglishTextNormalizer RUN = 10 ISLAND = 3 _norm = EnglishTextNormalizer({}) def tokens_from_text(text): return [t for w in text.split() for t in _norm(w).split()] def tokens_from_words(words): """[(token, start, end)] from word dicts, one entry per normalised token.""" out = [] for w in words: for t in _norm(w["word"]).split(): out.append((t, float(w["start"]), float(w["end"]))) return out def _edit_counts(a, b): n, m = len(a), len(b) d = [[0] * (m + 1) for _ in range(n + 1)] for i in range(n + 1): d[i][0] = i for j in range(m + 1): d[0][j] = j for i in range(1, n + 1): for j in range(1, m + 1): d[i][j] = min(d[i - 1][j] + 1, d[i][j - 1] + 1, d[i - 1][j - 1] + (a[i - 1] != b[j - 1])) s = i_ = de = 0 i, j = n, m while i > 0 or j > 0: if i > 0 and j > 0 and d[i][j] == d[i - 1][j - 1] + (a[i - 1] != b[j - 1]): s += a[i - 1] != b[j - 1] i, j = i - 1, j - 1 elif i > 0 and d[i][j] == d[i - 1][j] + 1: de += 1 i -= 1 else: i_ += 1 j -= 1 return s, i_, de def score(ref, hyp, hyp_times=None): """ref, hyp: token lists. hyp_times: [(start, end)] per hyp token, optional.""" sm = difflib.SequenceMatcher(None, ref, hyp, autojunk=False) ops = sm.get_opcodes() S = I = D = 0 for tag, i1, i2, j1, j2 in ops: if tag != "equal": s, i_, de = _edit_counts(ref[i1:i2], hyp[j1:j2]) S, I, D = S + s, I + i_, D + de # gaps between solid matching blocks solid = [(i1, i2, j1, j2) for tag, i1, i2, j1, j2 in ops if tag == "equal" and i2 - i1 >= ISLAND] bounds = [(0, 0, 0, 0)] + solid + [(len(ref), len(ref), len(hyp), len(hyp))] drops, ins = [], [] for (_, a_i2, _, a_j2), (b_i1, _, b_j1, _) in zip(bounds, bounds[1:]): gl, hl = b_i1 - a_i2, b_j1 - a_j2 t0 = t1 = None if hyp_times: t0 = hyp_times[a_j2 - 1][1] if a_j2 > 0 else 0.0 t1 = hyp_times[b_j1][0] if b_j1 < len(hyp) else hyp_times[-1][1] gap = dict(ref_start=a_i2, ref_words=gl, hyp_start=a_j2, hyp_words=hl, t0=t0, t1=t1) if gl >= RUN and hl < 0.5 * gl: drops.append(gap) elif hl >= RUN and gl < 0.5 * hl: ins.append(gap) n = len(ref) return dict(ref_words=n, hyp_words=len(hyp), S=S, I=I, D=D, wer=round((S + I + D) / n, 4) if n else None, dropouts=len(drops), dropout_words=sum(g["ref_words"] for g in drops), insertion_runs=len(ins), insertion_words=sum(g["hyp_words"] for g in ins), drop_gaps=drops, ins_gaps=ins) def load_hyp(path): doc = json.load(open(path)) if "word_timestamps" in doc and doc["word_timestamps"]: toks = tokens_from_words(doc["word_timestamps"]) return [t for t, _, _ in toks], [(s, e) for _, s, e in toks] return tokens_from_text(doc.get("transcription") or doc.get("text", "")), None if __name__ == "__main__": ref_path, hyp_path = sys.argv[1], sys.argv[2] ref = tokens_from_text(open(ref_path).read()) hyp, times = load_hyp(hyp_path) r = score(ref, hyp, times) print(json.dumps({k: v for k, v in r.items() if not k.endswith("_gaps")})) for g in r["drop_gaps"]: print(" DROP", g) for g in r["ins_gaps"]: print(" INS ", g)