From 2dafe7ce81c217609ff2c8616e43b0d72255e170 Mon Sep 17 00:00:00 2001 From: Vuong Hoang Date: Wed, 30 Sep 2026 11:41:07 -0700 Subject: [PATCH] fix(parakeet): overlap buffered chunks and stitch at an agreed word parakeet_transcribe_buffered.py cut long audio at fixed --chunk-len marks with no overlap, so a word straddling a mark was chopped, dropped or transcribed twice. Adjacent chunks now overlap by --overlap seconds (default 4, counted inside --chunk-len so no chunk grows), and in each overlap the chunks hand over at the word nearest the cut that both transcribed with the same text at nearly the same time (within 0.5 s), keeping whichever copy leaves the words in time order. With no such word they split at the cut. Splitting both chunks at the cut by word start time is not enough on its own: a word after a pause can be timestamped anywhere in the pause, so the two chunks may place it on opposite sides of the cut. --pause-search N (opt-in) also moves each cut back to the quietest 0.3 s in the last N seconds before the limit. Measured neutral on top of the overlap, so it is off by default. The CLI and JSON the Go adapter reads are unchanged; the new flags are optional, and the JSON gains overlap_secs, pause_search_secs and cut_times. --overlap 0 reproduces the previous output exactly. NeMo is now imported inside transcribe_buffered() so the slicing and stitching helpers can be unit-tested without a GPU. If a chunk ever returns text without word timestamps, its text is kept rather than dropped. --- .../py/nvidia/parakeet_transcribe_buffered.py | 221 ++++++++++-- .../py/nvidia/tests/test_parakeet_slicing.py | 329 ++++++++++++++++++ 2 files changed, 525 insertions(+), 25 deletions(-) create mode 100644 internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py diff --git a/internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py b/internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py index 29d5047..ba755c1 100644 --- a/internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py +++ b/internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py @@ -2,6 +2,10 @@ """ NVIDIA Parakeet buffered inference for long audio files. Splits audio into chunks to avoid GPU memory issues. + +Adjacent chunks overlap slightly, and in each overlap the chunks hand over at +a word both transcribed alike, so a word near a cut is neither chopped, dropped +nor repeated. Optionally (--pause-search) each cut also moves into a pause. """ import argparse @@ -12,37 +16,178 @@ import librosa import soundfile as sf import numpy as np from pathlib import Path -import nemo.collections.asr as nemo_asr +DEFAULT_OVERLAP_SECS = 4.0 +DEFAULT_PAUSE_SEARCH_SECS = 0.0 # opt-in; measured no gain on top of the overlap +QUIET_WINDOW_SECS = 0.3 +SAME_WORD_SECS = 0.5 + + +def plan_slices(audio, sr, max_chunk_secs, overlap_secs=0.0, search_secs=0.0): + """Choose where to cut `audio` so no chunk exceeds `max_chunk_secs`. -def split_audio_file(audio_path, chunk_duration_secs=300): - """Split audio file into chunks of specified duration.""" + Returns (spans, cuts): `cuts` are the sample indices where one chunk's + share of the audio ends and the next one's begins; `spans` are the + (start, end) samples actually transcribed, each cut-to-cut range widened + by half the overlap on both sides. The overlap counts towards the limit. + + With `search_secs` > 0, each cut moves back from the limit to the middle + of the quietest QUIET_WINDOW_SECS window within the last `search_secs`. + """ + overlap_secs = max(0.0, min(overlap_secs, max_chunk_secs / 4)) + step = int((max_chunk_secs - overlap_secs) * sr) + if step < 1: + raise ValueError(f"chunk length must be positive, got {max_chunk_secs}s") + search = min(int(search_secs * sr), step // 2) + window = max(1, int(QUIET_WINDOW_SECS * sr)) + + cuts = [] + position = 0 + while len(audio) - position > step: + cut = position + step + if search > window: + cut = _quietest_point(audio, cut - search, cut, window) + cuts.append(cut) + position = cut + + pad = int(overlap_secs * sr) // 2 + edges = [0] + cuts + [len(audio)] + spans = [(max(0, start - pad), min(len(audio), end + pad)) + for start, end in zip(edges, edges[1:])] + return spans, cuts + + +def _quietest_point(audio, start, end, window): + """Middle of the lowest-energy `window` samples within audio[start:end]. + + Ties go to the latest window, which keeps chunks as long as allowed. + """ + x = audio[start:end].astype(np.float64) + cumulative = np.concatenate(([0.0], np.cumsum(x * x))) + energy = cumulative[window:] - cumulative[:-window] + latest_min = len(energy) - 1 - int(np.argmin(energy[::-1])) + return start + latest_min + window // 2 + + +def stitch_slices(slice_results, cut_times, chunk_spans=None): + """Merge per-chunk (words, segments), already shifted to absolute time. + + `chunk_spans` gives each chunk's (start, end) in seconds; omit it when the + chunks do not overlap. Each chunk contributes the words between its two + handovers. A handover is at the cut, unless the chunks overlap: then it + moves to the nearest word in the overlap that both chunks transcribed + alike, at the same time, and the left chunk keeps the words before it, + the right chunk that word and the ones after. (Splitting both chunks at + the cut is fragile: a word that follows a pause can be timestamped + anywhere in the pause, so the two chunks may put it on opposite sides of + the cut and keep it twice, or not at all.) + + Segments are trimmed to the words their chunk keeps, and dropped if none. + """ + first = [0] * len(slice_results) + last = [len(chunk_words) for chunk_words, _ in slice_results] + for k, cut in enumerate(cut_times): + overlap = (chunk_spans[k + 1][0], chunk_spans[k][1]) if chunk_spans else (cut, cut) + last[k], first[k + 1] = _handover( + slice_results[k][0], slice_results[k + 1][0], cut, overlap + ) + + words, segments = [], [] + for (chunk_words, chunk_segments), lo, hi in zip(slice_results, first, last): + words.extend(chunk_words[lo:hi]) + for seg, (start, stop) in zip(chunk_segments, _segment_ranges(chunk_words, chunk_segments)): + kept = chunk_words[max(start, lo):min(stop, hi)] + if start == stop: # a segment without words: keep it where its chunk does + if lo <= start < hi: + segments.append(seg) + elif len(kept) == stop - start: + segments.append(seg) + elif kept: + segments.append({ + **seg, + "segment": " ".join(w["word"] for w in kept), + "start_offset": kept[0]["start_offset"], + "end_offset": kept[-1]["end_offset"], + "start": kept[0]["start"], + "end": kept[-1]["end"], + }) + return words, segments + + +def _handover(left, right, cut, overlap): + """(i, j): the left chunk keeps left[:i] and the right chunk right[j:]. + + Anchors are words in the overlap that both chunks transcribed with the same + text at nearly the same time. At the anchor nearest the cut, the right + chunk's copy is kept, or the left chunk's if that is what keeps the words + in time order. Without an anchor, both chunks split at the cut. + """ + i = sum(1 for w in left if w["start"] < cut) + j = sum(1 for w in right if w["start"] < cut) + in_order = lambda a, b: a == 0 or b == len(right) or left[a - 1]["start"] <= right[b]["start"] + anchors = [] + for p, lw in enumerate(left): + text = _normalize(lw["word"]) + if not text or not overlap[0] <= lw["start"] < overlap[1]: + continue + partners = [q for q, rw in enumerate(right) if _normalize(rw["word"]) == text + and abs(rw["start"] - lw["start"]) <= SAME_WORD_SECS] + if partners: + q = min(partners, key=lambda q: abs(right[q]["start"] - lw["start"])) + options = [h for h in ((p, q), (p + 1, q + 1)) if in_order(*h)] + if options: + anchors.append((abs(lw["start"] + right[q]["start"] - 2 * cut), options[0])) + if anchors: + i, j = min(anchors)[1] + return i, j + + +def _normalize(word): + return "".join(c for c in word.lower() if c.isalnum() or c == "'") + + +def _segment_ranges(words, segments): + """[start, stop) word indices of each segment, matched in order by frame offsets.""" + ranges, i = [], 0 + for seg in segments: + while i < len(words) and words[i]["start_offset"] < seg["start_offset"]: + i += 1 + start = i + while i < len(words) and words[i]["end_offset"] <= seg["end_offset"]: + i += 1 + ranges.append((start, i)) + return ranges + + +def split_audio_file(audio_path, chunk_duration_secs=300, overlap_secs=0.0, search_secs=0.0): + """Split audio file into chunks of at most chunk_duration_secs.""" audio, sr = librosa.load(audio_path, sr=None, mono=True) - total_duration = len(audio) / sr - chunk_samples = int(chunk_duration_secs * sr) + spans, cuts = plan_slices(audio, sr, chunk_duration_secs, overlap_secs, search_secs) chunks = [] - for start_sample in range(0, len(audio), chunk_samples): - end_sample = min(start_sample + chunk_samples, len(audio)) + for start_sample, end_sample in spans: chunk_audio = audio[start_sample:end_sample] - start_time = start_sample / sr chunks.append({ 'audio': chunk_audio, - 'start_time': start_time, + 'start_time': start_sample / sr, 'duration': len(chunk_audio) / sr }) - return chunks, sr + return chunks, sr, [cut / sr for cut in cuts] def transcribe_buffered( audio_path: str, output_file: str = None, chunk_duration_secs: float = 300, # 5 minutes default + overlap_secs: float = DEFAULT_OVERLAP_SECS, + pause_search_secs: float = DEFAULT_PAUSE_SEARCH_SECS, ): """ Transcribe long audio by splitting into chunks and merging results. """ + import nemo.collections.asr as nemo_asr + # Determine model path model_filename = "parakeet-tdt-0.6b-v3.nemo" model_path = None @@ -79,13 +224,15 @@ def transcribe_buffered( asr_model.change_decoding_strategy(dec_cfg) print("✓ CUDA graphs disabled successfully") - print(f"Splitting audio into {chunk_duration_secs}s chunks...") - chunks, sr = split_audio_file(audio_path, chunk_duration_secs) + print(f"Splitting audio into chunks of at most {chunk_duration_secs}s " + f"(overlap {overlap_secs}s, pause search {pause_search_secs}s)...") + chunks, sr, cut_times = split_audio_file( + audio_path, chunk_duration_secs, overlap_secs, pause_search_secs + ) print(f"Created {len(chunks)} chunks") - all_words = [] - all_segments = [] - full_text = [] + slice_results = [] + chunk_texts = [] for i, chunk_info in enumerate(chunks): print(f"Transcribing chunk {i+1}/{len(chunks)} (duration: {chunk_info['duration']:.1f}s)...") @@ -104,25 +251,26 @@ def transcribe_buffered( result_data = output[0] chunk_text = result_data.text - full_text.append(chunk_text) + chunk_texts.append(chunk_text) + chunk_words = [] + chunk_segments = [] # Extract and adjust timestamps if hasattr(result_data, 'timestamp') and result_data.timestamp: - chunk_words = result_data.timestamp.get("word", []) - chunk_segments = result_data.timestamp.get("segment", []) - # Adjust timestamps by chunk start time - for word in chunk_words: + for word in result_data.timestamp.get("word", []): word_copy = dict(word) word_copy['start'] += chunk_info['start_time'] word_copy['end'] += chunk_info['start_time'] - all_words.append(word_copy) + chunk_words.append(word_copy) - for segment in chunk_segments: + for segment in result_data.timestamp.get("segment", []): seg_copy = dict(segment) seg_copy['start'] += chunk_info['start_time'] seg_copy['end'] += chunk_info['start_time'] - all_segments.append(seg_copy) + chunk_segments.append(seg_copy) + + slice_results.append((chunk_words, chunk_segments)) print(f"Chunk {i+1} complete: {len(chunk_text)} characters") @@ -131,7 +279,15 @@ def transcribe_buffered( if os.path.exists(chunk_path): os.remove(chunk_path) - final_text = " ".join(full_text) + chunk_spans = [(c['start_time'], c['start_time'] + c['duration']) for c in chunks] + all_words, all_segments = stitch_slices(slice_results, cut_times, chunk_spans) + if any(text.strip() and not words for (words, _), text in zip(slice_results, chunk_texts)): + # A chunk came back without word timestamps, so there is nothing to + # stitch it by; keep its text rather than lose it. + print("Warning: a chunk has text but no word timestamps; joining chunk texts") + final_text = " ".join(chunk_texts) + else: + final_text = " ".join(w["word"] for w in all_words) print(f"Transcription complete: {len(final_text)} characters total") output_data = { @@ -144,6 +300,9 @@ def transcribe_buffered( "buffered": True, "chunk_duration_secs": chunk_duration_secs, "num_chunks": len(chunks), + "overlap_secs": overlap_secs, + "pause_search_secs": pause_search_secs, + "cut_times": cut_times, } if output_file: @@ -162,7 +321,17 @@ def main(): parser.add_argument("--output", "-o", help="Output file path", required=True) parser.add_argument( "--chunk-len", type=float, default=300, - help="Chunk duration in seconds (default: 300 = 5 minutes)" + help="Maximum chunk duration in seconds, overlap included (default: 300 = 5 minutes)" + ) + parser.add_argument( + "--overlap", type=float, default=DEFAULT_OVERLAP_SECS, + help=f"Seconds shared by adjacent chunks, capped at a quarter of --chunk-len " + f"(default: {DEFAULT_OVERLAP_SECS}; 0 disables)" + ) + parser.add_argument( + "--pause-search", type=float, default=DEFAULT_PAUSE_SEARCH_SECS, + help=f"Seconds before each chunk limit searched for the quietest point to cut at, " + f"e.g. 25 (default: {DEFAULT_PAUSE_SEARCH_SECS}, cut at the limit)" ) args = parser.parse_args() @@ -175,6 +344,8 @@ def main(): audio_path=args.audio_file, output_file=args.output, chunk_duration_secs=args.chunk_len, + overlap_secs=args.overlap, + pause_search_secs=args.pause_search, ) diff --git a/internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py b/internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py new file mode 100644 index 0000000..6a35947 --- /dev/null +++ b/internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py @@ -0,0 +1,329 @@ +"""Unit tests for the slicing and stitching helpers in parakeet_transcribe_buffered.py. + +These are pure functions: they need numpy, librosa and soundfile (imported by the +script) but no GPU, no NeMo and no model. +""" +import sys +from pathlib import Path + +import numpy as np +import pytest + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +from parakeet_transcribe_buffered import plan_slices, stitch_slices # noqa: E402 + +SR = 16000 + + +def speech(seconds, seed=0): + """Stand-in for continuous speech: broadband noise at a speech-like level.""" + rng = np.random.default_rng(seed) + return (0.1 * rng.standard_normal(int(seconds * SR))).astype(np.float32) + + +def with_pauses(audio, pauses, level=0.001, seed=1): + """Replace each (start_s, end_s) span with low-level room noise (or zeros).""" + rng = np.random.default_rng(seed) + out = audio.copy() + for start, end in pauses: + a, b = int(start * SR), int(end * SR) + out[a:b] = level * rng.standard_normal(b - a) + return out + + +def assert_valid_plan(spans, cuts, num_samples, max_secs): + assert spans[0][0] == 0 and spans[-1][1] == num_samples + assert len(spans) == len(cuts) + 1 + for start, end in spans: + assert 0 < end - start <= max_secs * SR + for (a0, a1), (b0, b1), cut in zip(spans, spans[1:], cuts): + assert b0 <= cut <= a1, "each cut must lie inside both neighbouring slices" + assert a0 < cut < b1 + + +# -- plan_slices: cut placement ------------------------------------------------- + + +def test_audio_shorter_than_one_slice_is_not_cut(): + audio = speech(60) + spans, cuts = plan_slices(audio, SR, 120, overlap_secs=4, search_secs=25) + assert cuts == [] + assert spans == [(0, len(audio))] + + +def test_audio_exactly_one_slice_long_is_not_cut(): + audio = speech(120) + spans, cuts = plan_slices(audio, SR, 120) + assert cuts == [] + assert spans == [(0, len(audio))] + + +def test_without_search_or_overlap_the_legacy_fixed_grid_is_reproduced(): + audio = speech(300) + spans, cuts = plan_slices(audio, SR, 120, overlap_secs=0, search_secs=0) + assert cuts == [120 * SR, 240 * SR] + assert spans == [(0, 120 * SR), (120 * SR, 240 * SR), (240 * SR, 300 * SR)] + + +def test_cuts_land_in_the_pauses_before_the_limit(): + pauses = [(100.0, 100.5), (215.0, 215.5), (330.0, 330.5)] + audio = with_pauses(speech(400), pauses) + spans, cuts = plan_slices(audio, SR, 120, overlap_secs=0, search_secs=25) + assert len(cuts) == 3 + for cut, (start, end) in zip(cuts, pauses): + assert start <= cut / SR <= end + assert_valid_plan(spans, cuts, len(audio), 120) + + +def test_the_quietest_pause_wins(): + # Two pauses inside the same search window; the later one is louder. + audio = with_pauses(speech(200), [(100.0, 100.6)], level=0.0) + audio = with_pauses(audio, [(115.0, 115.6)], level=0.01) + _, cuts = plan_slices(audio, SR, 120, search_secs=25) + assert 100.0 <= cuts[0] / SR <= 100.6 + + +def test_audio_with_no_pause_is_still_cut_within_the_limit(): + audio = speech(400) + spans, cuts = plan_slices(audio, SR, 120, overlap_secs=0, search_secs=25) + assert_valid_plan(spans, cuts, len(audio), 120) + edges = [0] + cuts + for prev, cut in zip(edges, cuts): + assert 95 * SR <= cut - prev <= 120 * SR, "cut must fall inside its search window" + + +def test_a_pause_at_the_very_start_is_never_a_cut(): + audio = with_pauses(speech(200), [(0.0, 5.0)], level=0.0) + spans, cuts = plan_slices(audio, SR, 120, search_secs=25) + assert len(cuts) == 1 and 95 <= cuts[0] / SR <= 120 + assert_valid_plan(spans, cuts, len(audio), 120) + + +def test_a_pause_at_the_very_end_leaves_no_empty_slice(): + # First cut ~100.35 s, so the second search window is ~[195, 220] s and + # holds the start of the trailing silence (217-222 s). + audio = with_pauses(speech(222), [(100.0, 100.5), (217.0, 222.0)], level=0.0) + spans, cuts = plan_slices(audio, SR, 120, search_secs=25) + assert len(cuts) == 2 + assert 217.0 <= cuts[1] / SR < 222.0 + assert_valid_plan(spans, cuts, len(audio), 120) + + +# -- plan_slices: overlap ------------------------------------------------------- + + +def test_overlap_is_included_in_the_slice_limit_and_centred_on_each_cut(): + pauses = [(100.0, 100.5), (215.0, 215.5), (330.0, 330.5)] + audio = with_pauses(speech(400), pauses) + spans, cuts = plan_slices(audio, SR, 120, overlap_secs=4, search_secs=25) + assert_valid_plan(spans, cuts, len(audio), 120) + for (_, a1), (b0, _), cut in zip(spans, spans[1:], cuts): + assert a1 - b0 == 4 * SR + assert cut - b0 == a1 - cut + for cut, (start, end) in zip(cuts, pauses): + assert start <= cut / SR <= end + + +def test_overlap_without_pause_search_uses_a_fixed_grid(): + audio = speech(300) + spans, cuts = plan_slices(audio, SR, 120, overlap_secs=4, search_secs=0) + assert cuts == [116 * SR, 232 * SR] + assert spans == [(0, 118 * SR), (114 * SR, 234 * SR), (230 * SR, 300 * SR)] + + +def test_short_slice_lengths_clamp_overlap_and_search(): + # Upstream's own buffered test runs a 19 s clip with --chunk-len 10. + audio = speech(19) + spans, cuts = plan_slices(audio, SR, 10, overlap_secs=4, search_secs=25) + assert len(spans) >= 2 + assert_valid_plan(spans, cuts, len(audio), 10) + + +# -- stitch_slices ------------------------------------------------------------------ + +FRAME = 0.08 +OVERLAPPING = [(0.0, 12.0), (8.0, 20.0)] # two chunks sharing 8-12 s, cut at 10 s + + +def word(text, start, end, slice_start): + return { + "word": text, + "start_offset": round((start - slice_start) / FRAME), + "end_offset": round((end - slice_start) / FRAME), + "start": start, + "end": end, + } + + +def segment(words): + return { + "segment": " ".join(w["word"] for w in words), + "start_offset": words[0]["start_offset"], + "end_offset": words[-1]["end_offset"], + "start": words[0]["start"], + "end": words[-1]["end"], + } + + +def texts(items, key="word"): + return [item[key] for item in items] + + +def test_no_overlap_stitching_is_plain_concatenation(): + left = [word("one", 1.0, 1.4, 0), word("two", 5.0, 5.3, 0)] + right = [word("three", 10.5, 10.9, 10), word("four", 14.0, 14.4, 10)] + words, segments = stitch_slices( + [(left, [segment(left)]), (right, [segment(right)])], [10.0] + ) + assert words == left + right + assert segments == [segment(left), segment(right)] + + +def test_overlapping_slices_keep_every_word_exactly_once(): + # Slices [0, 12] and [8, 20], cut at 10. Both transcribe the overlap and + # their timestamps for the same word differ by a few ms. + left = [ + word("a", 1.0, 1.3, 0), + word("b", 5.0, 5.4, 0), + word("c", 9.00, 9.40, 0), + word("d", 10.50, 10.90, 0), + word("e", 11.50, 11.80, 0), + ] + right = [ + word("c", 9.02, 9.40, 8), + word("d", 10.48, 10.90, 8), + word("e", 11.52, 11.80, 8), + word("f", 15.00, 15.40, 8), + ] + words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING) + assert texts(words) == ["a", "b", "c", "d", "e", "f"] + assert words[2] is left[2] and words[3] is right[1] + + +def test_a_word_straddling_the_cut_is_kept_once(): + left = [word("over", 9.90, 10.30, 0), word("the", 10.40, 10.55, 0)] + right = [word("over", 9.92, 10.30, 8), word("the", 10.40, 10.55, 8)] + words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING) + assert texts(words) == ["over", "the"] + # Both chunks agree on it, so the right chunk takes over from it. + assert words[0] is right[0] and words[1] is right[1] + + +def test_a_segment_straddling_the_cut_is_trimmed_to_the_words_each_slice_owns(): + left_words = [ + word("Hello", 8.5, 8.9, 0), + word("there", 9.2, 9.6, 0), + word("friend.", 10.4, 10.9, 0), + ] + right_words = [ + word("there", 9.21, 9.6, 8), + word("friend.", 10.41, 10.9, 8), + word("Bye.", 13.0, 13.4, 8), + ] + words, segments = stitch_slices( + [ + (left_words, [segment(left_words)]), + (right_words, [segment(right_words[:2]), segment(right_words[2:])]), + ], + [10.0], + OVERLAPPING, + ) + assert texts(words) == ["Hello", "there", "friend.", "Bye."] + assert texts(segments, "segment") == ["Hello there", "friend.", "Bye."] + assert segments[0]["end"] == 9.6 and segments[1]["start"] == 10.41 + # A segment that needed no trimming is passed through untouched. + assert segments[2] == segment(right_words[2:]) + # Every word appears in exactly one segment, in order. + assert " ".join(texts(segments, "segment")) == " ".join(texts(words)) + + +def test_a_segment_wholly_inside_the_other_slices_share_is_dropped(): + left_words = [word("a", 2.0, 2.3, 0), word("b.", 10.6, 11.0, 0)] + right_words = [word("b.", 10.61, 11.0, 8), word("c", 12.0, 12.3, 8)] + _, segments = stitch_slices( + [ + (left_words, [segment(left_words[:1]), segment(left_words[1:])]), + (right_words, [segment(right_words[:1]), segment(right_words[1:])]), + ], + [10.0], + OVERLAPPING, + ) + assert texts(segments, "segment") == ["a", "b.", "c"] + assert segments[1]["start"] == 10.61 + + +def test_a_non_positive_chunk_length_is_rejected_rather_than_looping(): + with pytest.raises(ValueError): + plan_slices(speech(5), SR, 0) + + +def test_a_word_the_two_chunks_timestamp_either_side_of_the_cut_is_kept_once(): + # After a pause TDT may place a word's start anywhere in the pause, so the + # two chunks can disagree about which side of the cut it starts on. + left = [word("so", 8.2, 8.5, 0), word("then", 9.98, 10.3, 0), word("we", 10.4, 10.6, 0)] + right = [word("so", 8.2, 8.5, 8), word("then", 10.03, 10.3, 8), word("we", 10.4, 10.6, 8)] + words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING) + assert texts(words) == ["so", "then", "we"] + # ...and the mirror image, where splitting both at the cut would drop it. + left = [word("so", 8.2, 8.5, 0), word("then", 10.03, 10.3, 0), word("we", 10.4, 10.6, 0)] + right = [word("so", 8.2, 8.5, 8), word("then", 9.98, 10.3, 8), word("we", 10.4, 10.6, 8)] + words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING) + assert texts(words) == ["so", "then", "we"] + + +def test_handover_happens_at_the_agreed_word_nearest_the_cut(): + # The chunks differ in casing/punctuation and the left one drops "really" + # near its end; the right chunk's version of the overlap after the cut wins. + left = [word("It", 8.5, 8.7, 0), word("was", 9.6, 9.9, 0), word("good,", 11.0, 11.4, 0)] + right = [word("it", 8.5, 8.7, 8), word("was", 9.62, 9.9, 8), word("really", 10.3, 10.7, 8), + word("good.", 11.0, 11.4, 8)] + words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING) + assert texts(words) == ["It", "was", "really", "good."] + assert words[0] is left[0] and words[1] is right[1] + + +def test_without_an_agreed_word_the_split_falls_back_to_the_cut(): + left = [word("alpha", 9.0, 9.4, 0), word("beta", 10.5, 10.9, 0)] + right = [word("gamma", 9.1, 9.4, 8), word("delta", 10.6, 10.9, 8)] + words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING) + assert texts(words) == ["alpha", "delta"] + + +def test_another_occurrence_of_the_word_elsewhere_in_the_overlap_is_not_an_anchor(): + # The chunks disagree everywhere except on "the", but the left chunk's + # "the" (9.0 s) and the right chunk's (11.0 s) are different words. As an + # anchor they would average to the cut and discard the left one. + left = [word("the", 9.0, 9.2, 0), word("dog", 10.5, 10.8, 0)] + right = [word("cat", 9.3, 9.6, 8), word("the", 11.0, 11.2, 8)] + words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING) + assert [w["start"] for w in words] == [9.0, 11.0] + + +def test_the_handover_never_puts_words_out_of_time_order(): + # "y" agrees (0.45 s apart), but the right chunk's "y" (9.65) after the + # left chunk's "x" (9.70) would run time backwards, so the left chunk's + # copy is kept instead. Splitting at the cut would lose "y" altogether. + left = [word("x", 9.70, 9.90, 0), word("y", 10.10, 10.30, 0)] + right = [word("z", 9.40, 9.60, 8), word("y", 9.65, 9.90, 8), word("w", 10.6, 10.8, 8)] + words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING) + assert texts(words) == ["x", "y", "w"] and words[1] is left[1] + starts = [w["start"] for w in words] + assert starts == sorted(starts) + + +def test_a_co_timed_anchor_is_found_even_when_a_longer_match_lies_elsewhere(): + # "x y" recurs later in the right chunk, a longer text match than "z", but + # at a different time. Only "z" is the same word in both chunks, and it + # straddles the cut, so splitting both at the cut would keep it twice. + left = [word("x", 8.2, 8.3, 0), word("y", 8.4, 8.5, 0), word("z", 9.98, 10.2, 0)] + right = [word("z", 10.03, 10.2, 8), word("x", 11.0, 11.1, 8), word("y", 11.2, 11.3, 8)] + words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING) + assert texts(words) == ["x", "y", "z", "x", "y"] + + +def test_punctuation_alone_is_never_an_anchor(): + # As an anchor the dash would hand the whole overlap to the right chunk. + left = [word("-", 9.50, 9.55, 0), word("yes", 10.4, 10.6, 0)] + right = [word("-", 9.52, 9.55, 8), word("no", 10.4, 10.6, 8)] + words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING) + assert words[0] is left[0] and texts(words) == ["-", "no"] -- 2.39.5