Carry patches/0001 on our Scriberr build (upstream a353078): adjacent buffered chunks overlap by 4 s inside --chunk-len and hand over at a word both chunks transcribed alike, instead of cutting at fixed marks with no overlap. Pause-aware cutting is included as an opt-in (--pause-search); it measured neutral once the stitch was right. The Go<->Python CLI and JSON seam is unchanged. Bench (4 recordings, 118 min, 3 cut placements each, against a no-cut whole-file reference; metrics only, private audio stays on fv-ml1): cuts with an error within +-3 s fall from 52% (93/179) to 22% (41/184) against a 19% background; floor +-0.08. Positive control: upstream's cutter +0.33 over background. A-vs-A byte-identical in-process and across CLI processes. Peak GPU memory unchanged at 5,496 MiB (n=3). Also found: Parakeet skips runs of >=10 words mid-chunk with any slicer, upstream's included; not addressed here. scripts/scriberr-rebuild clones a pinned upstream sha into a new /opt/docker/src dir, git-apply-checks the patches, builds a distinct tag, and checks embed, unit tests, the JSON seam (scriberr-seam-check.py) and the memory budget on idle GPU 3. Deploy stays manual. The upstream PR is prepared under patches/upstream-pr/ and not opened.
687 lines
29 KiB
Diff
687 lines
29 KiB
Diff
From 2dafe7ce81c217609ff2c8616e43b0d72255e170 Mon Sep 17 00:00:00 2001
|
|
From: Vuong Hoang <vh@phasefinal.com>
|
|
Date: Wed, 30 Sep 2026 11:41:07 -0700
|
|
Subject: [PATCH] fix(parakeet): overlap buffered chunks and stitch at an
|
|
agreed word
|
|
|
|
parakeet_transcribe_buffered.py cut long audio at fixed --chunk-len marks
|
|
with no overlap, so a word straddling a mark was chopped, dropped or
|
|
transcribed twice. Adjacent chunks now overlap by --overlap seconds
|
|
(default 4, counted inside --chunk-len so no chunk grows), and in each
|
|
overlap the chunks hand over at the word nearest the cut that both
|
|
transcribed with the same text at nearly the same time (within 0.5 s),
|
|
keeping whichever copy leaves the words in time order. With no such word
|
|
they split at the cut. Splitting both chunks at the cut by word start time
|
|
is not enough on its own: a word after a pause can be timestamped anywhere
|
|
in the pause, so the two chunks may place it on opposite sides of the cut.
|
|
|
|
--pause-search N (opt-in) also moves each cut back to the quietest 0.3 s
|
|
in the last N seconds before the limit. Measured neutral on top of the
|
|
overlap, so it is off by default.
|
|
|
|
The CLI and JSON the Go adapter reads are unchanged; the new flags are
|
|
optional, and the JSON gains overlap_secs, pause_search_secs and
|
|
cut_times. --overlap 0 reproduces the previous output exactly. NeMo is
|
|
now imported inside transcribe_buffered() so the slicing and stitching
|
|
helpers can be unit-tested without a GPU. If a chunk ever returns text
|
|
without word timestamps, its text is kept rather than dropped.
|
|
---
|
|
.../py/nvidia/parakeet_transcribe_buffered.py | 221 ++++++++++--
|
|
.../py/nvidia/tests/test_parakeet_slicing.py | 329 ++++++++++++++++++
|
|
2 files changed, 525 insertions(+), 25 deletions(-)
|
|
create mode 100644 internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py
|
|
|
|
diff --git a/internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py b/internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py
|
|
index 29d5047..ba755c1 100644
|
|
--- a/internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py
|
|
+++ b/internal/transcription/adapters/py/nvidia/parakeet_transcribe_buffered.py
|
|
@@ -2,6 +2,10 @@
|
|
"""
|
|
NVIDIA Parakeet buffered inference for long audio files.
|
|
Splits audio into chunks to avoid GPU memory issues.
|
|
+
|
|
+Adjacent chunks overlap slightly, and in each overlap the chunks hand over at
|
|
+a word both transcribed alike, so a word near a cut is neither chopped, dropped
|
|
+nor repeated. Optionally (--pause-search) each cut also moves into a pause.
|
|
"""
|
|
|
|
import argparse
|
|
@@ -12,37 +16,178 @@ import librosa
|
|
import soundfile as sf
|
|
import numpy as np
|
|
from pathlib import Path
|
|
-import nemo.collections.asr as nemo_asr
|
|
|
|
+DEFAULT_OVERLAP_SECS = 4.0
|
|
+DEFAULT_PAUSE_SEARCH_SECS = 0.0 # opt-in; measured no gain on top of the overlap
|
|
+QUIET_WINDOW_SECS = 0.3
|
|
+SAME_WORD_SECS = 0.5
|
|
+
|
|
+
|
|
+def plan_slices(audio, sr, max_chunk_secs, overlap_secs=0.0, search_secs=0.0):
|
|
+ """Choose where to cut `audio` so no chunk exceeds `max_chunk_secs`.
|
|
|
|
-def split_audio_file(audio_path, chunk_duration_secs=300):
|
|
- """Split audio file into chunks of specified duration."""
|
|
+ Returns (spans, cuts): `cuts` are the sample indices where one chunk's
|
|
+ share of the audio ends and the next one's begins; `spans` are the
|
|
+ (start, end) samples actually transcribed, each cut-to-cut range widened
|
|
+ by half the overlap on both sides. The overlap counts towards the limit.
|
|
+
|
|
+ With `search_secs` > 0, each cut moves back from the limit to the middle
|
|
+ of the quietest QUIET_WINDOW_SECS window within the last `search_secs`.
|
|
+ """
|
|
+ overlap_secs = max(0.0, min(overlap_secs, max_chunk_secs / 4))
|
|
+ step = int((max_chunk_secs - overlap_secs) * sr)
|
|
+ if step < 1:
|
|
+ raise ValueError(f"chunk length must be positive, got {max_chunk_secs}s")
|
|
+ search = min(int(search_secs * sr), step // 2)
|
|
+ window = max(1, int(QUIET_WINDOW_SECS * sr))
|
|
+
|
|
+ cuts = []
|
|
+ position = 0
|
|
+ while len(audio) - position > step:
|
|
+ cut = position + step
|
|
+ if search > window:
|
|
+ cut = _quietest_point(audio, cut - search, cut, window)
|
|
+ cuts.append(cut)
|
|
+ position = cut
|
|
+
|
|
+ pad = int(overlap_secs * sr) // 2
|
|
+ edges = [0] + cuts + [len(audio)]
|
|
+ spans = [(max(0, start - pad), min(len(audio), end + pad))
|
|
+ for start, end in zip(edges, edges[1:])]
|
|
+ return spans, cuts
|
|
+
|
|
+
|
|
+def _quietest_point(audio, start, end, window):
|
|
+ """Middle of the lowest-energy `window` samples within audio[start:end].
|
|
+
|
|
+ Ties go to the latest window, which keeps chunks as long as allowed.
|
|
+ """
|
|
+ x = audio[start:end].astype(np.float64)
|
|
+ cumulative = np.concatenate(([0.0], np.cumsum(x * x)))
|
|
+ energy = cumulative[window:] - cumulative[:-window]
|
|
+ latest_min = len(energy) - 1 - int(np.argmin(energy[::-1]))
|
|
+ return start + latest_min + window // 2
|
|
+
|
|
+
|
|
+def stitch_slices(slice_results, cut_times, chunk_spans=None):
|
|
+ """Merge per-chunk (words, segments), already shifted to absolute time.
|
|
+
|
|
+ `chunk_spans` gives each chunk's (start, end) in seconds; omit it when the
|
|
+ chunks do not overlap. Each chunk contributes the words between its two
|
|
+ handovers. A handover is at the cut, unless the chunks overlap: then it
|
|
+ moves to the nearest word in the overlap that both chunks transcribed
|
|
+ alike, at the same time, and the left chunk keeps the words before it,
|
|
+ the right chunk that word and the ones after. (Splitting both chunks at
|
|
+ the cut is fragile: a word that follows a pause can be timestamped
|
|
+ anywhere in the pause, so the two chunks may put it on opposite sides of
|
|
+ the cut and keep it twice, or not at all.)
|
|
+
|
|
+ Segments are trimmed to the words their chunk keeps, and dropped if none.
|
|
+ """
|
|
+ first = [0] * len(slice_results)
|
|
+ last = [len(chunk_words) for chunk_words, _ in slice_results]
|
|
+ for k, cut in enumerate(cut_times):
|
|
+ overlap = (chunk_spans[k + 1][0], chunk_spans[k][1]) if chunk_spans else (cut, cut)
|
|
+ last[k], first[k + 1] = _handover(
|
|
+ slice_results[k][0], slice_results[k + 1][0], cut, overlap
|
|
+ )
|
|
+
|
|
+ words, segments = [], []
|
|
+ for (chunk_words, chunk_segments), lo, hi in zip(slice_results, first, last):
|
|
+ words.extend(chunk_words[lo:hi])
|
|
+ for seg, (start, stop) in zip(chunk_segments, _segment_ranges(chunk_words, chunk_segments)):
|
|
+ kept = chunk_words[max(start, lo):min(stop, hi)]
|
|
+ if start == stop: # a segment without words: keep it where its chunk does
|
|
+ if lo <= start < hi:
|
|
+ segments.append(seg)
|
|
+ elif len(kept) == stop - start:
|
|
+ segments.append(seg)
|
|
+ elif kept:
|
|
+ segments.append({
|
|
+ **seg,
|
|
+ "segment": " ".join(w["word"] for w in kept),
|
|
+ "start_offset": kept[0]["start_offset"],
|
|
+ "end_offset": kept[-1]["end_offset"],
|
|
+ "start": kept[0]["start"],
|
|
+ "end": kept[-1]["end"],
|
|
+ })
|
|
+ return words, segments
|
|
+
|
|
+
|
|
+def _handover(left, right, cut, overlap):
|
|
+ """(i, j): the left chunk keeps left[:i] and the right chunk right[j:].
|
|
+
|
|
+ Anchors are words in the overlap that both chunks transcribed with the same
|
|
+ text at nearly the same time. At the anchor nearest the cut, the right
|
|
+ chunk's copy is kept, or the left chunk's if that is what keeps the words
|
|
+ in time order. Without an anchor, both chunks split at the cut.
|
|
+ """
|
|
+ i = sum(1 for w in left if w["start"] < cut)
|
|
+ j = sum(1 for w in right if w["start"] < cut)
|
|
+ in_order = lambda a, b: a == 0 or b == len(right) or left[a - 1]["start"] <= right[b]["start"]
|
|
+ anchors = []
|
|
+ for p, lw in enumerate(left):
|
|
+ text = _normalize(lw["word"])
|
|
+ if not text or not overlap[0] <= lw["start"] < overlap[1]:
|
|
+ continue
|
|
+ partners = [q for q, rw in enumerate(right) if _normalize(rw["word"]) == text
|
|
+ and abs(rw["start"] - lw["start"]) <= SAME_WORD_SECS]
|
|
+ if partners:
|
|
+ q = min(partners, key=lambda q: abs(right[q]["start"] - lw["start"]))
|
|
+ options = [h for h in ((p, q), (p + 1, q + 1)) if in_order(*h)]
|
|
+ if options:
|
|
+ anchors.append((abs(lw["start"] + right[q]["start"] - 2 * cut), options[0]))
|
|
+ if anchors:
|
|
+ i, j = min(anchors)[1]
|
|
+ return i, j
|
|
+
|
|
+
|
|
+def _normalize(word):
|
|
+ return "".join(c for c in word.lower() if c.isalnum() or c == "'")
|
|
+
|
|
+
|
|
+def _segment_ranges(words, segments):
|
|
+ """[start, stop) word indices of each segment, matched in order by frame offsets."""
|
|
+ ranges, i = [], 0
|
|
+ for seg in segments:
|
|
+ while i < len(words) and words[i]["start_offset"] < seg["start_offset"]:
|
|
+ i += 1
|
|
+ start = i
|
|
+ while i < len(words) and words[i]["end_offset"] <= seg["end_offset"]:
|
|
+ i += 1
|
|
+ ranges.append((start, i))
|
|
+ return ranges
|
|
+
|
|
+
|
|
+def split_audio_file(audio_path, chunk_duration_secs=300, overlap_secs=0.0, search_secs=0.0):
|
|
+ """Split audio file into chunks of at most chunk_duration_secs."""
|
|
audio, sr = librosa.load(audio_path, sr=None, mono=True)
|
|
- total_duration = len(audio) / sr
|
|
- chunk_samples = int(chunk_duration_secs * sr)
|
|
+ spans, cuts = plan_slices(audio, sr, chunk_duration_secs, overlap_secs, search_secs)
|
|
|
|
chunks = []
|
|
- for start_sample in range(0, len(audio), chunk_samples):
|
|
- end_sample = min(start_sample + chunk_samples, len(audio))
|
|
+ for start_sample, end_sample in spans:
|
|
chunk_audio = audio[start_sample:end_sample]
|
|
- start_time = start_sample / sr
|
|
chunks.append({
|
|
'audio': chunk_audio,
|
|
- 'start_time': start_time,
|
|
+ 'start_time': start_sample / sr,
|
|
'duration': len(chunk_audio) / sr
|
|
})
|
|
|
|
- return chunks, sr
|
|
+ return chunks, sr, [cut / sr for cut in cuts]
|
|
|
|
|
|
def transcribe_buffered(
|
|
audio_path: str,
|
|
output_file: str = None,
|
|
chunk_duration_secs: float = 300, # 5 minutes default
|
|
+ overlap_secs: float = DEFAULT_OVERLAP_SECS,
|
|
+ pause_search_secs: float = DEFAULT_PAUSE_SEARCH_SECS,
|
|
):
|
|
"""
|
|
Transcribe long audio by splitting into chunks and merging results.
|
|
"""
|
|
+ import nemo.collections.asr as nemo_asr
|
|
+
|
|
# Determine model path
|
|
model_filename = "parakeet-tdt-0.6b-v3.nemo"
|
|
model_path = None
|
|
@@ -79,13 +224,15 @@ def transcribe_buffered(
|
|
asr_model.change_decoding_strategy(dec_cfg)
|
|
print("✓ CUDA graphs disabled successfully")
|
|
|
|
- print(f"Splitting audio into {chunk_duration_secs}s chunks...")
|
|
- chunks, sr = split_audio_file(audio_path, chunk_duration_secs)
|
|
+ print(f"Splitting audio into chunks of at most {chunk_duration_secs}s "
|
|
+ f"(overlap {overlap_secs}s, pause search {pause_search_secs}s)...")
|
|
+ chunks, sr, cut_times = split_audio_file(
|
|
+ audio_path, chunk_duration_secs, overlap_secs, pause_search_secs
|
|
+ )
|
|
print(f"Created {len(chunks)} chunks")
|
|
|
|
- all_words = []
|
|
- all_segments = []
|
|
- full_text = []
|
|
+ slice_results = []
|
|
+ chunk_texts = []
|
|
|
|
for i, chunk_info in enumerate(chunks):
|
|
print(f"Transcribing chunk {i+1}/{len(chunks)} (duration: {chunk_info['duration']:.1f}s)...")
|
|
@@ -104,25 +251,26 @@ def transcribe_buffered(
|
|
|
|
result_data = output[0]
|
|
chunk_text = result_data.text
|
|
- full_text.append(chunk_text)
|
|
+ chunk_texts.append(chunk_text)
|
|
+ chunk_words = []
|
|
+ chunk_segments = []
|
|
|
|
# Extract and adjust timestamps
|
|
if hasattr(result_data, 'timestamp') and result_data.timestamp:
|
|
- chunk_words = result_data.timestamp.get("word", [])
|
|
- chunk_segments = result_data.timestamp.get("segment", [])
|
|
-
|
|
# Adjust timestamps by chunk start time
|
|
- for word in chunk_words:
|
|
+ for word in result_data.timestamp.get("word", []):
|
|
word_copy = dict(word)
|
|
word_copy['start'] += chunk_info['start_time']
|
|
word_copy['end'] += chunk_info['start_time']
|
|
- all_words.append(word_copy)
|
|
+ chunk_words.append(word_copy)
|
|
|
|
- for segment in chunk_segments:
|
|
+ for segment in result_data.timestamp.get("segment", []):
|
|
seg_copy = dict(segment)
|
|
seg_copy['start'] += chunk_info['start_time']
|
|
seg_copy['end'] += chunk_info['start_time']
|
|
- all_segments.append(seg_copy)
|
|
+ chunk_segments.append(seg_copy)
|
|
+
|
|
+ slice_results.append((chunk_words, chunk_segments))
|
|
|
|
print(f"Chunk {i+1} complete: {len(chunk_text)} characters")
|
|
|
|
@@ -131,7 +279,15 @@ def transcribe_buffered(
|
|
if os.path.exists(chunk_path):
|
|
os.remove(chunk_path)
|
|
|
|
- final_text = " ".join(full_text)
|
|
+ chunk_spans = [(c['start_time'], c['start_time'] + c['duration']) for c in chunks]
|
|
+ all_words, all_segments = stitch_slices(slice_results, cut_times, chunk_spans)
|
|
+ if any(text.strip() and not words for (words, _), text in zip(slice_results, chunk_texts)):
|
|
+ # A chunk came back without word timestamps, so there is nothing to
|
|
+ # stitch it by; keep its text rather than lose it.
|
|
+ print("Warning: a chunk has text but no word timestamps; joining chunk texts")
|
|
+ final_text = " ".join(chunk_texts)
|
|
+ else:
|
|
+ final_text = " ".join(w["word"] for w in all_words)
|
|
print(f"Transcription complete: {len(final_text)} characters total")
|
|
|
|
output_data = {
|
|
@@ -144,6 +300,9 @@ def transcribe_buffered(
|
|
"buffered": True,
|
|
"chunk_duration_secs": chunk_duration_secs,
|
|
"num_chunks": len(chunks),
|
|
+ "overlap_secs": overlap_secs,
|
|
+ "pause_search_secs": pause_search_secs,
|
|
+ "cut_times": cut_times,
|
|
}
|
|
|
|
if output_file:
|
|
@@ -162,7 +321,17 @@ def main():
|
|
parser.add_argument("--output", "-o", help="Output file path", required=True)
|
|
parser.add_argument(
|
|
"--chunk-len", type=float, default=300,
|
|
- help="Chunk duration in seconds (default: 300 = 5 minutes)"
|
|
+ help="Maximum chunk duration in seconds, overlap included (default: 300 = 5 minutes)"
|
|
+ )
|
|
+ parser.add_argument(
|
|
+ "--overlap", type=float, default=DEFAULT_OVERLAP_SECS,
|
|
+ help=f"Seconds shared by adjacent chunks, capped at a quarter of --chunk-len "
|
|
+ f"(default: {DEFAULT_OVERLAP_SECS}; 0 disables)"
|
|
+ )
|
|
+ parser.add_argument(
|
|
+ "--pause-search", type=float, default=DEFAULT_PAUSE_SEARCH_SECS,
|
|
+ help=f"Seconds before each chunk limit searched for the quietest point to cut at, "
|
|
+ f"e.g. 25 (default: {DEFAULT_PAUSE_SEARCH_SECS}, cut at the limit)"
|
|
)
|
|
|
|
args = parser.parse_args()
|
|
@@ -175,6 +344,8 @@ def main():
|
|
audio_path=args.audio_file,
|
|
output_file=args.output,
|
|
chunk_duration_secs=args.chunk_len,
|
|
+ overlap_secs=args.overlap,
|
|
+ pause_search_secs=args.pause_search,
|
|
)
|
|
|
|
|
|
diff --git a/internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py b/internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py
|
|
new file mode 100644
|
|
index 0000000..6a35947
|
|
--- /dev/null
|
|
+++ b/internal/transcription/adapters/py/nvidia/tests/test_parakeet_slicing.py
|
|
@@ -0,0 +1,329 @@
|
|
+"""Unit tests for the slicing and stitching helpers in parakeet_transcribe_buffered.py.
|
|
+
|
|
+These are pure functions: they need numpy, librosa and soundfile (imported by the
|
|
+script) but no GPU, no NeMo and no model.
|
|
+"""
|
|
+import sys
|
|
+from pathlib import Path
|
|
+
|
|
+import numpy as np
|
|
+import pytest
|
|
+
|
|
+sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
+from parakeet_transcribe_buffered import plan_slices, stitch_slices # noqa: E402
|
|
+
|
|
+SR = 16000
|
|
+
|
|
+
|
|
+def speech(seconds, seed=0):
|
|
+ """Stand-in for continuous speech: broadband noise at a speech-like level."""
|
|
+ rng = np.random.default_rng(seed)
|
|
+ return (0.1 * rng.standard_normal(int(seconds * SR))).astype(np.float32)
|
|
+
|
|
+
|
|
+def with_pauses(audio, pauses, level=0.001, seed=1):
|
|
+ """Replace each (start_s, end_s) span with low-level room noise (or zeros)."""
|
|
+ rng = np.random.default_rng(seed)
|
|
+ out = audio.copy()
|
|
+ for start, end in pauses:
|
|
+ a, b = int(start * SR), int(end * SR)
|
|
+ out[a:b] = level * rng.standard_normal(b - a)
|
|
+ return out
|
|
+
|
|
+
|
|
+def assert_valid_plan(spans, cuts, num_samples, max_secs):
|
|
+ assert spans[0][0] == 0 and spans[-1][1] == num_samples
|
|
+ assert len(spans) == len(cuts) + 1
|
|
+ for start, end in spans:
|
|
+ assert 0 < end - start <= max_secs * SR
|
|
+ for (a0, a1), (b0, b1), cut in zip(spans, spans[1:], cuts):
|
|
+ assert b0 <= cut <= a1, "each cut must lie inside both neighbouring slices"
|
|
+ assert a0 < cut < b1
|
|
+
|
|
+
|
|
+# -- plan_slices: cut placement -------------------------------------------------
|
|
+
|
|
+
|
|
+def test_audio_shorter_than_one_slice_is_not_cut():
|
|
+ audio = speech(60)
|
|
+ spans, cuts = plan_slices(audio, SR, 120, overlap_secs=4, search_secs=25)
|
|
+ assert cuts == []
|
|
+ assert spans == [(0, len(audio))]
|
|
+
|
|
+
|
|
+def test_audio_exactly_one_slice_long_is_not_cut():
|
|
+ audio = speech(120)
|
|
+ spans, cuts = plan_slices(audio, SR, 120)
|
|
+ assert cuts == []
|
|
+ assert spans == [(0, len(audio))]
|
|
+
|
|
+
|
|
+def test_without_search_or_overlap_the_legacy_fixed_grid_is_reproduced():
|
|
+ audio = speech(300)
|
|
+ spans, cuts = plan_slices(audio, SR, 120, overlap_secs=0, search_secs=0)
|
|
+ assert cuts == [120 * SR, 240 * SR]
|
|
+ assert spans == [(0, 120 * SR), (120 * SR, 240 * SR), (240 * SR, 300 * SR)]
|
|
+
|
|
+
|
|
+def test_cuts_land_in_the_pauses_before_the_limit():
|
|
+ pauses = [(100.0, 100.5), (215.0, 215.5), (330.0, 330.5)]
|
|
+ audio = with_pauses(speech(400), pauses)
|
|
+ spans, cuts = plan_slices(audio, SR, 120, overlap_secs=0, search_secs=25)
|
|
+ assert len(cuts) == 3
|
|
+ for cut, (start, end) in zip(cuts, pauses):
|
|
+ assert start <= cut / SR <= end
|
|
+ assert_valid_plan(spans, cuts, len(audio), 120)
|
|
+
|
|
+
|
|
+def test_the_quietest_pause_wins():
|
|
+ # Two pauses inside the same search window; the later one is louder.
|
|
+ audio = with_pauses(speech(200), [(100.0, 100.6)], level=0.0)
|
|
+ audio = with_pauses(audio, [(115.0, 115.6)], level=0.01)
|
|
+ _, cuts = plan_slices(audio, SR, 120, search_secs=25)
|
|
+ assert 100.0 <= cuts[0] / SR <= 100.6
|
|
+
|
|
+
|
|
+def test_audio_with_no_pause_is_still_cut_within_the_limit():
|
|
+ audio = speech(400)
|
|
+ spans, cuts = plan_slices(audio, SR, 120, overlap_secs=0, search_secs=25)
|
|
+ assert_valid_plan(spans, cuts, len(audio), 120)
|
|
+ edges = [0] + cuts
|
|
+ for prev, cut in zip(edges, cuts):
|
|
+ assert 95 * SR <= cut - prev <= 120 * SR, "cut must fall inside its search window"
|
|
+
|
|
+
|
|
+def test_a_pause_at_the_very_start_is_never_a_cut():
|
|
+ audio = with_pauses(speech(200), [(0.0, 5.0)], level=0.0)
|
|
+ spans, cuts = plan_slices(audio, SR, 120, search_secs=25)
|
|
+ assert len(cuts) == 1 and 95 <= cuts[0] / SR <= 120
|
|
+ assert_valid_plan(spans, cuts, len(audio), 120)
|
|
+
|
|
+
|
|
+def test_a_pause_at_the_very_end_leaves_no_empty_slice():
|
|
+ # First cut ~100.35 s, so the second search window is ~[195, 220] s and
|
|
+ # holds the start of the trailing silence (217-222 s).
|
|
+ audio = with_pauses(speech(222), [(100.0, 100.5), (217.0, 222.0)], level=0.0)
|
|
+ spans, cuts = plan_slices(audio, SR, 120, search_secs=25)
|
|
+ assert len(cuts) == 2
|
|
+ assert 217.0 <= cuts[1] / SR < 222.0
|
|
+ assert_valid_plan(spans, cuts, len(audio), 120)
|
|
+
|
|
+
|
|
+# -- plan_slices: overlap -------------------------------------------------------
|
|
+
|
|
+
|
|
+def test_overlap_is_included_in_the_slice_limit_and_centred_on_each_cut():
|
|
+ pauses = [(100.0, 100.5), (215.0, 215.5), (330.0, 330.5)]
|
|
+ audio = with_pauses(speech(400), pauses)
|
|
+ spans, cuts = plan_slices(audio, SR, 120, overlap_secs=4, search_secs=25)
|
|
+ assert_valid_plan(spans, cuts, len(audio), 120)
|
|
+ for (_, a1), (b0, _), cut in zip(spans, spans[1:], cuts):
|
|
+ assert a1 - b0 == 4 * SR
|
|
+ assert cut - b0 == a1 - cut
|
|
+ for cut, (start, end) in zip(cuts, pauses):
|
|
+ assert start <= cut / SR <= end
|
|
+
|
|
+
|
|
+def test_overlap_without_pause_search_uses_a_fixed_grid():
|
|
+ audio = speech(300)
|
|
+ spans, cuts = plan_slices(audio, SR, 120, overlap_secs=4, search_secs=0)
|
|
+ assert cuts == [116 * SR, 232 * SR]
|
|
+ assert spans == [(0, 118 * SR), (114 * SR, 234 * SR), (230 * SR, 300 * SR)]
|
|
+
|
|
+
|
|
+def test_short_slice_lengths_clamp_overlap_and_search():
|
|
+ # Upstream's own buffered test runs a 19 s clip with --chunk-len 10.
|
|
+ audio = speech(19)
|
|
+ spans, cuts = plan_slices(audio, SR, 10, overlap_secs=4, search_secs=25)
|
|
+ assert len(spans) >= 2
|
|
+ assert_valid_plan(spans, cuts, len(audio), 10)
|
|
+
|
|
+
|
|
+# -- stitch_slices ------------------------------------------------------------------
|
|
+
|
|
+FRAME = 0.08
|
|
+OVERLAPPING = [(0.0, 12.0), (8.0, 20.0)] # two chunks sharing 8-12 s, cut at 10 s
|
|
+
|
|
+
|
|
+def word(text, start, end, slice_start):
|
|
+ return {
|
|
+ "word": text,
|
|
+ "start_offset": round((start - slice_start) / FRAME),
|
|
+ "end_offset": round((end - slice_start) / FRAME),
|
|
+ "start": start,
|
|
+ "end": end,
|
|
+ }
|
|
+
|
|
+
|
|
+def segment(words):
|
|
+ return {
|
|
+ "segment": " ".join(w["word"] for w in words),
|
|
+ "start_offset": words[0]["start_offset"],
|
|
+ "end_offset": words[-1]["end_offset"],
|
|
+ "start": words[0]["start"],
|
|
+ "end": words[-1]["end"],
|
|
+ }
|
|
+
|
|
+
|
|
+def texts(items, key="word"):
|
|
+ return [item[key] for item in items]
|
|
+
|
|
+
|
|
+def test_no_overlap_stitching_is_plain_concatenation():
|
|
+ left = [word("one", 1.0, 1.4, 0), word("two", 5.0, 5.3, 0)]
|
|
+ right = [word("three", 10.5, 10.9, 10), word("four", 14.0, 14.4, 10)]
|
|
+ words, segments = stitch_slices(
|
|
+ [(left, [segment(left)]), (right, [segment(right)])], [10.0]
|
|
+ )
|
|
+ assert words == left + right
|
|
+ assert segments == [segment(left), segment(right)]
|
|
+
|
|
+
|
|
+def test_overlapping_slices_keep_every_word_exactly_once():
|
|
+ # Slices [0, 12] and [8, 20], cut at 10. Both transcribe the overlap and
|
|
+ # their timestamps for the same word differ by a few ms.
|
|
+ left = [
|
|
+ word("a", 1.0, 1.3, 0),
|
|
+ word("b", 5.0, 5.4, 0),
|
|
+ word("c", 9.00, 9.40, 0),
|
|
+ word("d", 10.50, 10.90, 0),
|
|
+ word("e", 11.50, 11.80, 0),
|
|
+ ]
|
|
+ right = [
|
|
+ word("c", 9.02, 9.40, 8),
|
|
+ word("d", 10.48, 10.90, 8),
|
|
+ word("e", 11.52, 11.80, 8),
|
|
+ word("f", 15.00, 15.40, 8),
|
|
+ ]
|
|
+ words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING)
|
|
+ assert texts(words) == ["a", "b", "c", "d", "e", "f"]
|
|
+ assert words[2] is left[2] and words[3] is right[1]
|
|
+
|
|
+
|
|
+def test_a_word_straddling_the_cut_is_kept_once():
|
|
+ left = [word("over", 9.90, 10.30, 0), word("the", 10.40, 10.55, 0)]
|
|
+ right = [word("over", 9.92, 10.30, 8), word("the", 10.40, 10.55, 8)]
|
|
+ words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING)
|
|
+ assert texts(words) == ["over", "the"]
|
|
+ # Both chunks agree on it, so the right chunk takes over from it.
|
|
+ assert words[0] is right[0] and words[1] is right[1]
|
|
+
|
|
+
|
|
+def test_a_segment_straddling_the_cut_is_trimmed_to_the_words_each_slice_owns():
|
|
+ left_words = [
|
|
+ word("Hello", 8.5, 8.9, 0),
|
|
+ word("there", 9.2, 9.6, 0),
|
|
+ word("friend.", 10.4, 10.9, 0),
|
|
+ ]
|
|
+ right_words = [
|
|
+ word("there", 9.21, 9.6, 8),
|
|
+ word("friend.", 10.41, 10.9, 8),
|
|
+ word("Bye.", 13.0, 13.4, 8),
|
|
+ ]
|
|
+ words, segments = stitch_slices(
|
|
+ [
|
|
+ (left_words, [segment(left_words)]),
|
|
+ (right_words, [segment(right_words[:2]), segment(right_words[2:])]),
|
|
+ ],
|
|
+ [10.0],
|
|
+ OVERLAPPING,
|
|
+ )
|
|
+ assert texts(words) == ["Hello", "there", "friend.", "Bye."]
|
|
+ assert texts(segments, "segment") == ["Hello there", "friend.", "Bye."]
|
|
+ assert segments[0]["end"] == 9.6 and segments[1]["start"] == 10.41
|
|
+ # A segment that needed no trimming is passed through untouched.
|
|
+ assert segments[2] == segment(right_words[2:])
|
|
+ # Every word appears in exactly one segment, in order.
|
|
+ assert " ".join(texts(segments, "segment")) == " ".join(texts(words))
|
|
+
|
|
+
|
|
+def test_a_segment_wholly_inside_the_other_slices_share_is_dropped():
|
|
+ left_words = [word("a", 2.0, 2.3, 0), word("b.", 10.6, 11.0, 0)]
|
|
+ right_words = [word("b.", 10.61, 11.0, 8), word("c", 12.0, 12.3, 8)]
|
|
+ _, segments = stitch_slices(
|
|
+ [
|
|
+ (left_words, [segment(left_words[:1]), segment(left_words[1:])]),
|
|
+ (right_words, [segment(right_words[:1]), segment(right_words[1:])]),
|
|
+ ],
|
|
+ [10.0],
|
|
+ OVERLAPPING,
|
|
+ )
|
|
+ assert texts(segments, "segment") == ["a", "b.", "c"]
|
|
+ assert segments[1]["start"] == 10.61
|
|
+
|
|
+
|
|
+def test_a_non_positive_chunk_length_is_rejected_rather_than_looping():
|
|
+ with pytest.raises(ValueError):
|
|
+ plan_slices(speech(5), SR, 0)
|
|
+
|
|
+
|
|
+def test_a_word_the_two_chunks_timestamp_either_side_of_the_cut_is_kept_once():
|
|
+ # After a pause TDT may place a word's start anywhere in the pause, so the
|
|
+ # two chunks can disagree about which side of the cut it starts on.
|
|
+ left = [word("so", 8.2, 8.5, 0), word("then", 9.98, 10.3, 0), word("we", 10.4, 10.6, 0)]
|
|
+ right = [word("so", 8.2, 8.5, 8), word("then", 10.03, 10.3, 8), word("we", 10.4, 10.6, 8)]
|
|
+ words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING)
|
|
+ assert texts(words) == ["so", "then", "we"]
|
|
+ # ...and the mirror image, where splitting both at the cut would drop it.
|
|
+ left = [word("so", 8.2, 8.5, 0), word("then", 10.03, 10.3, 0), word("we", 10.4, 10.6, 0)]
|
|
+ right = [word("so", 8.2, 8.5, 8), word("then", 9.98, 10.3, 8), word("we", 10.4, 10.6, 8)]
|
|
+ words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING)
|
|
+ assert texts(words) == ["so", "then", "we"]
|
|
+
|
|
+
|
|
+def test_handover_happens_at_the_agreed_word_nearest_the_cut():
|
|
+ # The chunks differ in casing/punctuation and the left one drops "really"
|
|
+ # near its end; the right chunk's version of the overlap after the cut wins.
|
|
+ left = [word("It", 8.5, 8.7, 0), word("was", 9.6, 9.9, 0), word("good,", 11.0, 11.4, 0)]
|
|
+ right = [word("it", 8.5, 8.7, 8), word("was", 9.62, 9.9, 8), word("really", 10.3, 10.7, 8),
|
|
+ word("good.", 11.0, 11.4, 8)]
|
|
+ words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING)
|
|
+ assert texts(words) == ["It", "was", "really", "good."]
|
|
+ assert words[0] is left[0] and words[1] is right[1]
|
|
+
|
|
+
|
|
+def test_without_an_agreed_word_the_split_falls_back_to_the_cut():
|
|
+ left = [word("alpha", 9.0, 9.4, 0), word("beta", 10.5, 10.9, 0)]
|
|
+ right = [word("gamma", 9.1, 9.4, 8), word("delta", 10.6, 10.9, 8)]
|
|
+ words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING)
|
|
+ assert texts(words) == ["alpha", "delta"]
|
|
+
|
|
+
|
|
+def test_another_occurrence_of_the_word_elsewhere_in_the_overlap_is_not_an_anchor():
|
|
+ # The chunks disagree everywhere except on "the", but the left chunk's
|
|
+ # "the" (9.0 s) and the right chunk's (11.0 s) are different words. As an
|
|
+ # anchor they would average to the cut and discard the left one.
|
|
+ left = [word("the", 9.0, 9.2, 0), word("dog", 10.5, 10.8, 0)]
|
|
+ right = [word("cat", 9.3, 9.6, 8), word("the", 11.0, 11.2, 8)]
|
|
+ words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING)
|
|
+ assert [w["start"] for w in words] == [9.0, 11.0]
|
|
+
|
|
+
|
|
+def test_the_handover_never_puts_words_out_of_time_order():
|
|
+ # "y" agrees (0.45 s apart), but the right chunk's "y" (9.65) after the
|
|
+ # left chunk's "x" (9.70) would run time backwards, so the left chunk's
|
|
+ # copy is kept instead. Splitting at the cut would lose "y" altogether.
|
|
+ left = [word("x", 9.70, 9.90, 0), word("y", 10.10, 10.30, 0)]
|
|
+ right = [word("z", 9.40, 9.60, 8), word("y", 9.65, 9.90, 8), word("w", 10.6, 10.8, 8)]
|
|
+ words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING)
|
|
+ assert texts(words) == ["x", "y", "w"] and words[1] is left[1]
|
|
+ starts = [w["start"] for w in words]
|
|
+ assert starts == sorted(starts)
|
|
+
|
|
+
|
|
+def test_a_co_timed_anchor_is_found_even_when_a_longer_match_lies_elsewhere():
|
|
+ # "x y" recurs later in the right chunk, a longer text match than "z", but
|
|
+ # at a different time. Only "z" is the same word in both chunks, and it
|
|
+ # straddles the cut, so splitting both at the cut would keep it twice.
|
|
+ left = [word("x", 8.2, 8.3, 0), word("y", 8.4, 8.5, 0), word("z", 9.98, 10.2, 0)]
|
|
+ right = [word("z", 10.03, 10.2, 8), word("x", 11.0, 11.1, 8), word("y", 11.2, 11.3, 8)]
|
|
+ words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING)
|
|
+ assert texts(words) == ["x", "y", "z", "x", "y"]
|
|
+
|
|
+
|
|
+def test_punctuation_alone_is_never_an_anchor():
|
|
+ # As an anchor the dash would hand the whole overlap to the right chunk.
|
|
+ left = [word("-", 9.50, 9.55, 0), word("yes", 10.4, 10.6, 0)]
|
|
+ right = [word("-", 9.52, 9.55, 8), word("no", 10.4, 10.6, 8)]
|
|
+ words, _ = stitch_slices([(left, []), (right, [])], [10.0], OVERLAPPING)
|
|
+ assert words[0] is left[0] and texts(words) == ["-", "no"]
|
|
--
|
|
2.39.5
|
|
|