Carry patches/0001 on our Scriberr build (upstream a353078): adjacent buffered chunks overlap by 4 s inside --chunk-len and hand over at a word both chunks transcribed alike, instead of cutting at fixed marks with no overlap. Pause-aware cutting is included as an opt-in (--pause-search); it measured neutral once the stitch was right. The Go<->Python CLI and JSON seam is unchanged. Bench (4 recordings, 118 min, 3 cut placements each, against a no-cut whole-file reference; metrics only, private audio stays on fv-ml1): cuts with an error within +-3 s fall from 52% (93/179) to 22% (41/184) against a 19% background; floor +-0.08. Positive control: upstream's cutter +0.33 over background. A-vs-A byte-identical in-process and across CLI processes. Peak GPU memory unchanged at 5,496 MiB (n=3). Also found: Parakeet skips runs of >=10 words mid-chunk with any slicer, upstream's included; not addressed here. scripts/scriberr-rebuild clones a pinned upstream sha into a new /opt/docker/src dir, git-apply-checks the patches, builds a distinct tag, and checks embed, unit tests, the JSON seam (scriberr-seam-check.py) and the memory budget on idle GPU 3. Deploy stays manual. The upstream PR is prepared under patches/upstream-pr/ and not opened.
87 lines
3.5 KiB
Python
Executable File
87 lines
3.5 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Validate a parakeet_transcribe_buffered.py result against the Go seam.
|
|
|
|
Scriberr's parakeet_adapter.go (parseResult) unmarshals this JSON into a struct
|
|
with typed fields; a float where Go expects an int, or a missing key, fails the
|
|
job. This checks the shape Go reads plus the stitching invariants the slicer
|
|
patch promises. Stdlib only, so it runs under any python3. Prints counts, never
|
|
transcript text.
|
|
|
|
usage: scriberr-seam-check.py RESULT.json [--min-chunks N]
|
|
"""
|
|
import argparse
|
|
import json
|
|
import sys
|
|
|
|
NUMBER = (int, float)
|
|
|
|
|
|
def fail(msg):
|
|
print(f"SEAM FAIL: {msg}")
|
|
sys.exit(1)
|
|
|
|
|
|
def check_items(items, text_key, name):
|
|
for i, item in enumerate(items):
|
|
if not isinstance(item, dict):
|
|
fail(f"{name}[{i}] is not an object")
|
|
if not isinstance(item.get(text_key), str):
|
|
fail(f"{name}[{i}].{text_key} is not a string")
|
|
for key in ("start_offset", "end_offset"):
|
|
if type(item.get(key)) is not int:
|
|
fail(f"{name}[{i}].{key} is not an integer (Go field is int)")
|
|
for key in ("start", "end"):
|
|
if not isinstance(item.get(key), NUMBER) or isinstance(item.get(key), bool):
|
|
fail(f"{name}[{i}].{key} is not a number")
|
|
if item["start"] > item["end"]:
|
|
fail(f"{name}[{i}] starts after it ends")
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description="Validate a buffered Parakeet result for Go.")
|
|
parser.add_argument("result", help="result JSON written by parakeet_transcribe_buffered.py")
|
|
parser.add_argument("--min-chunks", type=int, default=1,
|
|
help="fail unless the run used at least this many chunks")
|
|
args = parser.parse_args()
|
|
min_chunks = args.min_chunks
|
|
try:
|
|
data = json.load(open(args.result, encoding="utf-8"))
|
|
except (OSError, ValueError) as e:
|
|
fail(f"cannot read {args.result}: {e}")
|
|
if not isinstance(data, dict):
|
|
fail("the result is not a JSON object")
|
|
|
|
required = {"transcription": str, "language": str, "word_timestamps": list,
|
|
"segment_timestamps": list, "audio_file": str, "model": str}
|
|
for key, kind in required.items():
|
|
if not isinstance(data.get(key), kind):
|
|
fail(f"'{key}' missing or not {kind.__name__}")
|
|
if data.get("buffered") is not True:
|
|
fail("'buffered' is not true")
|
|
if not isinstance(data.get("chunk_duration_secs"), NUMBER):
|
|
fail("'chunk_duration_secs' is not a number")
|
|
if type(data.get("num_chunks")) is not int or data["num_chunks"] < min_chunks:
|
|
fail(f"'num_chunks' is not an integer >= {min_chunks}")
|
|
|
|
words, segments = data["word_timestamps"], data["segment_timestamps"]
|
|
if not words or not data["transcription"].strip():
|
|
fail("empty transcript")
|
|
check_items(words, "word", "word_timestamps")
|
|
check_items(segments, "segment", "segment_timestamps")
|
|
|
|
starts = [w["start"] for w in words]
|
|
if starts != sorted(starts):
|
|
fail("word start times go backwards (a stitch repeated or reordered words)")
|
|
joined = " ".join(w["word"] for w in words)
|
|
if data["transcription"] != joined:
|
|
fail("'transcription' is not the stitched words joined by spaces")
|
|
if " ".join(s["segment"] for s in segments) != joined:
|
|
fail("segments do not cover the stitched words exactly once, in order")
|
|
|
|
print(f"SEAM OK: {len(words)} words, {len(segments)} segments, "
|
|
f"{data['num_chunks']} chunks, cuts at {len(data.get('cut_times', []))} points")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|