#!/usr/bin/env python3 """Validate a parakeet_transcribe_buffered.py result against the Go seam. Scriberr's parakeet_adapter.go (parseResult) unmarshals this JSON into a struct with typed fields; a float where Go expects an int, or a missing key, fails the job. This checks the shape Go reads plus the stitching invariants the slicer patch promises. Stdlib only, so it runs under any python3. Prints counts, never transcript text. usage: scriberr-seam-check.py RESULT.json [--min-chunks N] """ import argparse import json import sys NUMBER = (int, float) def fail(msg): print(f"SEAM FAIL: {msg}") sys.exit(1) def check_items(items, text_key, name): for i, item in enumerate(items): if not isinstance(item, dict): fail(f"{name}[{i}] is not an object") if not isinstance(item.get(text_key), str): fail(f"{name}[{i}].{text_key} is not a string") for key in ("start_offset", "end_offset"): if type(item.get(key)) is not int: fail(f"{name}[{i}].{key} is not an integer (Go field is int)") for key in ("start", "end"): if not isinstance(item.get(key), NUMBER) or isinstance(item.get(key), bool): fail(f"{name}[{i}].{key} is not a number") if item["start"] > item["end"]: fail(f"{name}[{i}] starts after it ends") def main(): parser = argparse.ArgumentParser(description="Validate a buffered Parakeet result for Go.") parser.add_argument("result", help="result JSON written by parakeet_transcribe_buffered.py") parser.add_argument("--min-chunks", type=int, default=1, help="fail unless the run used at least this many chunks") args = parser.parse_args() min_chunks = args.min_chunks try: data = json.load(open(args.result, encoding="utf-8")) except (OSError, ValueError) as e: fail(f"cannot read {args.result}: {e}") if not isinstance(data, dict): fail("the result is not a JSON object") required = {"transcription": str, "language": str, "word_timestamps": list, "segment_timestamps": list, "audio_file": str, "model": str} for key, kind in required.items(): if not isinstance(data.get(key), kind): fail(f"'{key}' missing or not {kind.__name__}") if data.get("buffered") is not True: fail("'buffered' is not true") if not isinstance(data.get("chunk_duration_secs"), NUMBER): fail("'chunk_duration_secs' is not a number") if type(data.get("num_chunks")) is not int or data["num_chunks"] < min_chunks: fail(f"'num_chunks' is not an integer >= {min_chunks}") words, segments = data["word_timestamps"], data["segment_timestamps"] if not words or not data["transcription"].strip(): fail("empty transcript") check_items(words, "word", "word_timestamps") check_items(segments, "segment", "segment_timestamps") starts = [w["start"] for w in words] if starts != sorted(starts): fail("word start times go backwards (a stitch repeated or reordered words)") joined = " ".join(w["word"] for w in words) if data["transcription"] != joined: fail("'transcription' is not the stitched words joined by spaces") if " ".join(s["segment"] for s in segments) != joined: fail("segments do not cover the stitched words exactly once, in order") print(f"SEAM OK: {len(words)} words, {len(segments)} segments, " f"{data['num_chunks']} chunks, cuts at {len(data.get('cut_times', []))} points") if __name__ == "__main__": main()