"""Fetch pinned artifacts; verify each against its published digest. Prints a provenance line per file.""" import hashlib, json, os, sys, urllib.request TOKEN = open("/tank/aimodels/huggingface/token").read().strip() DL = "/tank/spikes/parakeet-ab/dl" def sha256(p): h = hashlib.sha256() with open(p, "rb") as f: for b in iter(lambda: f.read(1 << 24), b""): h.update(b) return h.hexdigest() def fetch(url, out, auth=False): if os.path.exists(out): return req = urllib.request.Request(url, headers={"Authorization": f"Bearer {TOKEN}"} if auth else {}) with urllib.request.urlopen(req, timeout=120) as r, open(out + ".part", "wb") as f: while True: b = r.read(1 << 22) if not b: break f.write(b) os.rename(out + ".part", out) prov = [] # GitHub release assets (digest from the GitHub API, recorded 2026-09-30) gh = {"sherpa-onnx-nemo-parakeet-unified-en-0.6b-int8-non-streaming.tar.bz2": "99f63605b3a85a54c250c0869670a687b7d6598a47bf2421515e1f839a76e150", "sherpa-onnx-nemo-parakeet-tdt-0.6b-v2-int8.tar.bz2": "157c157bc51155e03e37d2466522a3a737dd9c72bb25f36eb18912964161e1ad"} for name, want in gh.items(): out = f"{DL}/{name}" fetch(f"https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/{name}", out) got = sha256(out) prov.append(dict(src=f"github:k2-fsa/sherpa-onnx@asr-models/{name}", sha256=got, expected=want, ok=got == want)) # HF datasets, revision-pinned; expected sha = LFS oid from the authenticated API hf = [("openslr/librispeech_asr", "71cacbfb7e2354c4226d01e70d77d5fca3d04ba1", ["all/test.clean/0000.parquet", "all/test.other/0000.parquet"]), ("edinburghcstr/ami", "46f28f2503e2ec48f8867a84eef356c70476beab", [f"ihm/test-0000{i}-of-00004.parquet" for i in range(4)])] for repo, rev, files in hf: req = urllib.request.Request(f"https://huggingface.co/api/datasets/{repo}/revision/{rev}?blobs=true", headers={"Authorization": f"Bearer {TOKEN}"}) meta = json.load(urllib.request.urlopen(req, timeout=60)) oid = {s["rfilename"]: (s.get("lfs") or {}).get("sha256") for s in meta["siblings"]} for fn in files: out = f"{DL}/{repo.replace('/', '__')}__{fn.replace('/', '__')}" fetch(f"https://huggingface.co/datasets/{repo}/resolve/{rev}/{fn}", out, auth=True) got = sha256(out) prov.append(dict(src=f"hf-dataset:{repo}@{rev}/{fn}", sha256=got, expected=oid.get(fn), ok=got == oid.get(fn))) for p in prov: print(json.dumps(p)) json.dump(prov, open("/tank/spikes/parakeet-ab/out/provenance-fetch.json", "w"), indent=1)