"""Did the quant actually do what the recipe says, on the tensors it claims? `rc=0` and a plausible file size prove neither. The two things the operator asked for -- W4A16, vision towers intact -- are properties of the tensor table, so read the tensor table. Parses safetensors headers directly (u64 length + JSON), so no torch, no GPU, and no 20 GB load. Checks, per module family: * language-model Linears -> must be NVFP4-packed (uint8 blobs + *_scale companions) * vision / audio towers -> must still be BF16, i.e. PRESERVED not dropped * embeddings / lm_head / norms -> BF16 per the ignore list Run it against a known-good tree as well. A checker that has only ever seen the tree it was written for cannot tell "correct" from "blind". """ import argparse import json import struct from collections import defaultdict from pathlib import Path def tensors(root: Path): """Yield (name, dtype, shape) for every tensor in a sharded or single-file tree.""" idx = root / "model.safetensors.index.json" files = sorted({Path(v) for v in json.loads(idx.read_text())["weight_map"].values()}) \ if idx.exists() else [Path("model.safetensors")] for f in files: p = root / f with p.open("rb") as fh: n = struct.unpack(" str: if "vision_tower" in name or "embed_vision" in name: return "vision_tower" if "audio_tower" in name or "embed_audio" in name: return "audio_tower" if "multi_modal_projector" in name or "mm_projector" in name: return "projector" if "embed_tokens" in name: return "embeddings" if name.startswith("lm_head") or ".lm_head" in name: return "lm_head" if "norm" in name: return "norms" if "language_model" in name or ".layers." in name: return "language_model" return "other" ap = argparse.ArgumentParser() ap.add_argument("trees", nargs="+") a = ap.parse_args() for t in a.trees: root = Path(t) print(f"\n{'='*70}\n{root}") cfg = json.loads((root / "config.json").read_text()) q = cfg.get("quantization_config", {}) groups = q.get("config_groups", {}) for gname, g in groups.items(): w = g.get("weights", {}) i = g.get("input_activations") print(f" {gname}: weights num_bits={w.get('num_bits')} type={w.get('type')} " f"strategy={w.get('strategy')} | input_activations=" f"{'None (WEIGHT-ONLY)' if i is None else i}") print(f" format={q.get('format')} kv_cache_scheme={q.get('kv_cache_scheme')} " f"status={q.get('quantization_status')}") tc = cfg.get("text_config", {}) print(f" text_config: per_layer_config={'PRESENT' if 'per_layer_config' in tc else 'absent'}" f" head_dim={tc.get('head_dim')} global_head_dim={tc.get('global_head_dim')}" f" num_key_value_heads={tc.get('num_key_value_heads')}" f" num_global_key_value_heads={tc.get('num_global_key_value_heads')}") by = defaultdict(lambda: defaultdict(int)) packed = defaultdict(int) for name, dt, shape in tensors(root): f = family(name) by[f][dt] += 1 if name.endswith("weight_packed") or name.endswith("weight_scale"): packed[f] += 1 print(" tensor dtypes by family:") for f in sorted(by): dts = ", ".join(f"{d}x{c}" for d, c in sorted(by[f].items())) note = f" [{packed[f]} packed/scale tensors]" if packed[f] else "" print(f" {f:16} {dts}{note}")