#!/usr/bin/env python3 """NVFP4 quantize the MTP-grafted Heretic2 model (llm-compressor / compressed-tensors). Runs AFTER graft_mtp.py. Fleet-proven compressed-tensors NVFP4 path (pantheon-27b- mtp-nvfp4 serves this on our Blackwell) with the ignore-list from robbatt's on-fleet deckard-nvfp4 recipe PLUS the MTP head: keep GDN/linear-attn, vision, lm-head, all norms, AND mtp.* in BF16; NVFP4 only the dense Linear layers. (R36 fast-seat spike.) API validated against llmcompressor 0.12.0: oneshot(model, dataset, recipe, num_calibration_samples, max_seq_length) — dataset is a pre-tokenized datasets.Dataset. Two calib modes: --calib-mode text : AEON-baseline control (neuralmagic/calibration LLM split) --calib-mode chat : production 512-row mix (JSONL {messages, tools}); each row rendered via apply_chat_template(enable_thinking=True) so the forward-pass sees the qwen3_coder tool-call XML the seat emits. Run in a vLLM container on the freed GPU0: docker run --gpus '"device=0"' --ipc host -v /tank/aimodels:/tank/aimodels \ --entrypoint bash vllm/vllm-openai:v0.24.0 -c \ "pip install -q llmcompressor tiktoken sentencepiece && python3 quant_nvfp4.py ..." """ import argparse import json import sys # Ignore list = robbatt deckard-nvfp4 recipe + MTP head (brokkr: keep mtp.* BF16). IGNORE = [ "lm_head", "re:.*embed_tokens$", "re:visual.*", "re:model.visual.*", "re:.*linear_attn.*", "re:.*norm.*", "re:.*q_norm.*", "re:.*k_norm.*", "re:.*mtp.*", # MTP head stays BF16 for qwen3_5_mtp spec-decode (match anywhere: # module path is model.mtp.*, so an anchored re:mtp.* misses it) ] def _tok_rows_to_dataset(tok_rows): from datasets import Dataset return Dataset.from_list(tok_rows) def load_calib_text(name, tokenizer, n, seqlen): from datasets import load_dataset ds = load_dataset(name, split="train").shuffle(seed=42).select(range(min(n, 100000))) col = "text" if "text" in ds.column_names else ds.column_names[0] rows = [tokenizer(x[col], truncation=True, max_length=seqlen) for x in ds.select(range(n))] return _tok_rows_to_dataset(rows) def load_calib_chat(path, tokenizer, seqlen, n): rows = [json.loads(l) for l in open(path) if l.strip()][:n] out = [] for r in rows: # Tool_call arguments may be OpenAI wire-form JSON strings; the Qwen3.6 template # does .items() on them → needs a dict. Parse string→dict (render_verify finding; # belt-and-suspenders even though brokkr canonicalized the rows to dicts). for m in r["messages"]: for tc in (m.get("tool_calls") or []): a = tc.get("function", {}).get("arguments") if isinstance(a, str): tc["function"]["arguments"] = json.loads(a) text = tokenizer.apply_chat_template( r["messages"], tools=r.get("tools"), tokenize=False, add_generation_prompt=False, enable_thinking=True, ) out.append(tokenizer(text, truncation=True, max_length=seqlen)) return _tok_rows_to_dataset(out) def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("--model", required=True) ap.add_argument("--calib-mode", choices=["text", "chat"], required=True) ap.add_argument("--calib", required=True) ap.add_argument("--out", required=True) ap.add_argument("--num-samples", type=int, default=512) ap.add_argument("--seqlen", type=int, default=8192) args = ap.parse_args() from transformers import AutoModelForImageTextToText, AutoTokenizer from llmcompressor import oneshot from llmcompressor.modifiers.quantization import QuantizationModifier print(f"loading grafted model: {args.model}", flush=True) # Load as the FULL multimodal Qwen3_5ForConditionalGeneration (NOT AutoModelForCausalLM). # AutoModelForCausalLM resolves qwen3_5 -> Qwen3_5ForCausalLM (text-only), whose weight # keys are flat `model.layers.*` with no vision tower. But vLLM 0.24 only registers # Qwen3_5ForConditionalGeneration, and its hf_to_vllm_mapper expects the checkpoint keyed # `model.language_model.layers.*` (+ `model.visual.*`) — a bare `model.layers.` prefix has # NO mapping rule, so every transformer-layer weight fails to load -> uninitialized weights # -> degenerate `!!!!` output. AutoModelForImageTextToText resolves qwen3_5 -> # Qwen3_5ForConditionalGeneration, so keys are born `model.language_model.*` / `model.visual.*` # matching the working pantheon-27b-mtp-nvfp4 reference. The vision tower loads in BF16 and is # ignored by the quant (re:.*visual.*); calibration is text-only (no pixel_values needed). # (R36 fast-seat namespace fix, 2026-07-14 — the config-merge in the prior recipe was a doomed # patch over a checkpoint quantized in the wrong namespace.) model = AutoModelForImageTextToText.from_pretrained( args.model, torch_dtype="auto", device_map="auto", trust_remote_code=True, ) tok = AutoTokenizer.from_pretrained(args.model, trust_remote_code=True) print(f"building calibration ({args.calib_mode}, up to {args.num_samples} @ seq {args.seqlen})", flush=True) if args.calib_mode == "text": calib = load_calib_text(args.calib, tok, args.num_samples, args.seqlen) else: calib = load_calib_chat(args.calib, tok, args.seqlen, args.num_samples) print(f" {len(calib)} calibration rows", flush=True) recipe = QuantizationModifier(targets="Linear", scheme="NVFP4", ignore=IGNORE) print("running NVFP4 oneshot (Linear-only; GDN/vision/lm-head/norms/MTP kept BF16)", flush=True) oneshot( model=model, dataset=calib, recipe=recipe, num_calibration_samples=len(calib), max_seq_length=args.seqlen, ) print(f"saving -> {args.out}", flush=True) model.save_pretrained(args.out, save_compressed=True) tok.save_pretrained(args.out) print("DONE. serve: vllm --quantization compressed-tensors " "--speculative-config '{\"method\":\"qwen3_5_mtp\",\"num_speculative_tokens\":3}' " "--reasoning-parser qwen3 --tool-call-parser qwen3_coder --enable-auto-tool-choice", flush=True) return 0 if __name__ == "__main__": sys.exit(main())