feat(heretic2-nvfp4): MTP-graft + NVFP4 quant scripts + pipeline README (fire-ready)

graft_mtp.py: grafts the 15 base-Qwen3.6 MTP tensors into Heretic2 BF16 (CPU-only).
quant_nvfp4.py: llm-compressor NVFP4 (Linear only; GDN/vision/lm-head/norms/MTP
kept BF16 per robbatt's deckard recipe + brokkr's spec); text (AEON-baseline) or
chat (production, apply_chat_template renders qwen3_coder XML) calib modes.
README: fire sequence + gates (GPU window, production calib) + artifacts.

Spike gated only on: (1) off-peak Blackwell GPU window, (2) brokkr's production calib.
This commit is contained in:
vh
2026-07-14 08:57:25 -07:00
parent bbbfe5502e
commit 920f9a3709
3 changed files with 305 additions and 0 deletions
@@ -0,0 +1,118 @@
#!/usr/bin/env python3
"""NVFP4 quantize the MTP-grafted Heretic2 model (llm-compressor / compressed-tensors).
Runs AFTER graft_mtp.py. Uses the fleet-proven compressed-tensors NVFP4 path
(the pantheon-27b-mtp-nvfp4 serving pattern already works on our Blackwell) with
the ignore-list from robbatt's on-fleet deckard-nvfp4 recipe PLUS the MTP head:
keep GDN/linear-attn, vision tower, lm-head, all norms, AND mtp.* in BF16;
NVFP4 only the dense Linear layers. (R36 fast-seat spike, brokkr calib spec.)
⚠️ NEEDS: (a) an env with llmcompressor + a CUDA torch (run inside a vLLM
container: `pip install llmcompressor` on vllm/vllm-openai:v0.24.0), (b) a freed
Blackwell GPU (~55 GB — both ana-ml2 GPUs are normally full; needs an off-peak
window). API validated against llm-compressor at run time — treat the exact
symbol names as first-draft until a dry import confirms them.
Two calib modes (brokkr's two artifacts):
--calib-mode text : AEON-baseline control (neuralmagic/calibration LLM split)
--calib-mode chat : production 512-row mix (JSONL rows {messages, tools});
each row rendered via apply_chat_template(enable_thinking=True)
so the forward-pass sees the qwen3_coder tool-call XML =
the seat's native activations (the #355-preservation point).
Usage:
python3 quant_nvfp4.py --model /tank/aimodels/heretic2-mtp-bf16 \
--calib-mode text --calib neuralmagic/calibration --num-samples 160 \
--out /tank/aimodels/heretic2-mtp-nvfp4-baseline
python3 quant_nvfp4.py --model /tank/aimodels/heretic2-mtp-bf16 \
--calib-mode chat --calib /path/to/production_calib_512.jsonl \
--out /tank/aimodels/heretic2-mtp-nvfp4-prod
"""
import argparse
import json
import sys
# Ignore list = robbatt deckard-nvfp4 recipe + MTP head (brokkr: keep mtp.* BF16).
# Keeps in BF16: lm-head, embeddings, vision tower, GDN/linear-attn, all norms, MTP.
IGNORE = [
"lm_head",
"re:.*embed_tokens$",
"re:visual.*",
"re:model.visual.*",
"re:.*linear_attn.*",
"re:.*norm.*",
"re:.*q_norm.*",
"re:.*k_norm.*",
"re:mtp.*", # MTP head stays BF16 for the qwen3_5_mtp spec-decode head
]
def load_calib_text(name, tokenizer, n, seqlen):
from datasets import load_dataset
ds = load_dataset(name, split="train").shuffle(seed=42).select(range(n))
col = "text" if "text" in ds.column_names else ds.column_names[0]
return [tokenizer(x[col], truncation=True, max_length=seqlen) for x in ds]
def load_calib_chat(path, tokenizer, seqlen):
"""Render {messages, tools} JSONL rows through the chat template with thinking on.
The assistant tool_calls (OpenAI form) serialize to qwen3_coder XML here, so the
calibration forward-pass sees the EXACT tool-call token distribution the seat emits."""
rows = [json.loads(l) for l in open(path) if l.strip()]
out = []
for r in rows:
text = tokenizer.apply_chat_template(
r["messages"], tools=r.get("tools"),
tokenize=False, add_generation_prompt=False, enable_thinking=True,
)
out.append(tokenizer(text, truncation=True, max_length=seqlen))
return out
def main() -> int:
ap = argparse.ArgumentParser()
ap.add_argument("--model", required=True, help="grafted BF16 model dir (graft_mtp.py output)")
ap.add_argument("--calib-mode", choices=["text", "chat"], required=True)
ap.add_argument("--calib", required=True, help="HF dataset name (text) or JSONL path (chat)")
ap.add_argument("--out", required=True)
ap.add_argument("--num-samples", type=int, default=512)
ap.add_argument("--seqlen", type=int, default=8192)
args = ap.parse_args()
from transformers import AutoModelForCausalLM, AutoTokenizer
from llmcompressor import oneshot
from llmcompressor.modifiers.quantization import QuantizationModifier
print(f"loading grafted model: {args.model}")
model = AutoModelForCausalLM.from_pretrained(
args.model, torch_dtype="auto", device_map="auto", trust_remote_code=True,
)
tok = AutoTokenizer.from_pretrained(args.model, trust_remote_code=True)
print(f"building calibration ({args.calib_mode}, {args.num_samples} samples, seq {args.seqlen})")
if args.calib_mode == "text":
calib = load_calib_text(args.calib, tok, args.num_samples, args.seqlen)
else:
calib = load_calib_chat(args.calib, tok, args.seqlen)
print(f" {len(calib)} calibration rows")
recipe = QuantizationModifier(targets="Linear", scheme="NVFP4", ignore=IGNORE)
print("running NVFP4 oneshot (Linear-only; GDN/vision/lm-head/norms/MTP kept BF16)")
oneshot(
model=model, dataset=calib, recipe=recipe,
max_seq_length=args.seqlen, num_calibration_samples=len(calib),
)
print(f"saving -> {args.out}")
model.save_pretrained(args.out, save_compressed=True)
tok.save_pretrained(args.out)
print("DONE. serve: vllm --quantization compressed-tensors "
"--speculative-config '{\"method\":\"qwen3_5_mtp\",\"num_speculative_tokens\":3}' "
"--reasoning-parser qwen3 --tool-call-parser qwen3_coder --enable-auto-tool-choice")
print("NEXT: ping brokkr -> P00 rig (soong 9-tool k5); acceptance = hold ~0.967/perfect attach_tool")
return 0
if __name__ == "__main__":
sys.exit(main())