feat(gen-seat): mixed NVFP4+FP8 requant — +18% decode at equal MTP acceptance
Re-quantizes the fleet `gen` seat from weight-only NVFP4A16 to a mixed-precision build: NVFP4 W4A4 for layers 0-55 MLPs, FP8 W8A8 for the attention projections / linear_attn / lm_head / layers 56-63 MLPs, FP8 KV cache. Replicates the scheme of unsloth/Qwen3.8-27B-NVFP4 on the abliterated weights. The queued task named this "W4A8" (NVFP4 weights + FP8 activations). That checkpoint cannot be served: vLLM 0.24's compressed-tensors dispatcher (compressed_tensors.py:704-713) accepts NVFP4 weights with either no input quantization (W4A16, which forces the Marlin kernel) or NVFP4 input quantization (W4A4) -- anything else, FP8 included, raises ValueError at load. CompressedTensorsW4A8Fp8 is INT4 weights gated on an exact-sm90 check, so it is closed on Blackwell twice over. The ~20% intuition was correct; the scheme name was not. Getting FP8 into the mix has to be done per-layer-group. Established the gain before spending GPU time: unsloth's build was already on-box, so serving it as a probe measured +19.1% over our seat at identical MTP acceptance -- a kernel-level result, no requant needed to learn it. Measured, cache-busted, bs=1: decode 80.12 -> 94.53 tok/s (+18.0%) MTP acceptance 47.8% -> 47.7% (unchanged) perplexity (n=6) 6.941 -> 7.059 (+1.7%) abliteration 4/4 -> 4/4 (preserved) weights on disk 27.7 -> 22.5 GB (-19%) Surface test green on the live seat: plain chat, vision, tool calling, thinking split, 36K-token needle retrieval, streaming. All 7 LiteLLM aliases verified routing. GEN_GPU_MEM_UTIL 0.45 -> 0.43: the new weights are 5.2 GB smaller, and at 0.45 the seat absorbed that slack as KV, leaving meromero-charrp 0.18 GiB short of its budget on the shared GPU0 -- it crash-looped. Handing the space back leaves gen 422K tokens of KV (1.6x its 262K context) and both seats co-resident at 89.8/97.9 GB. Also records two measured negatives so they are not re-chased: GEN_SPEC_TOKENS is already optimal at 3 (swept 2/3/4/5 -> 77.1/80.1/78.7/ 75.9 tok/s), and vLLM's prompt_logprobs are ~uniform while speculative decoding is on, so perplexity must be measured with spec off. Pipeline, acceptance harness and raw measurements land in services/gen-seat-mixed-quant/. Rollback is one .env line; the previous build is untouched at /tank/aimodels/qwen38-27b-uncensored-nvfp4.
This commit is contained in:
@@ -0,0 +1,103 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Mandatory post-steps after quantizing Qwen3.8-27B via the wrapper class.
|
||||
|
||||
The wrapper-class save drops the MTP head and the vision preprocessor configs.
|
||||
All three of these have bitten previous rounds:
|
||||
|
||||
1. graft `model-mtp.safetensors` verbatim from the bf16 source and register its
|
||||
tensors in the output index (else no speculative decoding at all);
|
||||
2. restore preprocessor_config.json / processor_config.json /
|
||||
video_preprocessor_config.json (else the vision tower can't preprocess);
|
||||
3. VERIFY `re:^mtp.*` is in quantization_config.ignore -- if it is missing,
|
||||
vLLM loads the grafted bf16 MTP head as though it were quantized and it
|
||||
comes up uninitialised, giving 0% acceptance. This is THE bug that cost two
|
||||
prior rounds; it is verified here rather than assumed.
|
||||
"""
|
||||
import json, os, shutil, sys
|
||||
|
||||
def main():
|
||||
src, out = sys.argv[1], sys.argv[2]
|
||||
fail = []
|
||||
|
||||
# --- 1. MTP graft ---------------------------------------------------------
|
||||
mtp_src = os.path.join(src, "model-mtp.safetensors")
|
||||
mtp_dst = os.path.join(out, "model-mtp.safetensors")
|
||||
if not os.path.exists(mtp_src):
|
||||
fail.append(f"missing MTP shard at {mtp_src}")
|
||||
else:
|
||||
if not os.path.exists(mtp_dst):
|
||||
print(f"copying MTP shard ({os.path.getsize(mtp_src)/1e9:.2f} GB) ...", flush=True)
|
||||
shutil.copy2(mtp_src, mtp_dst)
|
||||
else:
|
||||
print("MTP shard already present")
|
||||
|
||||
src_idx = json.load(open(os.path.join(src, "model.safetensors.index.json")))
|
||||
mtp_keys = [k for k in src_idx["weight_map"] if k.startswith("mtp")]
|
||||
out_idx_p = os.path.join(out, "model.safetensors.index.json")
|
||||
out_idx = json.load(open(out_idx_p))
|
||||
added = 0
|
||||
for k in mtp_keys:
|
||||
if k not in out_idx["weight_map"]:
|
||||
out_idx["weight_map"][k] = "model-mtp.safetensors"
|
||||
added += 1
|
||||
if added:
|
||||
json.dump(out_idx, open(out_idx_p, "w"), indent=2)
|
||||
print(f"MTP tensors in source: {len(mtp_keys)}; added to output index: {added}; "
|
||||
f"now present: {sum(1 for k in out_idx['weight_map'] if k.startswith('mtp'))}")
|
||||
if len(mtp_keys) == 0:
|
||||
fail.append("source index had NO mtp tensors")
|
||||
|
||||
# --- 2. preprocessor / processor configs ---------------------------------
|
||||
for fn in ("preprocessor_config.json", "processor_config.json",
|
||||
"video_preprocessor_config.json", "chat_template.jinja",
|
||||
"generation_config.json"):
|
||||
s = os.path.join(src, fn)
|
||||
d = os.path.join(out, fn)
|
||||
if os.path.exists(s) and not os.path.exists(d):
|
||||
shutil.copy2(s, d)
|
||||
print(f"restored {fn}")
|
||||
elif os.path.exists(d):
|
||||
print(f"{fn} already present")
|
||||
else:
|
||||
print(f"NOTE: {fn} absent in source, skipped")
|
||||
if not os.path.exists(os.path.join(out, "preprocessor_config.json")):
|
||||
fail.append("preprocessor_config.json missing from output (vision will break)")
|
||||
|
||||
# --- 3. verify the mtp ignore --------------------------------------------
|
||||
cfg_p = os.path.join(out, "config.json")
|
||||
cfg = json.load(open(cfg_p))
|
||||
ig = cfg.get("quantization_config", {}).get("ignore", [])
|
||||
has = any("mtp" in x for x in ig)
|
||||
if not has:
|
||||
# llm-compressor PRUNES ignore entries that matched no module at quant
|
||||
# time. The wrapper class never loads the MTP head, so `re:^mtp.*`
|
||||
# matches nothing and silently vanishes from the saved config -- and
|
||||
# then vLLM treats the freshly grafted bf16 MTP head as quantized and
|
||||
# brings it up uninitialised (0% acceptance). Re-inject it here, AFTER
|
||||
# the graft. This is the two-rounds-lost bug; repair, then re-verify.
|
||||
ig.append("re:^mtp.*")
|
||||
cfg["quantization_config"]["ignore"] = ig
|
||||
json.dump(cfg, open(cfg_p, "w"), indent=2)
|
||||
print("REPAIRED: re-injected 're:^mtp.*' into quantization_config.ignore "
|
||||
"(llm-compressor pruned it -- it matched no module at quant time)")
|
||||
cfg = json.load(open(cfg_p))
|
||||
ig = cfg["quantization_config"]["ignore"]
|
||||
has = any("mtp" in x for x in ig)
|
||||
print(f"quantization_config.ignore has an mtp entry: {has} "
|
||||
f"({[x for x in ig if 'mtp' in x]})")
|
||||
if not has:
|
||||
fail.append("re:^mtp.* NOT in ignore -- MTP would load uninitialised (0% acceptance)")
|
||||
|
||||
# --- report ---------------------------------------------------------------
|
||||
print("\nformat:", cfg.get("quantization_config", {}).get("format"))
|
||||
print("config_groups:", list(cfg.get("quantization_config", {}).get("config_groups", {})))
|
||||
if fail:
|
||||
print("\nFAILED CHECKS:")
|
||||
for f in fail:
|
||||
print(" -", f)
|
||||
return 1
|
||||
print("\nall post-steps OK")
|
||||
return 0
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user