#!/bin/bash # Launch the modelopt NVFP4 quant of the grafted Heretic2 on ana-ml2 GPU0 (detached, survives ssh # drop). bare nvidia-modelopt (0.45). The FusedMoE-compat guard + single-shard export + the # multimodal load class are all inside quant_modelopt.py. ~18 min. Output: heretic2-modelopt-nvfp4 # (no mtp yet — run finalize_modelopt_mtp.py after). quant_modelopt.py must be at /lk/quant_modelopt.py # (mount /home/lkraven as /lk, or scp it there first). set -euo pipefail docker rm -f vllm-heretic2-modelopt-quant 2>/dev/null || true rm -rf /tank/aimodels/heretic2-nvfp4-work/heretic2-modelopt-nvfp4 2>/dev/null || true docker run -d --name vllm-heretic2-modelopt-quant --gpus '"device=0"' --ipc host \ -v /tank/aimodels:/tank/aimodels -v /home/lkraven:/lk \ --entrypoint bash vllm/vllm-openai:v0.24.0 -c ' set -e pip install -q nvidia-modelopt tiktoken sentencepiece 2>&1 | tail -1 python3 /lk/quant_modelopt.py \ --model /tank/aimodels/heretic2-nvfp4-work/heretic2-mtp-bf16 \ --calib-mode chat --calib /tank/aimodels/heretic2-nvfp4-work/production_calib_512.jsonl \ --num-samples 512 --seqlen 8192 \ --out /tank/aimodels/heretic2-nvfp4-work/heretic2-modelopt-nvfp4' echo "LAUNCHED: $(docker ps --filter name=vllm-heretic2-modelopt-quant --format '{{.Status}}')"