#!/bin/bash # Serve the modelopt NVFP4 + MTP Heretic2 seat (the WORKING fast char-rp-reasoning seat). # ~77 tok/s, MTP acceptance 32-40%. Requires: (1) heretic2-modelopt-nvfp4-mtp built (quant -> # finalize), (2) the sitecustomize MTP workaround mounted on PYTHONPATH (vLLM 0.24 draft-model # exclude bug — see runbook landmine #4; without it the engine crashes on a shape mismatch). set -euo pipefail MODEL="${1:-/tank/aimodels/heretic2-nvfp4-work/heretic2-modelopt-nvfp4-mtp}" # Dir containing sitecustomize.py (a copy of sitecustomize-mtp-workaround.py named sitecustomize.py): WORKAROUND_DIR="${MTP_WORKAROUND_DIR:-/home/lkraven/isls_debug}" docker rm -f vllm-charrp-modelopt 2>/dev/null || true docker run -d --name vllm-charrp-modelopt --gpus '"device=0"' --ipc host \ -v /tank/aimodels:/tank/aimodels \ -v "${WORKAROUND_DIR}":/lk_debug -e PYTHONPATH=/lk_debug \ -p 8018:8000 \ vllm/vllm-openai:v0.24.0 \ "$MODEL" \ --quantization modelopt \ --speculative-config '{"method":"qwen3_5_mtp","num_speculative_tokens":3}' \ --language-model-only \ --mamba-cache-dtype float32 \ --reasoning-parser qwen3 --tool-call-parser qwen3_coder --enable-auto-tool-choice \ --served-model-name char-rp-reasoning \ --max-model-len 40960 --max-num-seqs 32 --gpu-memory-utilization 0.5 --trust-remote-code echo "started: $(docker ps --filter name=vllm-charrp-modelopt --format '{{.Status}}')" echo "verify MTP: docker logs vllm-charrp-modelopt 2>&1 | grep -E 'mtp-workaround|SpecDecoding'"