# qwen36-27b-aeon tunables — copy to .env on the host, edit there (never commit .env). # The real .env lives at /opt/docker/compose/qwen36-27b-aeon/.env on ana-ml2. # Per-service vLLM image. Both on latest (0.24.0) after validating it on rp. Reasoning works # on 0.24.0 (and did on 0.23.0 too — the trace bug was a LiteLLM shared-config mutation, not # vLLM). Rollback: set either back to vllm/vllm-openai:known-good-0.23.0 (tagged on the host). AEON_GEN_IMAGE=vllm/vllm-openai:latest AEON_RP_IMAGE=vllm/vllm-openai:latest # Shared vLLM API key (matches the litellm VLLM_API_KEY). Blank = no auth. API_KEY= # Both instances pin to the same physical GPU (the freed qwopus slot). AEON_GPU_ID=0 # ── General serve (MTP off) → gen / gen-reasoning / summarizer-large ── AEON_GEN_CONTAINER_NAME=vllm-aeon-gen AEON_GEN_PORT=8015 AEON_GEN_SERVED_NAME=qwen3.6-27b-aeon AEON_GEN_SERVED_NAME_THINK=qwen3.6-27b-aeon-thinking AEON_GEN_MODEL=/tank/aimodels/qwen36-27b-aeon-nvfp4 # NOTE: live gen serves qwen3.6-35b-a3b-heretic since 2026-07-08 (aeon-27b displaced); # SERVED_NAME/MODEL above lag live — pending reconciliation with the model swap. # util/ctx/seqs RE-TUNED 2026-07-16 (char-rp left GPU-0 → room): util 0.30->0.42, # ctx 131072->262144 (256K native), seqs 16->32. gen MoE KV is cheap → 5.43x @ 256K. AEON_GEN_GPU_MEM_UTIL=0.42 AEON_GEN_MAX_MODEL_LEN=262144 AEON_GEN_MAX_NUM_SEQS=32 AEON_GEN_KV_CACHE_DTYPE=fp8 # Reasoning parser — KEEP qwen3. This checkpoint's chat_template.jinja injects the opening # into the PROMPT (output has only ). qwen3 handles the non-thinking path # correctly. NOTE: enable_thinking:true does NOT surface reasoning_content under EITHER qwen3 # or deepseek_r1 (the reasoning is generated but dropped) — that needs a chat_template fix, # NOT a parser swap. deepseek_r1 was tested and is WORSE: it breaks the non-thinking path # too (both content and reasoning_content come back empty). Do not use deepseek_r1 here. AEON_GEN_REASONING_PARSER=qwen3 # ── RP seat (native MTP on) → char-rp ── AEON_RP_CONTAINER_NAME=vllm-aeon-rp AEON_RP_PORT=8016 AEON_RP_SERVED_NAME=qwen3.6-27b-aeon-rp AEON_RP_SERVED_NAME_THINK=qwen3.6-27b-aeon-rp-thinking # Full 27GB variant by default; switch to the 21GB XS for ~6GB more co-location margin: # AEON_RP_MODEL=/tank/aimodels/qwen36-27b-aeon-nvfp4-xs AEON_RP_MODEL=/tank/aimodels/qwen36-27b-aeon-nvfp4 AEON_RP_GPU_MEM_UTIL=0.40 AEON_RP_MAX_MODEL_LEN=65536 AEON_RP_MAX_NUM_SEQS=2 AEON_RP_KV_CACHE_DTYPE=fp8 AEON_RP_REASONING_PARSER=qwen3 # Native MTP head. If stock vLLM names it differently, try method=mtp. AEON_RP_SPEC_METHOD=qwen3_5_mtp AEON_RP_SPEC_TOKENS=3