# qwen36-27b-aeon — AEON-7/Qwen3.6-27B-AEON-Ultimate-Uncensored on ana-ml2 GPU 0, # REPLACING qwopus3.5-122b as the `gen` model (operator 2026-07-05). # # Dense 27B, qwen3_5 GDN-hybrid arch (full-attn + Gated DeltaNet SSM) — same family # as qwopus, VISION-INTACT (Qwen3_5ForConditionalGeneration, vision tower preserved # at bf16), abliterated (abliterix v1.4, 0/100 refusals), native MTP head grafted, # Apache-2.0, 131K default ctx. Served NVFP4 (ModelOpt) on Blackwell's FP4 cores. # # TWO CO-LOCATED INSTANCES on GPU 0 (operator wants both behaviours resident at once; # MTP is a serve-time config, NOT per-request, so one endpoint can't do both): # vllm-aeon-gen (:8015, served qwen3.6-27b-aeon) — MTP OFF, general/concurrent # serve. Backs gateway gen / gen-reasoning / summarizer-large. # vllm-aeon-rp (:8016, served qwen3.6-27b-aeon-rp) — native MTP ON (qwen3_5_mtp # n=3), low-concurrency single-seat RP. Backs gateway char-rp. # Why MTP off for the general serve: measured on qwopus, MTP helps single-stream # (+12% N=1) but HURTS moderate concurrency (-15..-20% N=4) and silently drops # min_p/logit_bias — wrong for a shared multi-consumer endpoint. Right only for a # dedicated single-stream seat (the RP one). [[reference_gen_qwopus_122b]] # # VRAM budget (2 weight copies, no sharing): full NVFP4 = 27GB ea. gen util 0.45 # (~43GB) + rp util 0.40 (~38GB) = ~81GB / 96GB, ~15GB margin. depends_on: # service_healthy sequences gen-first so the util reservation doesn't race → OOM. # If margin bites at warmup (vision-encoder + big-vocab sampler warmup, cf. qwen36-vl), # point AEON_RP_MODEL at the 21GB XS variant (frees ~6GB) via .env — no compose edit. # # NVFP4 is ModelOpt format → --quantization modelopt (vLLM also auto-detects; explicit # is belt-and-suspenders). --mamba-cache-dtype float32 for the GDN/SSM state (AEON # deploy guide + vLLM recipe Mamba-cache note). Tool-calling qwen3_coder + reasoning # qwen3 (per the AEON card), same as qwopus. # # ⚠️ DEPLOYABILITY — load-test before trusting: multimodal + ModelOpt-NVFP4 on THIS # brand-new arch, and native qwen3_5_mtp spec-decode, are unproven on our stock vLLM # image. If stock can't serve it, the AEON patched image (ghcr.io/aeon-7/aeon-vllm- # ultimate, PRs #41703/#40898) is the fallback — but that's really for DFlash; native # MTP + base inference should ride stock >= 0.23.0. Set AEON_IMAGE in .env. # # REVERT: `docker compose down` here + `docker compose up -d` the qwopus3.5-122b stack # (still staged) + revert the litellm gen/gen-reasoning/summarizer-large records. # All tunables in .env — edit that, not this file. name: qwen36-27b-aeon services: # ── General serve — MTP OFF, concurrent. gen / gen-reasoning / summarizer-large. ── vllm-aeon-gen: # Per-service image so gen can stay pinned to a known-good vLLM while rp tests a new one. image: ${AEON_GEN_IMAGE:-vllm/vllm-openai:latest} container_name: ${AEON_GEN_CONTAINER_NAME:-vllm-aeon-gen} restart: unless-stopped ipc: host ports: - "${AEON_GEN_PORT:-8015}:8000" volumes: - /tank/aimodels/huggingface:/hfcache - ${AEON_GEN_MODEL:-/tank/aimodels/qwen36-27b-aeon-nvfp4}:/model:ro environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub - VLLM_API_KEY=${API_KEY:-} # Reclaims PyTorch reserved-but-unallocated fragmentation so the co-located # util split doesn't strand VRAM (same knob qwopus needed for the MoE workspace). - PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True command: - /model # TWO served-names: base + a `-thinking` alias. LiteLLM keys deployments by # (model, api_base), so gen and gen-reasoning MUST use distinct model names or a # thinking-off request mutates the shared litellm_params and clobbers the other's # enable_thinking (the shared-config-mutation footgun). gen-reasoning routes to the # `-thinking` name; gen/summarizer-large route to the base name. - --served-model-name - ${AEON_GEN_SERVED_NAME:-qwen3.6-27b-aeon} - ${AEON_GEN_SERVED_NAME_THINK:-qwen3.6-27b-aeon-thinking} - --host - 0.0.0.0 - --port - "8000" - --quantization - modelopt - --gpu-memory-utilization - ${AEON_GEN_GPU_MEM_UTIL:-0.45} - --max-model-len - ${AEON_GEN_MAX_MODEL_LEN:-131072} # Keep concurrency modest: big-vocab sampler warmup allocates a large tensor # (qwen36-vl OOM'd at the default 1024 on a shared GPU). 16 is ample here. - --max-num-seqs - ${AEON_GEN_MAX_NUM_SEQS:-16} - --max-num-batched-tokens - "16384" - --trust-remote-code - --dtype - auto # GDN/SSM (Gated DeltaNet) state cache — float32 per the AEON deploy guide. - --mamba-cache-dtype - float32 - --kv-cache-dtype - ${AEON_GEN_KV_CACHE_DTYPE:-fp8} - --enable-prefix-caching - --enable-chunked-prefill - --limit-mm-per-prompt - '{"image": 4}' - --reasoning-parser - ${AEON_GEN_REASONING_PARSER:-qwen3} - --enable-auto-tool-choice - --tool-call-parser - qwen3_coder deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${AEON_GPU_ID:-0}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 900s networks: - tnet labels: - homepage.group=AI Systems - homepage.name=Qwen3.6-27B AEON (NVFP4, vision) — gen - homepage.icon=mdi-creation - homepage.description=Uncensored Qwen3.6-27B multimodal NVFP4, the `gen` model (ana-ml2 GPU 0) - homepage.href=http://10.250.50.54:${AEON_GEN_PORT:-8015}/docs # ── RP seat — native MTP ON, low concurrency, single-seat. char-rp. ── vllm-aeon-rp: image: ${AEON_RP_IMAGE:-vllm/vllm-openai:latest} container_name: ${AEON_RP_CONTAINER_NAME:-vllm-aeon-rp} restart: unless-stopped ipc: host # Sequence AFTER the general serve is healthy so the two util reservations on the # shared GPU don't race into an OOM (gen reserves its 0.45 first, then rp its 0.40). depends_on: vllm-aeon-gen: condition: service_healthy ports: - "${AEON_RP_PORT:-8016}:8000" volumes: - /tank/aimodels/huggingface:/hfcache # Point at the XS (21GB) variant via .env to buy ~6GB co-location margin. - ${AEON_RP_MODEL:-/tank/aimodels/qwen36-27b-aeon-nvfp4}:/model:ro environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub - VLLM_API_KEY=${API_KEY:-} - PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True command: - /model # base + `-thinking` alias (see the gen note): char-rp -> base, char-rp-reasoning -> -thinking. - --served-model-name - ${AEON_RP_SERVED_NAME:-qwen3.6-27b-aeon-rp} - ${AEON_RP_SERVED_NAME_THINK:-qwen3.6-27b-aeon-rp-thinking} - --host - 0.0.0.0 - --port - "8000" - --quantization - modelopt - --gpu-memory-utilization - ${AEON_RP_GPU_MEM_UTIL:-0.40} - --max-model-len - ${AEON_RP_MAX_MODEL_LEN:-65536} # Single-seat: low concurrency keeps warmup + KV small so it fits alongside gen. - --max-num-seqs - ${AEON_RP_MAX_NUM_SEQS:-2} - --trust-remote-code - --dtype - auto - --mamba-cache-dtype - float32 - --kv-cache-dtype - ${AEON_RP_KV_CACHE_DTYPE:-fp8} - --enable-prefix-caching - --limit-mm-per-prompt - '{"image": 4}' - --reasoning-parser - ${AEON_RP_REASONING_PARSER:-qwen3} # Native MTP speculative decode (the grafted head). n=3 per the AEON card's # measured accept length (~3.3/3). MTP is why this seat exists separately. - --speculative-config - '{"method": "${AEON_RP_SPEC_METHOD:-qwen3_5_mtp}", "num_speculative_tokens": ${AEON_RP_SPEC_TOKENS:-3}}' deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${AEON_GPU_ID:-0}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 900s networks: - tnet labels: - homepage.group=AI Systems - homepage.name=Qwen3.6-27B AEON RP (NVFP4 + MTP) — char-rp - homepage.icon=mdi-drama-masks - homepage.description=Uncensored Qwen3.6-27B, native MTP single-seat RP (char-rp), ana-ml2 GPU 0 - homepage.href=http://10.250.50.54:${AEON_RP_PORT:-8016}/docs networks: tnet: name: traefik-net external: true