gen-seat: deploy Qwen3.8-27B-Uncensored gen seat, rename qwen36-27b-aeon->gen-seat

- New uncensored gen seat: JonathanColetti/Qwen3.8-27B-Uncensored, in-house NVFP4
  W4A16 (compressed-tensors) + grafted bf16 MTP (config ignore re:^mtp.*), vision-intact,
  262K ctx, MTP n=3 (~42% accept, ~68 tok/s). Replaces the qwen3.6-35b-a3b-heretic MoE.
- Rename compose project qwen36-27b-aeon -> gen-seat, container vllm-aeon-gen -> vllm-gen,
  env vars AEON_GEN_* -> GEN_*; drop the dormant vllm-aeon-rp service.
- litellm: repoint 7 aliases (gen/summarizer/summarizer-large/classifier/image-judge/
  qwen-image-bench -> qwen3.8-27b-uncensored; gen-reasoning -> -thinking).
- servers/ana-ml2/README: refresh the gen hero-seat row.
This commit is contained in:
vh
2026-08-15 00:19:20 -07:00
parent dac4acf0c5
commit 680c30e778
6 changed files with 148 additions and 295 deletions
+18
View File
@@ -0,0 +1,18 @@
# gen-seat tunables — fleet `gen` seat (ana-ml2 GPU 0, :8015). Edit here, never commit.
GEN_IMAGE=vllm/vllm-openai:latest
API_KEY=
GEN_GPU_ID=0
GEN_CONTAINER_NAME=vllm-gen
GEN_PORT=8015
GEN_SERVED_NAME=qwen3.8-27b-uncensored
GEN_SERVED_NAME_THINK=qwen3.8-27b-uncensored-thinking
GEN_MODEL=/tank/aimodels/qwen38-27b-uncensored-nvfp4
GEN_QUANT=compressed-tensors
GEN_GPU_MEM_UTIL=0.45
GEN_MAX_MODEL_LEN=262144
GEN_MAX_NUM_SEQS=16
GEN_KV_CACHE_DTYPE=fp8
GEN_REASONING_PARSER=qwen3
GEN_SPEC_METHOD=qwen3_5_mtp
GEN_SPEC_TOKENS=3
+97
View File
@@ -0,0 +1,97 @@
# gen-seat — the fleet's general `gen` seat on ana-ml2 GPU 0 (:8015).
# Serves JonathanColetti/Qwen3.8-27B-Uncensored (Heretic abliteration KL 0.12, vision-intact
# Qwen3_5ForConditionalGeneration, grafted MTP head), quantized in-house to NVFP4 W4A16
# (compressed-tensors) with the bf16 MTP grafted back. ⚠ the grafted MTP requires
# `re:^mtp.*` in config.json quantization_config.ignore or vLLM loads it uninitialized (0% accept).
# Backs gateway aliases: gen, gen-reasoning, summarizer, summarizer-large, classifier,
# image-judge, qwen-image-bench (all via api_base :8015).
#
# Two served-names (base + `-thinking`): LiteLLM keys deployments by (model, api_base), so gen
# and gen-reasoning use distinct names to avoid the shared-config enable_thinking clobber.
# --mamba-cache-dtype float32 for the GDN/hybrid-linear-attn state; MTP via qwen3_5_mtp n=3;
# full 262K context (hybrid attention -> cheap KV, ~8.6GB at 262K). All tunables in .env.
# (Renamed from qwen36-27b-aeon/vllm-aeon-gen; dormant RP service dropped 2026-08-15.)
name: gen-seat
services:
vllm-gen:
image: ${GEN_IMAGE:-vllm/vllm-openai:latest}
container_name: ${GEN_CONTAINER_NAME:-vllm-gen}
restart: unless-stopped
ipc: host
ports:
- "${GEN_PORT:-8015}:8000"
volumes:
- /tank/aimodels/huggingface:/hfcache
- ${GEN_MODEL:-/tank/aimodels/qwen38-27b-uncensored-nvfp4}:/model:ro
environment:
- HF_HOME=/hfcache
- HF_HUB_CACHE=/hfcache/hub
- VLLM_API_KEY=${API_KEY:-}
- PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
command:
- /model
- --served-model-name
- ${GEN_SERVED_NAME:-qwen3.8-27b-uncensored}
- ${GEN_SERVED_NAME_THINK:-qwen3.8-27b-uncensored-thinking}
- --host
- 0.0.0.0
- --port
- "8000"
- --quantization
- ${GEN_QUANT:-compressed-tensors}
- --gpu-memory-utilization
- ${GEN_GPU_MEM_UTIL:-0.45}
- --max-model-len
- ${GEN_MAX_MODEL_LEN:-262144}
- --max-num-seqs
- ${GEN_MAX_NUM_SEQS:-16}
- --max-num-batched-tokens
- "16384"
- --trust-remote-code
- --dtype
- auto
- --mamba-cache-dtype
- float32
- --kv-cache-dtype
- ${GEN_KV_CACHE_DTYPE:-fp8}
- --enable-prefix-caching
- --enable-chunked-prefill
- --limit-mm-per-prompt
- '{"image": 4}'
- --reasoning-parser
- ${GEN_REASONING_PARSER:-qwen3}
- --enable-auto-tool-choice
- --tool-call-parser
- qwen3_coder
- --speculative-config
- '{"method": "${GEN_SPEC_METHOD:-qwen3_5_mtp}", "num_speculative_tokens": ${GEN_SPEC_TOKENS:-3}}'
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids:
- "${GEN_GPU_ID:-0}"
capabilities:
- gpu
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
interval: 30s
timeout: 10s
retries: 3
start_period: 900s
networks:
- tnet
labels:
- homepage.group=AI - Inference
- homepage.name=Qwen3.8-27B Uncensored (NVFP4, vision, MTP) — gen
- homepage.icon=mdi-creation
- homepage.description=Uncensored Qwen3.8-27B multimodal NVFP4+MTP, the `gen` seat (ana-ml2 GPU 0)
- homepage.href=http://10.250.50.54:${GEN_PORT:-8015}/docs
networks:
tnet:
name: traefik-net
external: true