gen-seat: deploy Qwen3.8-27B-Uncensored gen seat, rename qwen36-27b-aeon->gen-seat
- New uncensored gen seat: JonathanColetti/Qwen3.8-27B-Uncensored, in-house NVFP4 W4A16 (compressed-tensors) + grafted bf16 MTP (config ignore re:^mtp.*), vision-intact, 262K ctx, MTP n=3 (~42% accept, ~68 tok/s). Replaces the qwen3.6-35b-a3b-heretic MoE. - Rename compose project qwen36-27b-aeon -> gen-seat, container vllm-aeon-gen -> vllm-gen, env vars AEON_GEN_* -> GEN_*; drop the dormant vllm-aeon-rp service. - litellm: repoint 7 aliases (gen/summarizer/summarizer-large/classifier/image-judge/ qwen-image-bench -> qwen3.8-27b-uncensored; gen-reasoning -> -thinking). - servers/ana-ml2/README: refresh the gen hero-seat row.
This commit is contained in:
@@ -0,0 +1,18 @@
|
||||
# gen-seat tunables — fleet `gen` seat (ana-ml2 GPU 0, :8015). Edit here, never commit.
|
||||
GEN_IMAGE=vllm/vllm-openai:latest
|
||||
API_KEY=
|
||||
GEN_GPU_ID=0
|
||||
|
||||
GEN_CONTAINER_NAME=vllm-gen
|
||||
GEN_PORT=8015
|
||||
GEN_SERVED_NAME=qwen3.8-27b-uncensored
|
||||
GEN_SERVED_NAME_THINK=qwen3.8-27b-uncensored-thinking
|
||||
GEN_MODEL=/tank/aimodels/qwen38-27b-uncensored-nvfp4
|
||||
GEN_QUANT=compressed-tensors
|
||||
GEN_GPU_MEM_UTIL=0.45
|
||||
GEN_MAX_MODEL_LEN=262144
|
||||
GEN_MAX_NUM_SEQS=16
|
||||
GEN_KV_CACHE_DTYPE=fp8
|
||||
GEN_REASONING_PARSER=qwen3
|
||||
GEN_SPEC_METHOD=qwen3_5_mtp
|
||||
GEN_SPEC_TOKENS=3
|
||||
@@ -0,0 +1,97 @@
|
||||
# gen-seat — the fleet's general `gen` seat on ana-ml2 GPU 0 (:8015).
|
||||
# Serves JonathanColetti/Qwen3.8-27B-Uncensored (Heretic abliteration KL 0.12, vision-intact
|
||||
# Qwen3_5ForConditionalGeneration, grafted MTP head), quantized in-house to NVFP4 W4A16
|
||||
# (compressed-tensors) with the bf16 MTP grafted back. ⚠ the grafted MTP requires
|
||||
# `re:^mtp.*` in config.json quantization_config.ignore or vLLM loads it uninitialized (0% accept).
|
||||
# Backs gateway aliases: gen, gen-reasoning, summarizer, summarizer-large, classifier,
|
||||
# image-judge, qwen-image-bench (all via api_base :8015).
|
||||
#
|
||||
# Two served-names (base + `-thinking`): LiteLLM keys deployments by (model, api_base), so gen
|
||||
# and gen-reasoning use distinct names to avoid the shared-config enable_thinking clobber.
|
||||
# --mamba-cache-dtype float32 for the GDN/hybrid-linear-attn state; MTP via qwen3_5_mtp n=3;
|
||||
# full 262K context (hybrid attention -> cheap KV, ~8.6GB at 262K). All tunables in .env.
|
||||
# (Renamed from qwen36-27b-aeon/vllm-aeon-gen; dormant RP service dropped 2026-08-15.)
|
||||
|
||||
name: gen-seat
|
||||
|
||||
services:
|
||||
vllm-gen:
|
||||
image: ${GEN_IMAGE:-vllm/vllm-openai:latest}
|
||||
container_name: ${GEN_CONTAINER_NAME:-vllm-gen}
|
||||
restart: unless-stopped
|
||||
ipc: host
|
||||
ports:
|
||||
- "${GEN_PORT:-8015}:8000"
|
||||
volumes:
|
||||
- /tank/aimodels/huggingface:/hfcache
|
||||
- ${GEN_MODEL:-/tank/aimodels/qwen38-27b-uncensored-nvfp4}:/model:ro
|
||||
environment:
|
||||
- HF_HOME=/hfcache
|
||||
- HF_HUB_CACHE=/hfcache/hub
|
||||
- VLLM_API_KEY=${API_KEY:-}
|
||||
- PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True
|
||||
command:
|
||||
- /model
|
||||
- --served-model-name
|
||||
- ${GEN_SERVED_NAME:-qwen3.8-27b-uncensored}
|
||||
- ${GEN_SERVED_NAME_THINK:-qwen3.8-27b-uncensored-thinking}
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
- --port
|
||||
- "8000"
|
||||
- --quantization
|
||||
- ${GEN_QUANT:-compressed-tensors}
|
||||
- --gpu-memory-utilization
|
||||
- ${GEN_GPU_MEM_UTIL:-0.45}
|
||||
- --max-model-len
|
||||
- ${GEN_MAX_MODEL_LEN:-262144}
|
||||
- --max-num-seqs
|
||||
- ${GEN_MAX_NUM_SEQS:-16}
|
||||
- --max-num-batched-tokens
|
||||
- "16384"
|
||||
- --trust-remote-code
|
||||
- --dtype
|
||||
- auto
|
||||
- --mamba-cache-dtype
|
||||
- float32
|
||||
- --kv-cache-dtype
|
||||
- ${GEN_KV_CACHE_DTYPE:-fp8}
|
||||
- --enable-prefix-caching
|
||||
- --enable-chunked-prefill
|
||||
- --limit-mm-per-prompt
|
||||
- '{"image": 4}'
|
||||
- --reasoning-parser
|
||||
- ${GEN_REASONING_PARSER:-qwen3}
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- qwen3_coder
|
||||
- --speculative-config
|
||||
- '{"method": "${GEN_SPEC_METHOD:-qwen3_5_mtp}", "num_speculative_tokens": ${GEN_SPEC_TOKENS:-3}}'
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
device_ids:
|
||||
- "${GEN_GPU_ID:-0}"
|
||||
capabilities:
|
||||
- gpu
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 3
|
||||
start_period: 900s
|
||||
networks:
|
||||
- tnet
|
||||
labels:
|
||||
- homepage.group=AI - Inference
|
||||
- homepage.name=Qwen3.8-27B Uncensored (NVFP4, vision, MTP) — gen
|
||||
- homepage.icon=mdi-creation
|
||||
- homepage.description=Uncensored Qwen3.8-27B multimodal NVFP4+MTP, the `gen` seat (ana-ml2 GPU 0)
|
||||
- homepage.href=http://10.250.50.54:${GEN_PORT:-8015}/docs
|
||||
|
||||
networks:
|
||||
tnet:
|
||||
name: traefik-net
|
||||
external: true
|
||||
Reference in New Issue
Block a user