# vLLM — Qwen3 Embedding + Reranker + Skywork Reward-V2 classifier. # # Originally created to replace the unmaintained Infinity stack (embed + # rerank); generalized 2026-05-13 to host any vLLM-served model on ana-ml2, # starting with the Skywork-Reward-V2-Llama-3.1-8B reward classifier # (AWQ-quantized locally, served from /tank/aimodels/llm/). # # vLLM runs one model per process, so this stack brings up three containers # sharing a single GPU: # # vllm-embed — Qwen3-Embedding served as an OpenAI /v1/embeddings server # vllm-rerank — Qwen3-Reranker served as a /rerank + /score server # vllm-reward — Skywork-Reward-V2-Llama-3.1-8B-AWQ served as a /classify scorer # # The reranker is a causal-LM checkpoint; --hf-overrides re-maps it to # Qwen3ForSequenceClassification so vLLM's reranking endpoints work and the # model only emits two class logits (no/yes) instead of the full 151k vocab. # # All tunables live in .env — edit that, not this file. # # Pre-download models to avoid first-run delay: # scripts/elway ana-ml2 --playbook playbooks/pull-hf-repo.yaml \ # --var hf_repo=Qwen/Qwen3-Embedding-0.6B # scripts/elway ana-ml2 --playbook playbooks/pull-hf-repo.yaml \ # --var hf_repo=Qwen/Qwen3-Reranker-0.6B # # Skywork-Reward-V2-Llama-3.1-8B-AWQ is a locally-quantized model — lives at # /tank/aimodels/llm/Skywork-Reward-V2-Llama-3.1-8B-AWQ on ana-ml2 and is # bind-mounted into the reward service at /local-models. Not from HF Hub. services: vllm-embed: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-embed restart: unless-stopped ipc: host ports: - "${EMBED_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} - VLLM_API_KEY=${API_KEY:-} command: - ${EMBED_MODEL} - --served-model-name - ${EMBED_MODEL} - --runner - pooling - --host - 0.0.0.0 - --port - "8000" - --gpu-memory-utilization - ${EMBED_GPU_MEM_UTIL} - --max-model-len - ${EMBED_MAX_MODEL_LEN} - --dtype - auto deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${GPU_ID}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 180s networks: - tnet labels: - homepage.group=AI - Eval & Retrieval - homepage.name=vLLM Embed (Qwen3) - homepage.icon=mdi-vector-arrange-below - homepage.description=Qwen3 Embedding via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${EMBED_PORT}/docs vllm-rerank: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-rerank restart: unless-stopped ipc: host ports: - "${RERANK_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} - VLLM_API_KEY=${API_KEY:-} command: - ${RERANK_MODEL} - --served-model-name - ${RERANK_MODEL} - --runner - pooling - --hf-overrides - '{"architectures":["Qwen3ForSequenceClassification"],"classifier_from_token":["no","yes"],"is_original_qwen3_reranker":true}' - --host - 0.0.0.0 - --port - "8000" - --gpu-memory-utilization - ${RERANK_GPU_MEM_UTIL} - --max-model-len - ${RERANK_MAX_MODEL_LEN} - --dtype - auto deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${GPU_ID}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 180s networks: - tnet labels: - homepage.group=AI - Eval & Retrieval - homepage.name=vLLM Rerank (Qwen3) - homepage.icon=mdi-sort-variant - homepage.description=Qwen3 Reranker via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${RERANK_PORT}/docs vllm-reward: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-reward restart: unless-stopped ipc: host ports: - "${REWARD_PORT}:8000" volumes: # AWQ output lives in the legacy llama-swap models tree, not the HF cache # — bind-mount the LLM models dir read-only so the reward service can # load it as a local-path HF-format model. - /tank/aimodels/llm:/local-models:ro environment: - VLLM_API_KEY=${API_KEY:-} command: - /local-models/Skywork-Reward-V2-Llama-3.1-8B-AWQ - --served-model-name - Skywork/Skywork-Reward-V2-Llama-3.1-8B-AWQ # vLLM 0.19.1 deprecated --task in favor of --runner. The model's # config.json declares `LlamaForSequenceClassification` so the # pooling runner uses it as a classifier (single-label reward score) # without needing an explicit task flag. - --runner - pooling - --host - 0.0.0.0 - --port - "8000" - --gpu-memory-utilization - ${REWARD_GPU_MEM_UTIL} - --max-model-len - ${REWARD_MAX_MODEL_LEN} - --dtype - auto deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${GPU_ID}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 240s networks: - tnet labels: - homepage.group=AI - Eval & Retrieval - homepage.name=vLLM Reward (Skywork) - homepage.icon=mdi-scale-balance - homepage.description=Skywork-Reward-V2 8B classifier via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${REWARD_PORT}/docs # Granite 4.1 8B (FP8) — production summarizer (replaced phi4-mini # 2026-06-05, which had superseded the llama-swap granite-4-small pin). # Generative chat model (OpenAI /v1/chat/completions), so NO --runner # pooling. FP8 on RTX PRO 6000 Blackwell (cc 12.0): near-lossless, ~1.2x, ~6 GB. vllm-granite: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-granite restart: unless-stopped ipc: host ports: - "${GRANITE_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} - VLLM_API_KEY=${API_KEY:-} command: # Production summarizer (replaced phi4-mini 2026-06-05). Default = official # IBM pre-quantized FP8 (compressed-tensors), loaded directly; FP8 is native # on the RTX PRO 6000 Blackwell (cc 12.0). Fallback to vLLM-native dynamic FP8 from # BF16: GRANITE_MODEL=ibm-granite/granite-4.1-8b + GRANITE_QUANT=fp8. - ${GRANITE_MODEL} - --served-model-name - ${GRANITE_SERVED_NAME} - --quantization - ${GRANITE_QUANT} - --host - 0.0.0.0 - --port - "8000" - --gpu-memory-utilization - ${GRANITE_GPU_MEM_UTIL} - --max-model-len - ${GRANITE_MAX_MODEL_LEN} # Very high so the KV pool (not the seq cap) is the only concurrency bound — # granite is the fleet fan-out summarizer/classifier (many concurrent SHORT # calls). vLLM's default resolves to 128, capping below the KV bound # (~192 @ 1K-tok); 1024 unblocks it (VRAM-neutral — KV pool is util-bound). - --max-num-seqs - ${GRANITE_MAX_NUM_SEQS} - --dtype - auto # CUDA graphs ENABLED (no --enforce-eager) for decode throughput. Made # room 2026-06-05 by right-sizing the embed/rerank/reward trio's KV pools # (they were over-provisioned at 5.9x/2.0x/3.9x concurrency); GPU 1 now has # ~17 GB free after granite, so graph-capture buffers fit. If the trio # ever grows back, granite may need --enforce-eager again on this card. # FP8 KV cache — halves KV memory; near-lossless on Blackwell (cc 12.0). - --kv-cache-dtype - ${GRANITE_KV_CACHE_DTYPE} # Prefix caching pinned EXPLICIT (vLLM v1 defaults it on, but pin so a # version flip can't silently disable it). Benched 2026-06-13: ~6.5x faster # TTFT (45ms vs 292ms) on a shared ~4.5k-token summarizer template; soft/ # evictable KV, neutral when prefixes don't repeat — pure win for granite. - --enable-prefix-caching deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${GRANITE_GPU_ID}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 180s networks: - tnet labels: - homepage.group=AI - Inference - homepage.name=vLLM Granite 4.1 8B (summarizer) - homepage.icon=mdi-text-box-outline - homepage.description=Granite 4.1 8B FP8 via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${GRANITE_PORT}/docs vllm-coder: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-coder restart: unless-stopped ipc: host ports: - "${CODER_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} - VLLM_API_KEY=${API_KEY:-} command: # Qwen2.5-Coder-1.5B (BASE) — FIM code-completion seat for Zed edit-predictions # (deep-research pick 2026-07-27). Native fill-in-the-middle: <|fim_prefix|> / # <|fim_suffix|> / <|fim_middle|> (IDs 151659/151660/151661); Zed sends the # FIM-formatted prompt to /v1/completions and vLLM passes it through (the FIM # special tokens live in the tokenizer). BASE not -Instruct (FIM is a # pretraining objective; base completions are cleaner). Apache-2.0. Runner-up = # Qwen2.5-Coder-3B (higher HumanEval-FIM but non-commercial Qwen-Research license). - ${CODER_MODEL} - --served-model-name - ${CODER_SERVED_NAME} - --host - 0.0.0.0 - --port - "8000" - --gpu-memory-utilization - ${CODER_GPU_MEM_UTIL} - --max-model-len - ${CODER_MAX_MODEL_LEN} - --max-num-seqs - ${CODER_MAX_NUM_SEQS} - --dtype - auto - --kv-cache-dtype - ${CODER_KV_CACHE_DTYPE} - --enable-prefix-caching deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${CODER_GPU_ID}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 300s networks: - tnet labels: - homepage.group=AI - Inference - homepage.name=vLLM Qwen2.5-Coder 1.5B (FIM) - homepage.icon=mdi-code-braces - homepage.description=Qwen2.5-Coder-1.5B FIM code-completion (ana-ml2, Zed edit-predictions) - homepage.href=http://10.250.50.54:${CODER_PORT}/docs vllm-lfm25: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-lfm25 restart: unless-stopped ipc: host ports: - "${LFM25_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} - VLLM_API_KEY=${API_KEY:-} command: # LiquidAI/LFM2.5-2.6B (BF16, Lfm2ForCausalLM). NON-PRODUCTION bake-off alias # vs granite-4.1-8b on structured extraction / classification / tool-call # formatting (brokkr R-target, 2026-08-10). LFM Open License v1.0 (; deliberately NO --reasoning-parser, so the full # generation (thinking + answer) lands in `content` non-empty — brokkr's explicit # requirement (an empty content with the answer stranded in reasoning_content # reads as a degenerate model). Vendor sampling (temp 0.1 / top_k 50 / rep_pen 1.1) # is the LiteLLM alias default, not a launch arg. - ${LFM25_MODEL} - --served-model-name - ${LFM25_SERVED_NAME} - --host - 0.0.0.0 - --port - "8000" - --gpu-memory-utilization - ${LFM25_GPU_MEM_UTIL} - --max-model-len - ${LFM25_MAX_MODEL_LEN} - --max-num-seqs - ${LFM25_MAX_NUM_SEQS} - --dtype - auto - --kv-cache-dtype - ${LFM25_KV_CACHE_DTYPE} - --enable-prefix-caching # Split the reasoning (delimited by ; the opening tag is # injected by the chat template into the prompt) into reasoning_content, so # `content` is the clean post- answer — scoreable JSON for brokkr's # structured-extraction bake-off (raw-served, reasoning prepended, is not). - --reasoning-parser - deepseek_r1 deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${LFM25_GPU_ID}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 300s networks: - tnet labels: - homepage.group=AI - Inference - homepage.name=vLLM LFM2.5-2.6B (bake-off) - homepage.icon=mdi-flask-outline - homepage.description=LiquidAI LFM2.5-2.6B non-prod bake-off vs granite (ana-ml2) - homepage.href=http://10.250.50.54:${LFM25_PORT}/docs networks: tnet: name: traefik-net external: true