# vLLM — Qwen3 Embedding + Reranker + Skywork Reward-V2 classifier. # # Originally created to replace the unmaintained Infinity stack (embed + # rerank); generalized 2026-05-13 to host any vLLM-served model on ana-ml2, # starting with the Skywork-Reward-V2-Llama-3.1-8B reward classifier # (AWQ-quantized locally, served from /tank/aimodels/llm/). # # vLLM runs one model per process, so this stack brings up three containers # sharing a single GPU: # # vllm-embed — Qwen3-Embedding served as an OpenAI /v1/embeddings server # vllm-rerank — Qwen3-Reranker served as a /rerank + /score server # vllm-reward — Skywork-Reward-V2-Llama-3.1-8B-AWQ served as a /classify scorer # # The reranker is a causal-LM checkpoint; --hf-overrides re-maps it to # Qwen3ForSequenceClassification so vLLM's reranking endpoints work and the # model only emits two class logits (no/yes) instead of the full 151k vocab. # # All tunables live in .env — edit that, not this file. # # Pre-download models to avoid first-run delay: # scripts/elway ana-ml2 --playbook playbooks/pull-hf-repo.yaml \ # --var hf_repo=Qwen/Qwen3-Embedding-0.6B # scripts/elway ana-ml2 --playbook playbooks/pull-hf-repo.yaml \ # --var hf_repo=Qwen/Qwen3-Reranker-0.6B # # Skywork-Reward-V2-Llama-3.1-8B-AWQ is a locally-quantized model — lives at # /tank/aimodels/llm/Skywork-Reward-V2-Llama-3.1-8B-AWQ on ana-ml2 and is # bind-mounted into the reward service at /local-models. Not from HF Hub. services: vllm-embed: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-embed restart: unless-stopped ipc: host ports: - "${EMBED_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} - VLLM_API_KEY=${API_KEY:-} command: - ${EMBED_MODEL} - --served-model-name - ${EMBED_MODEL} - --runner - pooling - --host - 0.0.0.0 - --port - "8000" - --gpu-memory-utilization - ${EMBED_GPU_MEM_UTIL} - --max-model-len - ${EMBED_MAX_MODEL_LEN} - --dtype - auto deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${GPU_ID}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 180s networks: - tnet labels: - homepage.group=AI Systems - homepage.name=vLLM Embed (Qwen3) - homepage.icon=mdi-vector-arrange-below - homepage.description=Qwen3 Embedding via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${EMBED_PORT}/docs vllm-rerank: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-rerank restart: unless-stopped ipc: host ports: - "${RERANK_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} - VLLM_API_KEY=${API_KEY:-} command: - ${RERANK_MODEL} - --served-model-name - ${RERANK_MODEL} - --runner - pooling - --hf-overrides - '{"architectures":["Qwen3ForSequenceClassification"],"classifier_from_token":["no","yes"],"is_original_qwen3_reranker":true}' - --host - 0.0.0.0 - --port - "8000" - --gpu-memory-utilization - ${RERANK_GPU_MEM_UTIL} - --max-model-len - ${RERANK_MAX_MODEL_LEN} - --dtype - auto deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${GPU_ID}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 180s networks: - tnet labels: - homepage.group=AI Systems - homepage.name=vLLM Rerank (Qwen3) - homepage.icon=mdi-sort-variant - homepage.description=Qwen3 Reranker via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${RERANK_PORT}/docs vllm-reward: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-reward restart: unless-stopped ipc: host ports: - "${REWARD_PORT}:8000" volumes: # AWQ output lives in the legacy llama-swap models tree, not the HF cache # — bind-mount the LLM models dir read-only so the reward service can # load it as a local-path HF-format model. - /tank/aimodels/llm:/local-models:ro environment: - VLLM_API_KEY=${API_KEY:-} command: - /local-models/Skywork-Reward-V2-Llama-3.1-8B-AWQ - --served-model-name - Skywork/Skywork-Reward-V2-Llama-3.1-8B-AWQ # vLLM 0.19.1 deprecated --task in favor of --runner. The model's # config.json declares `LlamaForSequenceClassification` so the # pooling runner uses it as a classifier (single-label reward score) # without needing an explicit task flag. - --runner - pooling - --host - 0.0.0.0 - --port - "8000" - --gpu-memory-utilization - ${REWARD_GPU_MEM_UTIL} - --max-model-len - ${REWARD_MAX_MODEL_LEN} - --dtype - auto deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${GPU_ID}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 240s networks: - tnet labels: - homepage.group=AI Systems - homepage.name=vLLM Reward (Skywork) - homepage.icon=mdi-scale-balance - homepage.description=Skywork-Reward-V2 8B classifier via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${REWARD_PORT}/docs # Phi-4-mini (FP8) — summarizer + "dreaming" agent. Supersedes the # llama-swap granite-4-small pin. Generative chat model (OpenAI # /v1/chat/completions), so NO --runner pooling. FP8 on RTX 6000 Ada # (cc 8.9): near-lossless, ~1.2x, ~6 GB. vllm-granite: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-granite restart: unless-stopped ipc: host ports: - "${GRANITE_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} - VLLM_API_KEY=${API_KEY:-} command: # Production summarizer (replaced phi4-mini 2026-06-05). Default = official # IBM pre-quantized FP8 (compressed-tensors), loaded directly; FP8 is native # on the RTX 6000 Ada (cc 8.9). Fallback to vLLM-native dynamic FP8 from # BF16: GRANITE_MODEL=ibm-granite/granite-4.1-8b + GRANITE_QUANT=fp8. - ${GRANITE_MODEL} - --served-model-name - ${GRANITE_SERVED_NAME} - --quantization - ${GRANITE_QUANT} - --host - 0.0.0.0 - --port - "8000" - --gpu-memory-utilization - ${GRANITE_GPU_MEM_UTIL} - --max-model-len - ${GRANITE_MAX_MODEL_LEN} - --dtype - auto # CUDA graphs ENABLED (no --enforce-eager) for decode throughput. Made # room 2026-06-05 by right-sizing the embed/rerank/reward trio's KV pools # (they were over-provisioned at 5.9x/2.0x/3.9x concurrency); GPU 1 now has # ~17 GB free after granite, so graph-capture buffers fit. If the trio # ever grows back, granite may need --enforce-eager again on this card. # FP8 KV cache — halves KV memory; near-lossless on Ada (cc 8.9). - --kv-cache-dtype - ${GRANITE_KV_CACHE_DTYPE} deploy: resources: reservations: devices: - driver: nvidia device_ids: - "${GRANITE_GPU_ID}" capabilities: - gpu healthcheck: test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 start_period: 180s networks: - tnet labels: - homepage.group=AI Systems - homepage.name=vLLM Granite 4.1 8B (summarizer) - homepage.icon=mdi-text-box-outline - homepage.description=Granite 4.1 8B FP8 via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${GRANITE_PORT}/docs networks: tnet: name: traefik-net external: true