diff --git a/stacks/litellm/conf/config.yaml b/stacks/litellm/conf/config.yaml index 1b382ef..c5e0888 100644 --- a/stacks/litellm/conf/config.yaml +++ b/stacks/litellm/conf/config.yaml @@ -54,24 +54,12 @@ model_list: model_info: mode: chat - # ═══ DECKARD TRIAL (2026-07-01) — qwopus/`gen` DISPLACED on ana-ml2 GPU0:8013 ═══ - # robbatt/Qwen3.6-40B-Deckard-NVFP4 (served-name qwen3.6-40b-deckard) now occupies - # GPU0:8013; the qwopus container (vllm-qwopus35-122b) is STOPPED (not removed). - # Trial-by-fleet-traffic (operator 2026-07-01), may become permanent. - # REPOINTED to Deckard: the 5 role aliases summarizer-large / qwen-large[-reasoning] - # / gen[-reasoning] (model: field -> hosted_vllm/qwen3.6-40b-deckard; api_base :8013 - # unchanged; presence_penalty 1.0 + sampling CARRIED from qwopus, may retune). - # NOT repointed (auto-404 by design, no-false-alias): the true names - # qwen3.5-122-a10b[-reasoning] — Deckard doesn't serve that name -> clean "down". - # REVERT: docker rm -f vllm-deckard-40b; docker start vllm-qwopus35-122b; restore this - # file from config.yaml.bak-pre-deckard-20260701-001036; docker restart litellm. - # ───────────────────────────────────────────────────────────────────────────────── - # alias: summarizer-large -> DECKARD (was gen/qwen3.5-122-a10b). Heavier summarization. - # Thinking OFF. WARN semantic note: Deckard is a thinking/creative model, not a pure - # summarizer — acceptable for the trial; flag if summaries regress. + # alias: summarizer-large -> gen / qwen3.5-122-a10b (operator 2026-06-19). For heavier + # summarization that wants the 122B Qwopus instead of granite-8b. Thinking OFF (matches + # gen). Keep api_base (:8013) + enable_thinking in sync with the gen record below. - model_name: summarizer-large litellm_params: - model: hosted_vllm/qwen3.6-40b-deckard + model: hosted_vllm/qwen3.5-122-a10b api_base: http://10.250.50.54:8013/v1 api_key: os.environ/VLLM_API_KEY presence_penalty: 1.0 @@ -119,10 +107,8 @@ model_list: # --- Qwopus3.5-122B-A10B (Kimi-distilled, abliterated, NVFP4, VISION-INTACT) — the # general / `gen` model on ana-ml2 GPU 0. Replaced the bjk110 text-only qwen3.5-122b # 2026-06-19 (which had replaced mistral-small-4). Served on :8013 via vLLM as plain - # multimodal (no text-only patch), served-name qwen3.5-122-a10b. WARN DECKARD TRIAL - # (see banner above): qwopus is STOPPED — these two true-name records are left - # unchanged BY DESIGN and now 404 (Deckard serves qwen3.6-40b-deckard, not this - # name). Historical qwopus detail follows. Full 256K (262144) @ fp8 KV + CUDA graphs; + # multimodal (no text-only patch), served-name qwen3.5-122-a10b — so these records + # route UNCHANGED. Full 256K (262144) @ fp8 KV + CUDA graphs (92.7 tok/s warm); # tool-calling via qwen3_coder. Thinking split = chat_template_kwargs.enable_thinking # + --reasoning-parser qwen3. One upstream fanned out under qwen3.5-122-a10b[-reasoning] # + aliases qwen-large[-reasoning] + gen[-reasoning]; -reasoning variants enable @@ -164,7 +150,7 @@ model_list: mode: chat - model_name: qwen-large litellm_params: - model: hosted_vllm/qwen3.6-40b-deckard + model: hosted_vllm/qwen3.5-122-a10b api_base: http://10.250.50.54:8013/v1 api_key: os.environ/VLLM_API_KEY presence_penalty: 1.0 @@ -178,7 +164,7 @@ model_list: mode: chat - model_name: qwen-large-reasoning litellm_params: - model: hosted_vllm/qwen3.6-40b-deckard + model: hosted_vllm/qwen3.5-122-a10b api_base: http://10.250.50.54:8013/v1 api_key: os.environ/VLLM_API_KEY presence_penalty: 1.0 @@ -192,7 +178,7 @@ model_list: mode: chat - model_name: gen litellm_params: - model: hosted_vllm/qwen3.6-40b-deckard + model: hosted_vllm/qwen3.5-122-a10b api_base: http://10.250.50.54:8013/v1 api_key: os.environ/VLLM_API_KEY presence_penalty: 1.0 @@ -206,7 +192,7 @@ model_list: mode: chat - model_name: gen-reasoning litellm_params: - model: hosted_vllm/qwen3.6-40b-deckard + model: hosted_vllm/qwen3.5-122-a10b api_base: http://10.250.50.54:8013/v1 api_key: os.environ/VLLM_API_KEY presence_penalty: 1.0 @@ -219,26 +205,6 @@ model_list: model_info: mode: chat - # --- Qwen3.6-40B-Deckard (NVFP4) TRUE-NAME record — DECKARD TRIAL (2026-07-01). - # DavidAU 27B->40B "Deckard-Heretic" (Claude-4.6-Opus reasoning distill, uncensored), - # robbatt NVFP4 (llm-compressor). ana-ml2 GPU0:8013, stock vLLM 0.23.0, hybrid attn - # + multimodal + fp4_gemm all clean. Direct handle for Brokkr's A/B; params mirror - # `gen` (thinking off, pp 1.0 carried from qwopus). -reasoning folds into gen-reasoning. - - model_name: qwen3.6-40b-deckard - litellm_params: - model: hosted_vllm/qwen3.6-40b-deckard - api_base: http://10.250.50.54:8013/v1 - api_key: os.environ/VLLM_API_KEY - presence_penalty: 1.0 - temperature: 0.7 - top_p: 0.8 - extra_body: - top_k: 20 - chat_template_kwargs: - enable_thinking: false - model_info: - mode: chat - # --- Selene 1 Mini 8B (AtlaAI judge, FP8) — restored on GPU1 after the # llama-swap teardown (was the Q6_K GGUF in the swap zoo). vLLM dynamic fp8, # :8011. Explicit entry shadows the "*" wildcard (which used to reach it via