diff --git a/stacks/vllm/compose.yaml b/stacks/vllm/compose.yaml index 28ec332..1a0334d 100644 --- a/stacks/vllm/compose.yaml +++ b/stacks/vllm/compose.yaml @@ -210,7 +210,6 @@ services: - "${PHI4_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache - - /opt/docker/conf/vllm/phi4-chat-template.jinja:/config/phi4-chat-template.jinja:ro environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub @@ -238,10 +237,6 @@ services: # FP8 KV cache — halves KV memory at 128K ctx on Ada (cc 8.9); near-lossless. - --kv-cache-dtype - ${PHI4_KV_CACHE_DTYPE} - # Ollama-matching scaffold (drops the system-turn <|end|>) so vLLM reproduces - # brokkr's R15 canonical baseline. See conf/phi4-chat-template.jinja. - - --chat-template - - /config/phi4-chat-template.jinja deploy: resources: reservations: diff --git a/stacks/vllm/conf/phi4-chat-template.jinja b/stacks/vllm/conf/phi4-chat-template.jinja deleted file mode 100644 index 31b930b..0000000 --- a/stacks/vllm/conf/phi4-chat-template.jinja +++ /dev/null @@ -1,24 +0,0 @@ -{#- - Phi-4-mini chat template — OLLAMA-MATCHING variant. - - Why this exists: the official HF tokenizer template emits <|end|> after the - SYSTEM turn; Ollama's phi4 template does NOT (its system-block <|end|> is only - in the tools branch). That single boundary token regressed brokkr's R15 P02 - admission eval on vLLM vs the Ollama-measured canonical (type macro-F1 -33pp) - while valid_format held at 1.0. Operator chose (2026-06-04) to make vLLM match - Ollama's leaner scaffold globally so baseline == production. - - Renders (system + user, add_generation_prompt): - <|system|>{sys}<|user|>{usr}<|end|><|assistant|> - i.e. NO <|end|> after the system turn (the only delta from official). --#} -{%- for message in messages -%} -{%- if message['role'] == 'system' -%} -{{- '<|system|>' + message['content'] -}} -{%- else -%} -{{- '<|' + message['role'] + '|>' + message['content'] + '<|end|>' -}} -{%- endif -%} -{%- endfor -%} -{%- if add_generation_prompt -%} -{{- '<|assistant|>' -}} -{%- endif -%}