homepage-regroup, mog-sec-move-to-gpu0, pull-hf-repo and serve-qwen3.5-122b all carried runnable 'scripts/elway ana-ml2 --playbook ...' instructions or the old 10.250.50.54 address. Each would fail today against a dead name and a dead IP, so these are corrections rather than cosmetics. homepage-regroup is renamed to match; the other three keep their names, which never carried the host.
73 lines
2.6 KiB
YAML
73 lines
2.6 KiB
YAML
# Displace mistral-small-4 (heretic) on fv-ml1 GPU 0 and serve
|
|
# bjk110/Qwen3.5-122B-A10B-abliterated-NVFP4 (text-only) as the new `gen` model
|
|
# (operator 2026-06-19). Weights pre-staged at /tank/aimodels/qwen3.5-122b-a10b-nvfp4
|
|
# (incl. the repo's serving/entrypoint.sh + vllm_patches/ that the compose mounts).
|
|
#
|
|
# ⚠️ Downing mistral-small-4 takes down the Worldtree CHARACTER backend (vision-intact)
|
|
# until it's repointed — operator-acknowledged. REVERT = down qwen, up -d the heretic.
|
|
#
|
|
# scripts/elway fv-ml1 --playbook playbooks/serve-qwen3.5-122b.yaml
|
|
|
|
vars:
|
|
compose_dir: /opt/docker/compose/qwen3.5-122b
|
|
heretic_dir: /opt/docker/compose/mistral-small-4-heretic
|
|
model_dir: /tank/aimodels/qwen3.5-122b-a10b-nvfp4
|
|
host_port: "8013"
|
|
|
|
steps:
|
|
- name: Verify NVFP4 weights + the repo's patch/entrypoint are staged
|
|
shell: |
|
|
test -f {{ model_dir }}/model.safetensors.index.json \
|
|
&& test -f {{ model_dir }}/serving/entrypoint.sh \
|
|
&& test -f {{ model_dir }}/vllm_patches/patch_qwen35_moe_text.py
|
|
changed_when: "false"
|
|
|
|
- name: Ensure vLLM compile-cache dir exists (writable)
|
|
shell: mkdir -p {{ model_dir }}/.cache/vllm
|
|
creates: "{{ model_dir }}/.cache/vllm"
|
|
|
|
- name: Ensure compose dir exists
|
|
shell: mkdir -p {{ compose_dir }}
|
|
creates: "{{ compose_dir }}"
|
|
|
|
- name: Upload compose.yaml
|
|
upload:
|
|
src: stacks/qwen3.5-122b/compose.yaml
|
|
dest: "{{ compose_dir }}/compose.yaml"
|
|
mode: "0644"
|
|
|
|
- name: Seed .env from template (only if absent)
|
|
upload:
|
|
src: stacks/qwen3.5-122b/.env.example
|
|
dest: "{{ compose_dir }}/.env"
|
|
mode: "0644"
|
|
when: "[ ! -f {{ compose_dir }}/.env ]"
|
|
|
|
- name: Displace — down mistral-small-4-heretic (frees GPU 0; no-op if down)
|
|
shell: cd {{ heretic_dir }} && docker compose down
|
|
|
|
- name: Bring up qwen3.5-122b
|
|
shell: cd {{ compose_dir }} && docker compose up -d
|
|
|
|
- name: Wait for vLLM /health (allow ~15 min for patch + NVFP4 MoE load + warmup)
|
|
shell: |
|
|
for i in $(seq 1 180); do
|
|
curl -sf -o /dev/null --max-time 3 http://localhost:{{ host_port }}/health && exit 0
|
|
sleep 5
|
|
done
|
|
exit 1
|
|
changed_when: "false"
|
|
|
|
verify:
|
|
- name: /health returns 200
|
|
shell: curl -sf -o /dev/null http://localhost:{{ host_port }}/health
|
|
changed_when: "false"
|
|
|
|
- name: served model id is qwen3.5-122-a10b
|
|
shell: curl -sf http://localhost:{{ host_port }}/v1/models | grep -q qwen3.5-122-a10b
|
|
changed_when: "false"
|
|
|
|
- name: container running
|
|
shell: docker inspect vllm-qwen35-122b --format '{{.State.Status}}' | grep -q running
|
|
changed_when: "false"
|