# llama-swap — GGUF model server with on-demand model swapping. # # Proxies OpenAI-compatible API requests to llama.cpp server instances # and swaps which model is loaded into VRAM per request. Runs on # fv-ml1 using both GPUs dynamically (no explicit device pinning — # llama-swap picks per-model-definition). # # Model definitions live in /opt/docker/conf/llama-swap/config.yaml on # the server. Canonical copy of that config is config.yaml in this # workspace; deploy with scp + `docker compose restart` or the script # at the bottom of README.md. # # All tunables live in .env — edit that, not this file. services: llama-swap: image: ghcr.io/mostlygeek/llama-swap:${LLAMA_SWAP_VERSION} container_name: llama-swap restart: unless-stopped stdin_open: true tty: true runtime: nvidia ports: - "${LLAMA_SWAP_PORT}:8080" volumes: - /opt/docker/conf/llama-swap/config.yaml:/app/config.yaml - ${MODELS_DIR}:/models - ${HF_CACHE_DIR}:/hfcache environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub # Pin to GPU 0 — the reserved card for on-demand large-model hot-loads. # The always-on vLLM services (granite + embed/rerank/reward) own GPU 1; # keeping llama-swap off GPU 1 stops a hot-loaded model from contending # with them. llama.cpp then sees only GPU 0 (cuda:0), so --n-gpu-layers # 999 loads there with no per-model device targeting needed. - NVIDIA_VISIBLE_DEVICES=${LLAMA_SWAP_GPU:-0} healthcheck: test: ["CMD-SHELL", "curl -fsS http://localhost:8080/ >/dev/null || exit 1"] interval: 30s timeout: 10s retries: 3 start_period: 30s networks: - tnet labels: - homepage.group=AI Systems - homepage.name=llama-swap - homepage.icon=mdi-swap-horizontal - homepage.description=GGUF model swapper (llama.cpp; fv-ml1) - homepage.href=http://10.251.50.54:${LLAMA_SWAP_PORT} networks: tnet: name: traefik-net external: true