# llama-swap — GGUF model server with on-demand model swapping. # # Proxies OpenAI-compatible API requests to llama.cpp server instances # and swaps which model is loaded into VRAM per request. Runs on # ana-ml2 using both GPUs dynamically (no explicit device pinning — # llama-swap picks per-model-definition). # # Model definitions live in /opt/docker/conf/llama-swap/config.yaml on # the server. Canonical copy of that config is config.yaml in this # workspace; deploy with scp + `docker compose restart` or the script # at the bottom of README.md. # # All tunables live in .env — edit that, not this file. services: llama-swap: image: ghcr.io/mostlygeek/llama-swap:${LLAMA_SWAP_VERSION} container_name: llama-swap restart: unless-stopped stdin_open: true tty: true runtime: nvidia ports: - "${LLAMA_SWAP_PORT}:8080" volumes: - /opt/docker/conf/llama-swap/config.yaml:/app/config.yaml - ${MODELS_DIR}:/models - ${HF_CACHE_DIR}:/hfcache environment: - HF_HOME=/hfcache - HF_HUB_CACHE=/hfcache/hub healthcheck: test: ["CMD-SHELL", "curl -fsS http://localhost:8080/ >/dev/null || exit 1"] interval: 30s timeout: 10s retries: 3 start_period: 30s networks: - tnet labels: - homepage.group=AI Systems - homepage.name=llama-swap - homepage.icon=mdi-swap-horizontal - homepage.description=GGUF model swapper (llama.cpp; ana-ml2) - homepage.href=http://10.250.50.54:${LLAMA_SWAP_PORT} networks: tnet: name: traefik-net external: true