diff --git a/CLAUDE.md b/CLAUDE.md index a54970e..f05bcc0 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -88,7 +88,7 @@ Observed and standardized across servers: | Name | IP | Site | Role | Details | |------|-----|------|------|---------| -| ana-ml2 | 10.250.50.54 | Anaheim (`10.250.0.0/16`) | GPU / AI inference (bare metal, dual RTX 6000 Ada) | `servers/ana-ml2/README.md` | +| ana-ml2 | 10.250.50.54 | Anaheim (`10.250.0.0/16`) | GPU / AI inference (bare metal, dual RTX PRO 6000 Blackwell Max-Q, 96 GB each) | `servers/ana-ml2/README.md` | | irv-ml1 | 10.100.79.3 (WG) | Irvine — reachable only via WireGuard tunnel from NH3 | GPU / AI inference (bare metal, RTX 3090 + RTX A6000, native stacks) | `servers/irv-ml1/README.md` | | ana-docker | 10.250.50.70 | Anaheim | General-purpose Docker host (non-GPU VM on pfi-pve) | `servers/ana-docker/README.md` | | pfi-ana-webhost | 10.250.50.52 | Anaheim | VM on pfi-pve (VMID 110) — web workload | `servers/pfi-ana-webhost/README.md` | diff --git a/servers/ana-ml2/README.md b/servers/ana-ml2/README.md index 2aa1b1b..a384684 100644 --- a/servers/ana-ml2/README.md +++ b/servers/ana-ml2/README.md @@ -15,7 +15,7 @@ Primary AI inference host for PFI. NOT Dell / not the same box as sf-r630 / sfsrv-ana) - **CPU:** AMD EPYC 9254 24-core (96 threads) - **RAM:** 566 GB -- **GPUs:** 2x NVIDIA RTX 6000 Ada Generation (46 GB VRAM each, GPU 0 and GPU 1) +- **GPUs:** 2x NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition (96 GB VRAM each, cc 12.0 / sm_120, GPU 0 and GPU 1) — upgraded 2026-06 from 2x RTX 6000 Ada (46 GB, cc 8.9). Blackwell adds native FP4 (NVFP4) tensor cores and doubles VRAM. - **Storage:** ZFS `zroot` (434 GB root) + `tank` pool (8.6 TB at `/tank`) - **OS:** Debian 13 (trixie), kernel 6.12.x - **Docker:** 29.3.1, runtimes: runc (default), nvidia, io.containerd.runc.v2 diff --git a/servers/ana-ml2/system-details.txt b/servers/ana-ml2/system-details.txt index d43a031..e0bff6a 100644 --- a/servers/ana-ml2/system-details.txt +++ b/servers/ana-ml2/system-details.txt @@ -2,10 +2,10 @@ ===== HOST ===== Hostname: ana-ml2 -Date: 2026-04-19T22:15:48-07:00 -Uptime: up 33 weeks, 1 day, 14 minutes +Date: 2026-06-13T13:35:42-07:00 +Uptime: up 1 day, 6 minutes OS: Debian GNU/Linux 13 (trixie) -Kernel: 6.12.41+deb13-amd64 +Kernel: 6.12.74+deb13+1-amd64 Arch: x86_64 ===== HARDWARE ===== @@ -13,22 +13,22 @@ Arch: x86_64 CPU cores: 96 CPU model: AMD EPYC 9254 24-Core Processor MemTotal: 566.6 GB -MemAvailable: 500.9 GB +MemAvailable: 484.5 GB ===== GPUS ===== index, name, memory.total [MiB], memory.free [MiB], driver_version -0, NVIDIA RTX 6000 Ada Generation, 46068 MiB, 45456 MiB, 580.65.06 -1, NVIDIA RTX 6000 Ada Generation, 46068 MiB, 34072 MiB, 580.65.06 +0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 97887 MiB, 97247 MiB, 580.65.06 +1, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 97887 MiB, 3692 MiB, 580.65.06 ===== FILESYSTEMS (df) ===== Filesystem Size Used Avail Use% Mounted on -zroot/ROOT/debian 434G 169G 265G 39% / -efivarfs 128K 58K 66K 47% /sys/firmware/efi/efivars -zroot/home 280G 15G 265G 6% /home -/dev/sda1 511M 92M 420M 18% /boot/efi -tank 8.6T 1.3T 7.3T 16% /tank +zroot/ROOT/debian 394G 212G 182G 54% / +efivarfs 128K 67K 57K 55% /sys/firmware/efi/efivars +/dev/sdb1 511M 92M 420M 18% /boot/efi +tank 8.6T 1.6T 7.1T 18% /tank +zroot/home 244G 62G 182G 26% /home ===== PERSISTENT MOUNTS (/etc/fstab, non-comment) ===== @@ -36,11 +36,11 @@ UUID="3D9B-8E0C" /boot/efi vfat defaults 0 0 ===== TARGETED DATA PATHS ===== -/tank (total: 1.3T) +/tank (total: 1.6T) total 15 drwxrwxrwx 6 root root 6 2026-04-11 23:17 . drwxr-xr-x 18 root root 26 2026-03-29 16:08 .. - drwxrwxr-x 7 llmuser llm 8 2026-04-17 14:09 aimodels + drwxrwxr-x 9 llmuser llm 10 2026-06-12 18:01 aimodels drwxrwxr-x 3 llmuser llm 3 2025-09-08 13:21 comfy drwxrwxr-x 4 llmuser llmuser 4 2026-04-11 23:17 kokoro drwxrwxr-x 5 lkraven lkraven 6 2025-09-10 18:35 vibevoice @@ -65,29 +65,30 @@ UUID="3D9B-8E0C" /boot/efi vfat defaults 0 0 total 18 drwxrwxr-x 4 lkraven lkraven 4 2025-09-03 20:19 . drwxrwxrwx 13 root root 13 2026-04-17 23:00 .. - drwxrwxr-x 12 llmuser llm 12 2026-04-19 00:58 compose - drwxrwxr-x 3 llmuser llmuser 3 2026-04-18 22:40 conf + drwxrwxr-x 12 lkraven lkraven 12 2026-06-13 02:32 compose + drwxrwxr-x 4 lkraven lkraven 4 2026-06-04 00:26 conf /opt/docker/compose (total: 110M) - total 14 - drwxrwxr-x 12 llmuser llm 12 2026-04-19 00:58 . + total 30 + drwxrwxr-x 12 lkraven lkraven 12 2026-06-13 02:32 . drwxrwxr-x 4 lkraven lkraven 4 2025-09-03 20:19 .. - drwxr-xr-x 2 root root 4 2026-04-19 00:58 beszel-agent - drwxr-xr-x 2 llmuser llm 4 2025-09-05 20:36 comfyui - drwxrwxr-x 2 llmuser llm 3 2025-09-03 17:23 dockge - drwxr-xr-x 2 root root 4 2026-04-19 00:28 dozzle-agent - drwxrwxr-x 3 llmuser llmuser 4 2026-04-11 23:17 kokoro - drwxr-xr-x 2 llmuser llm 4 2025-09-03 17:34 llama-swap - drwxr-xr-x 2 llmuser llm 4 2025-09-24 12:32 parakeet - drwxr-xr-x 2 llmuser llmuser 2 2026-04-11 11:08 synapse - drwxr-xr-x 2 llmuser llm 4 2025-09-10 18:03 vibevoice - drwxr-xr-x 2 root root 4 2026-04-18 22:59 vllm-qwen3 + drwxr-xr-x 2 lkraven lkraven 4 2026-04-20 18:47 beszel-agent-ana + drwxr-xr-x 2 lkraven lkraven 4 2025-09-05 20:36 comfyui + drwxr-xr-x 2 lkraven lkraven 4 2026-04-20 20:47 dockge + drwxr-xr-x 2 lkraven lkraven 4 2026-04-20 20:44 dozzle-agent-ana + drwxrwxr-x 3 lkraven lkraven 4 2026-04-11 23:17 kokoro + drwxr-xr-x 2 lkraven lkraven 6 2026-06-12 15:17 llama-swap + drwxr-xr-x 2 lkraven lkraven 4 2025-09-24 12:32 parakeet + drwxrwxr-x 2 lkraven lkraven 4 2026-06-13 08:34 qwen35-vl + drwxr-xr-x 2 lkraven lkraven 4 2025-09-10 18:03 vibevoice + drwxr-xr-x 2 lkraven lkraven 9 2026-06-13 08:42 vllm -/opt/docker/conf (total: 18K) +/opt/docker/conf (total: 14K) total 2 - drwxrwxr-x 3 llmuser llmuser 3 2026-04-18 22:40 . + drwxrwxr-x 4 lkraven lkraven 4 2026-06-04 00:26 . drwxrwxr-x 4 lkraven lkraven 4 2025-09-03 20:19 .. - drwxrwxr-x 2 llmuser llmuser 4 2026-04-19 17:00 llama-swap + drwxr-xr-x 2 lkraven lkraven 3 2026-06-05 00:22 llama-swap + drwxr-xr-x 2 lkraven lkraven 2 2026-06-04 00:37 vllm /var/lib/docker (total: 8.5K) @@ -102,8 +103,8 @@ UUID="3D9B-8E0C" /boot/efi vfat defaults 0 0 Server: 29.3.1 Client: 29.3.1 ----- docker info ----- -Containers: 6 (running 6, paused 0, stopped 0) -Images: 39 +Containers: 9 (running 9, paused 0, stopped 0) +Images: 42 Runtimes: map[io.containerd.runc.v2:{{runc [] map[]} map[org.opencontainers.runtime-spec.features:{"ociVersionMin":"1.0.0","ociVersionMax":"1.2.1","hooks":["prestart","createRuntime","createContainer","startContainer","poststart","poststop"],"mountOptions":["async","atime","bind","defaults","dev","diratime","dirsync","exec","iversion","lazytime","loud","mand","noatime","nodev","nodiratime","noexec","noiversion","nolazytime","nomand","norelatime","nostrictatime","nosuid","nosymfollow","private","ratime","rbind","rdev","rdiratime","relatime","remount","rexec","rnoatime","rnodev","rnodiratime","rnoexec","rnorelatime","rnostrictatime","rnosuid","rnosymfollow","ro","rprivate","rrelatime","rro","rrw","rshared","rslave","rstrictatime","rsuid","rsymfollow","runbindable","rw","shared","silent","slave","strictatime","suid","symfollow","sync","tmpcopyup","unbindable"],"linux":{"namespaces":["cgroup","ipc","mount","network","pid","time","user","uts"],"capabilities":["CAP_CHOWN","CAP_DAC_OVERRIDE","CAP_DAC_READ_SEARCH","CAP_FOWNER","CAP_FSETID","CAP_KILL","CAP_SETGID","CAP_SETUID","CAP_SETPCAP","CAP_LINUX_IMMUTABLE","CAP_NET_BIND_SERVICE","CAP_NET_BROADCAST","CAP_NET_ADMIN","CAP_NET_RAW","CAP_IPC_LOCK","CAP_IPC_OWNER","CAP_SYS_MODULE","CAP_SYS_RAWIO","CAP_SYS_CHROOT","CAP_SYS_PTRACE","CAP_SYS_PACCT","CAP_SYS_ADMIN","CAP_SYS_BOOT","CAP_SYS_NICE","CAP_SYS_RESOURCE","CAP_SYS_TIME","CAP_SYS_TTY_CONFIG","CAP_MKNOD","CAP_LEASE","CAP_AUDIT_WRITE","CAP_AUDIT_CONTROL","CAP_SETFCAP","CAP_MAC_OVERRIDE","CAP_MAC_ADMIN","CAP_SYSLOG","CAP_WAKE_ALARM","CAP_BLOCK_SUSPEND","CAP_AUDIT_READ","CAP_PERFMON","CAP_BPF","CAP_CHECKPOINT_RESTORE"],"cgroup":{"v1":true,"v2":true,"systemd":true,"systemdUser":true,"rdma":true},"seccomp":{"enabled":true,"actions":["SCMP_ACT_ALLOW","SCMP_ACT_ERRNO","SCMP_ACT_KILL","SCMP_ACT_KILL_PROCESS","SCMP_ACT_KILL_THREAD","SCMP_ACT_LOG","SCMP_ACT_NOTIFY","SCMP_ACT_TRACE","SCMP_ACT_TRAP"],"operators":["SCMP_CMP_EQ","SCMP_CMP_GE","SCMP_CMP_GT","SCMP_CMP_LE","SCMP_CMP_LT","SCMP_CMP_MASKED_EQ","SCMP_CMP_NE"],"archs":["SCMP_ARCH_AARCH64","SCMP_ARCH_ARM","SCMP_ARCH_MIPS","SCMP_ARCH_MIPS64","SCMP_ARCH_MIPS64N32","SCMP_ARCH_MIPSEL","SCMP_ARCH_MIPSEL64","SCMP_ARCH_MIPSEL64N32","SCMP_ARCH_PPC","SCMP_ARCH_PPC64","SCMP_ARCH_PPC64LE","SCMP_ARCH_RISCV64","SCMP_ARCH_S390","SCMP_ARCH_S390X","SCMP_ARCH_X32","SCMP_ARCH_X86","SCMP_ARCH_X86_64"],"knownFlags":["SECCOMP_FILTER_FLAG_TSYNC","SECCOMP_FILTER_FLAG_SPEC_ALLOW","SECCOMP_FILTER_FLAG_LOG"],"supportedFlags":["SECCOMP_FILTER_FLAG_TSYNC","SECCOMP_FILTER_FLAG_SPEC_ALLOW","SECCOMP_FILTER_FLAG_LOG"]},"apparmor":{"enabled":true},"selinux":{"enabled":true},"intelRdt":{"enabled":true},"mountExtensions":{"idmap":{"enabled":true}}},"annotations":{"io.github.seccomp.libseccomp.version":"2.6.0","org.opencontainers.runc.checkpoint.enabled":"true","org.opencontainers.runc.commit":"v1.3.4-0-gd6d73eb8","org.opencontainers.runc.version":"1.3.4\n"},"potentiallyUnsafeConfigAnnotations":["bundle","org.systemd.property.","org.criu.config"]}]} nvidia:{{nvidia-container-runtime [] map[]} map[org.opencontainers.runtime-spec.features:{"ociVersionMin":"1.0.0","ociVersionMax":"1.2.1","hooks":["prestart","createRuntime","createContainer","startContainer","poststart","poststop"],"mountOptions":["async","atime","bind","defaults","dev","diratime","dirsync","exec","iversion","lazytime","loud","mand","noatime","nodev","nodiratime","noexec","noiversion","nolazytime","nomand","norelatime","nostrictatime","nosuid","nosymfollow","private","ratime","rbind","rdev","rdiratime","relatime","remount","rexec","rnoatime","rnodev","rnodiratime","rnoexec","rnorelatime","rnostrictatime","rnosuid","rnosymfollow","ro","rprivate","rrelatime","rro","rrw","rshared","rslave","rstrictatime","rsuid","rsymfollow","runbindable","rw","shared","silent","slave","strictatime","suid","symfollow","sync","tmpcopyup","unbindable"],"linux":{"namespaces":["cgroup","ipc","mount","network","pid","time","user","uts"],"capabilities":["CAP_CHOWN","CAP_DAC_OVERRIDE","CAP_DAC_READ_SEARCH","CAP_FOWNER","CAP_FSETID","CAP_KILL","CAP_SETGID","CAP_SETUID","CAP_SETPCAP","CAP_LINUX_IMMUTABLE","CAP_NET_BIND_SERVICE","CAP_NET_BROADCAST","CAP_NET_ADMIN","CAP_NET_RAW","CAP_IPC_LOCK","CAP_IPC_OWNER","CAP_SYS_MODULE","CAP_SYS_RAWIO","CAP_SYS_CHROOT","CAP_SYS_PTRACE","CAP_SYS_PACCT","CAP_SYS_ADMIN","CAP_SYS_BOOT","CAP_SYS_NICE","CAP_SYS_RESOURCE","CAP_SYS_TIME","CAP_SYS_TTY_CONFIG","CAP_MKNOD","CAP_LEASE","CAP_AUDIT_WRITE","CAP_AUDIT_CONTROL","CAP_SETFCAP","CAP_MAC_OVERRIDE","CAP_MAC_ADMIN","CAP_SYSLOG","CAP_WAKE_ALARM","CAP_BLOCK_SUSPEND","CAP_AUDIT_READ","CAP_PERFMON","CAP_BPF","CAP_CHECKPOINT_RESTORE"],"cgroup":{"v1":true,"v2":true,"systemd":true,"systemdUser":true,"rdma":true},"seccomp":{"enabled":true,"actions":["SCMP_ACT_ALLOW","SCMP_ACT_ERRNO","SCMP_ACT_KILL","SCMP_ACT_KILL_PROCESS","SCMP_ACT_KILL_THREAD","SCMP_ACT_LOG","SCMP_ACT_NOTIFY","SCMP_ACT_TRACE","SCMP_ACT_TRAP"],"operators":["SCMP_CMP_EQ","SCMP_CMP_GE","SCMP_CMP_GT","SCMP_CMP_LE","SCMP_CMP_LT","SCMP_CMP_MASKED_EQ","SCMP_CMP_NE"],"archs":["SCMP_ARCH_AARCH64","SCMP_ARCH_ARM","SCMP_ARCH_MIPS","SCMP_ARCH_MIPS64","SCMP_ARCH_MIPS64N32","SCMP_ARCH_MIPSEL","SCMP_ARCH_MIPSEL64","SCMP_ARCH_MIPSEL64N32","SCMP_ARCH_PPC","SCMP_ARCH_PPC64","SCMP_ARCH_PPC64LE","SCMP_ARCH_RISCV64","SCMP_ARCH_S390","SCMP_ARCH_S390X","SCMP_ARCH_X32","SCMP_ARCH_X86","SCMP_ARCH_X86_64"],"knownFlags":["SECCOMP_FILTER_FLAG_TSYNC","SECCOMP_FILTER_FLAG_SPEC_ALLOW","SECCOMP_FILTER_FLAG_LOG"],"supportedFlags":["SECCOMP_FILTER_FLAG_TSYNC","SECCOMP_FILTER_FLAG_SPEC_ALLOW","SECCOMP_FILTER_FLAG_LOG"]},"apparmor":{"enabled":true},"selinux":{"enabled":true},"intelRdt":{"enabled":true},"mountExtensions":{"idmap":{"enabled":true}}},"annotations":{"io.github.seccomp.libseccomp.version":"2.6.0","org.opencontainers.runc.checkpoint.enabled":"true","org.opencontainers.runc.commit":"v1.3.4-0-gd6d73eb8","org.opencontainers.runc.version":"1.3.4\n"},"potentiallyUnsafeConfigAnnotations":["bundle","org.systemd.property.","org.criu.config"]}]} runc:{{runc [] map[]} map[org.opencontainers.runtime-spec.features:{"ociVersionMin":"1.0.0","ociVersionMax":"1.2.1","hooks":["prestart","createRuntime","createContainer","startContainer","poststart","poststop"],"mountOptions":["async","atime","bind","defaults","dev","diratime","dirsync","exec","iversion","lazytime","loud","mand","noatime","nodev","nodiratime","noexec","noiversion","nolazytime","nomand","norelatime","nostrictatime","nosuid","nosymfollow","private","ratime","rbind","rdev","rdiratime","relatime","remount","rexec","rnoatime","rnodev","rnodiratime","rnoexec","rnorelatime","rnostrictatime","rnosuid","rnosymfollow","ro","rprivate","rrelatime","rro","rrw","rshared","rslave","rstrictatime","rsuid","rsymfollow","runbindable","rw","shared","silent","slave","strictatime","suid","symfollow","sync","tmpcopyup","unbindable"],"linux":{"namespaces":["cgroup","ipc","mount","network","pid","time","user","uts"],"capabilities":["CAP_CHOWN","CAP_DAC_OVERRIDE","CAP_DAC_READ_SEARCH","CAP_FOWNER","CAP_FSETID","CAP_KILL","CAP_SETGID","CAP_SETUID","CAP_SETPCAP","CAP_LINUX_IMMUTABLE","CAP_NET_BIND_SERVICE","CAP_NET_BROADCAST","CAP_NET_ADMIN","CAP_NET_RAW","CAP_IPC_LOCK","CAP_IPC_OWNER","CAP_SYS_MODULE","CAP_SYS_RAWIO","CAP_SYS_CHROOT","CAP_SYS_PTRACE","CAP_SYS_PACCT","CAP_SYS_ADMIN","CAP_SYS_BOOT","CAP_SYS_NICE","CAP_SYS_RESOURCE","CAP_SYS_TIME","CAP_SYS_TTY_CONFIG","CAP_MKNOD","CAP_LEASE","CAP_AUDIT_WRITE","CAP_AUDIT_CONTROL","CAP_SETFCAP","CAP_MAC_OVERRIDE","CAP_MAC_ADMIN","CAP_SYSLOG","CAP_WAKE_ALARM","CAP_BLOCK_SUSPEND","CAP_AUDIT_READ","CAP_PERFMON","CAP_BPF","CAP_CHECKPOINT_RESTORE"],"cgroup":{"v1":true,"v2":true,"systemd":true,"systemdUser":true,"rdma":true},"seccomp":{"enabled":true,"actions":["SCMP_ACT_ALLOW","SCMP_ACT_ERRNO","SCMP_ACT_KILL","SCMP_ACT_KILL_PROCESS","SCMP_ACT_KILL_THREAD","SCMP_ACT_LOG","SCMP_ACT_NOTIFY","SCMP_ACT_TRACE","SCMP_ACT_TRAP"],"operators":["SCMP_CMP_EQ","SCMP_CMP_GE","SCMP_CMP_GT","SCMP_CMP_LE","SCMP_CMP_LT","SCMP_CMP_MASKED_EQ","SCMP_CMP_NE"],"archs":["SCMP_ARCH_AARCH64","SCMP_ARCH_ARM","SCMP_ARCH_MIPS","SCMP_ARCH_MIPS64","SCMP_ARCH_MIPS64N32","SCMP_ARCH_MIPSEL","SCMP_ARCH_MIPSEL64","SCMP_ARCH_MIPSEL64N32","SCMP_ARCH_PPC","SCMP_ARCH_PPC64","SCMP_ARCH_PPC64LE","SCMP_ARCH_RISCV64","SCMP_ARCH_S390","SCMP_ARCH_S390X","SCMP_ARCH_X32","SCMP_ARCH_X86","SCMP_ARCH_X86_64"],"knownFlags":["SECCOMP_FILTER_FLAG_TSYNC","SECCOMP_FILTER_FLAG_SPEC_ALLOW","SECCOMP_FILTER_FLAG_LOG"],"supportedFlags":["SECCOMP_FILTER_FLAG_TSYNC","SECCOMP_FILTER_FLAG_SPEC_ALLOW","SECCOMP_FILTER_FLAG_LOG"]},"apparmor":{"enabled":true},"selinux":{"enabled":true},"intelRdt":{"enabled":true},"mountExtensions":{"idmap":{"enabled":true}}},"annotations":{"io.github.seccomp.libseccomp.version":"2.6.0","org.opencontainers.runc.checkpoint.enabled":"true","org.opencontainers.runc.commit":"v1.3.4-0-gd6d73eb8","org.opencontainers.runc.version":"1.3.4\n"},"potentiallyUnsafeConfigAnnotations":["bundle","org.systemd.property.","org.criu.config"]}]}] Default runtime: runc Storage driver: overlay2 @@ -111,22 +112,28 @@ Root dir: /var/lib/docker Server version: 29.3.1 ----- running containers ----- -NAMES IMAGE STATUS PORTS -beszel-agent henrygd/beszel-agent:latest Up 21 hours -dozzle-agent amir20/dozzle:latest Up 22 hours 0.0.0.0:7007->7007/tcp, 8080/tcp -llama-swap-llama-swap-1 ghcr.io/mostlygeek/llama-swap:cuda Up 5 hours (healthy) 0.0.0.0:9292->8080/tcp, [::]:9292->8080/tcp -vllm-rerank vllm/vllm-openai:latest Up 23 hours (healthy) 0.0.0.0:8002->8000/tcp, [::]:8002->8000/tcp -vllm-embed vllm/vllm-openai:latest Up 23 hours (healthy) 0.0.0.0:8001->8000/tcp, [::]:8001->8000/tcp -dockge-dockge-1 louislam/dockge:latest Up 3 weeks (healthy) 0.0.0.0:5001->5001/tcp, [::]:5001->5001/tcp +NAMES IMAGE STATUS PORTS +vllm-qwen35 vllm/vllm-openai Up About an hour (healthy) 0.0.0.0:8007->8000/tcp, [::]:8007->8000/tcp +vllm-granite vllm/vllm-openai:latest Up About an hour (healthy) 0.0.0.0:8004->8000/tcp, [::]:8004->8000/tcp +vllm-rerank vllm/vllm-openai:latest Up 13 hours (healthy) 0.0.0.0:8002->8000/tcp, [::]:8002->8000/tcp +vllm-embed vllm/vllm-openai:latest Up 13 hours (healthy) 0.0.0.0:8001->8000/tcp, [::]:8001->8000/tcp +vllm-reward vllm/vllm-openai:latest Up 13 hours (healthy) 0.0.0.0:8003->8000/tcp, [::]:8003->8000/tcp +llama-swap ghcr.io/mostlygeek/llama-swap:cuda Up 13 hours (healthy) 0.0.0.0:9292->8080/tcp, [::]:9292->8080/tcp +dockge louislam/dockge:latest Up 13 hours (healthy) 0.0.0.0:5001->5001/tcp, [::]:5001->5001/tcp +dozzle-agent amir20/dozzle:latest Up 13 hours 0.0.0.0:7007->7007/tcp, 8080/tcp +beszel-agent henrygd/beszel-agent:latest Up 13 hours (healthy) ----- all containers ----- -NAMES IMAGE STATUS -beszel-agent henrygd/beszel-agent:latest Up 21 hours -dozzle-agent amir20/dozzle:latest Up 22 hours -llama-swap-llama-swap-1 ghcr.io/mostlygeek/llama-swap:cuda Up 5 hours (healthy) -vllm-rerank vllm/vllm-openai:latest Up 23 hours (healthy) -vllm-embed vllm/vllm-openai:latest Up 23 hours (healthy) -dockge-dockge-1 louislam/dockge:latest Up 3 weeks (healthy) +NAMES IMAGE STATUS +vllm-qwen35 vllm/vllm-openai Up About an hour (healthy) +vllm-granite vllm/vllm-openai:latest Up About an hour (healthy) +vllm-rerank vllm/vllm-openai:latest Up 13 hours (healthy) +vllm-embed vllm/vllm-openai:latest Up 13 hours (healthy) +vllm-reward vllm/vllm-openai:latest Up 13 hours (healthy) +llama-swap ghcr.io/mostlygeek/llama-swap:cuda Up 13 hours (healthy) +dockge louislam/dockge:latest Up 13 hours (healthy) +dozzle-agent amir20/dozzle:latest Up 13 hours +beszel-agent henrygd/beszel-agent:latest Up 13 hours (healthy) ----- networks ----- NAME DRIVER SCOPE @@ -145,27 +152,25 @@ llama-swap_default traefik-net ----- named volumes ----- -VOLUME NAME DRIVER -88cea17a7577848d5a34a191ac1a01acb0cd196a2792cfdae6845053ede8b02b local -beszel-agent_beszel_agent_data local -ccb997d07399630358be058be83fe6f5ee64ca307f6dc03c669d9f137644b29a local -dockge_dockge_data local -dozzle-agent_dozzle_agent_data local -librechat_pgdata2 local -parakeet_parakeet_cache local -searxng_searxng-data local +VOLUME NAME DRIVER +beszel-agent-ana_beszel_agent_data local +dockge_dockge_data local +dozzle-agent-ana_dozzle_agent_data local +parakeet_parakeet_cache local +searxng_searxng-data local ----- compose projects currently running ----- -beszel-agent +beszel-agent-ana dockge -dozzle-agent +dozzle-agent-ana llama-swap -vllm-qwen3 +qwen35-vl +vllm ===== COMPOSE FILES (/opt/docker/compose/) ===== ->>> /opt/docker/compose/beszel-agent/compose.yaml +>>> /opt/docker/compose/beszel-agent-ana/compose.yaml # Beszel — lightweight server/container monitoring. # # Hub: single web UI with the SQLite store. Agents: per-host metric collectors @@ -174,7 +179,8 @@ vllm-qwen3 # Multi-host layout via compose profiles: # COMPOSE_PROFILES=hub → hub only (ana-docker) # COMPOSE_PROFILES=hub,agent → hub + local agent on the same host -# COMPOSE_PROFILES=agent → agent only (ana-ml2) +# COMPOSE_PROFILES=agent → agent only (ana-ml2, nh3-docker, +# esh-docker-vm, vm-esh-nas) # # The agent uses network_mode: host so it sees real host CPU/mem/net/disk # counters rather than container-scoped ones — that's why it can't share @@ -186,50 +192,63 @@ services: beszel: image: henrygd/beszel:${BESZEL_VERSION} container_name: beszel - profiles: - - hub + profiles: [hub] restart: unless-stopped ports: - - ${BESZEL_PORT}:8090 + - "${BESZEL_PORT}:8090" volumes: - beszel_data:/beszel_data healthcheck: - test: - - CMD - - wget - - -qO- - - http://localhost:8090/api/health - interval: 30s + # Hub image is distroless — no wget/curl. Use the bundled `/beszel` + # binary's built-in health subcommand (https://beszel.dev/guide/healthchecks). + test: ["CMD", "/beszel", "health", "--url", "http://localhost:8090"] + interval: 120s timeout: 10s retries: 3 start_period: 15s networks: - tnet labels: - - homepage.group=PFI-ANA + - homepage.group=Monitoring - homepage.name=Beszel - homepage.icon=mdi-chart-line - homepage.description=Server + container monitoring - homepage.href=http://10.250.50.70:${BESZEL_PORT} + beszel-agent: image: henrygd/beszel-agent:${BESZEL_VERSION} container_name: beszel-agent - profiles: - - agent + profiles: [agent] restart: unless-stopped network_mode: host volumes: - /var/run/docker.sock:/var/run/docker.sock:ro - beszel_agent_data:/var/lib/beszel-agent environment: + # Agent auth has two modes (v0.13+ supports both side-by-side): + # - KEY-mode: agent listens, hub connects inbound over SSH using KEY. + # Requires BESZEL_HUB_KEY in .env. + # - Token-mode: agent initiates an outbound connection to HUB_URL + # using TOKEN. Easier through NAT. Requires HUB_URL + BESZEL_TOKEN. + # Leave unused ones empty ("") in .env; both can be set simultaneously. - PORT=${BESZEL_AGENT_PORT:-45876} - - KEY=${BESZEL_HUB_KEY} - - HUB_URL=${HUB_URL} - - TOKEN=${BESZEL_TOKEN} + - KEY=${BESZEL_HUB_KEY:-} + - HUB_URL=${HUB_URL:-} + - TOKEN=${BESZEL_TOKEN:-} - EXTRA_FILESYSTEMS=${BESZEL_EXTRA_FS:-} + healthcheck: + # Agent image ships the `/agent` binary with a `health` subcommand. + # Verifies the agent process is up — not that the hub can reach it. + test: ["CMD", "/agent", "health"] + interval: 120s + timeout: 10s + retries: 3 + start_period: 15s + volumes: - beszel_data: null - beszel_agent_data: null + beszel_data: + beszel_agent_data: + networks: tnet: name: traefik-net @@ -278,38 +297,45 @@ networks: external: true >>> /opt/docker/compose/dockge/compose.yaml +# Dockge — per-host Docker Compose UI (https://dockge.kuma.pet/). +# +# One instance runs on every Docker host so the compose dir is manageable +# from a browser. Each host sets DOCKGE_HOST_LABEL + DOCKGE_HOST_IP in its +# .env so the homepage card points at the right place. +# +# All tunables live in .env — edit that, not this file. + services: dockge: - image: louislam/dockge:latest + image: louislam/dockge:${DOCKGE_VERSION:-latest} + container_name: dockge restart: unless-stopped ports: - # Host Port : Container Port - - 5001:5001 + - "${DOCKGE_PORT:-5001}:5001" volumes: - /var/run/docker.sock:/var/run/docker.sock - dockge_data:/app/data - - /opt/docker/compose:/opt/docker/compose - labels: - - homepage.group=PFI-ANA - - homepage.name=Dockge-ML2 - - homepage.icon=si-portainer - - homepage.description=Docker - - homepage.href=http://10.250.50.54:5001 + - /opt/docker:/opt/docker environment: - # Tell Dockge where is your stacks directory - DOCKGE_STACKS_DIR=/opt/docker/compose networks: - tnet + labels: + - homepage.group=Service Networking + - homepage.name=Dockge (${DOCKGE_HOST_LABEL}) + - homepage.icon=sh-dockge.png + - homepage.description=Compose UI on ${DOCKGE_HOST_LABEL} + - homepage.href=http://${DOCKGE_HOST_IP}:${DOCKGE_PORT:-5001} volumes: - dockge_data: null + dockge_data: + networks: tnet: name: traefik-net external: true - ->>> /opt/docker/compose/dozzle-agent/compose.yaml +>>> /opt/docker/compose/dozzle-agent-ana/compose.yaml # Dozzle — container log viewer. # # Multi-host layout via compose profiles: @@ -346,7 +372,7 @@ services: networks: - tnet labels: - - homepage.group=PFI-ANA + - homepage.group=Monitoring - homepage.name=Dozzle - homepage.icon=mdi-text-box-search - homepage.description=Container logs (ana-docker + ana-ml2) @@ -419,27 +445,63 @@ networks: >>> /opt/docker/compose/llama-swap/compose.yaml - services: - llama-swap: - stdin_open: true - tty: true - runtime: nvidia - volumes: - - /opt/docker/conf/llama-swap/config.yaml:/app/config.yaml - - /tank/aimodels/llm:/models - - /tank/aimodels/huggingface:/hfcache - environment: - - HF_HOME=/hfcache - - HF_HUB_CACHE=/hfcache/hub - ports: - - 9292:8080 - image: ghcr.io/mostlygeek/llama-swap:cuda - networks: - - tnet - networks: - tnet: - name: traefik-net - external: true +# llama-swap — GGUF model server with on-demand model swapping. +# +# Proxies OpenAI-compatible API requests to llama.cpp server instances +# and swaps which model is loaded into VRAM per request. Runs on +# ana-ml2 using both GPUs dynamically (no explicit device pinning — +# llama-swap picks per-model-definition). +# +# Model definitions live in /opt/docker/conf/llama-swap/config.yaml on +# the server. Canonical copy of that config is config.yaml in this +# workspace; deploy with scp + `docker compose restart` or the script +# at the bottom of README.md. +# +# All tunables live in .env — edit that, not this file. + +services: + llama-swap: + image: ghcr.io/mostlygeek/llama-swap:${LLAMA_SWAP_VERSION} + container_name: llama-swap + restart: unless-stopped + stdin_open: true + tty: true + runtime: nvidia + ports: + - "${LLAMA_SWAP_PORT}:8080" + volumes: + - /opt/docker/conf/llama-swap/config.yaml:/app/config.yaml + - ${MODELS_DIR}:/models + - ${HF_CACHE_DIR}:/hfcache + environment: + - HF_HOME=/hfcache + - HF_HUB_CACHE=/hfcache/hub + # Pin to GPU 0 — the reserved card for on-demand large-model hot-loads. + # The always-on vLLM services (granite + embed/rerank/reward) own GPU 1; + # keeping llama-swap off GPU 1 stops a hot-loaded model from contending + # with them. llama.cpp then sees only GPU 0 (cuda:0), so --n-gpu-layers + # 999 loads there with no per-model device targeting needed. + - NVIDIA_VISIBLE_DEVICES=${LLAMA_SWAP_GPU:-0} + healthcheck: + test: ["CMD-SHELL", "curl -fsS http://localhost:8080/ >/dev/null || exit 1"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 30s + networks: + - tnet + labels: + - homepage.group=AI Systems + - homepage.name=llama-swap + - homepage.icon=mdi-swap-horizontal + - homepage.description=GGUF model swapper (llama.cpp; ana-ml2) + - homepage.href=http://10.250.50.54:${LLAMA_SWAP_PORT} + +networks: + tnet: + name: traefik-net + external: true + >>> /opt/docker/compose/parakeet/compose.yaml services: parakeet-stt: @@ -474,6 +536,95 @@ networks: volumes: parakeet_cache: null +>>> /opt/docker/compose/qwen35-vl/compose.yaml +# qwen35-vl — Qwen3.5-9B vision-language model (FP8) on ana-ml2. +# +# Co-located on GPU 1 with the granite summarizer + embed/rerank/reward trio +# (GPU 0 is deliberately kept free for hot-reloading large models). Serves on +# :8007, fronted by the LiteLLM gateway as `qwen3.5-9b-fp8`. +# +# WHY A PINNED NIGHTLY DIGEST (not :latest): vLLM :latest (v0.19.1) quantizes +# the Qwen3.5-VL *vision tower* under --quantization fp8, producing garbage +# vision output (the language model is unaffected — it answers text fine but +# "sees" noise). The nightly correctly excludes the vision tower, so vision +# works while the LM still gets the FP8 throughput/VRAM win. We pin the exact +# nightly digest for reproducibility — a moving :nightly tag would silently +# change the engine. WATCH: once the vision-FP8 exclusion lands in a stable +# release, re-pin to :latest and drop this note. +# +# WHY util 0.40 (not the trio's tiny values): the model needs ~34 GB just to +# start at 32k context (FP8 weights + BF16 vision tower + graph capture + 32k +# profiling). On shared GPU 1 (prod uses ~46 GB, ~48 GB free) this vLLM build +# requires free >= util*total, capping util at ~0.51 here; 0.40 (~38 GB) sits +# above the ~34 GB floor with ~10 GB card headroom. +# +# All tunables live in .env — edit that, not this file. + +name: qwen35-vl + +services: + vllm-qwen35: + image: ${QWEN_IMAGE} + container_name: ${QWEN_CONTAINER_NAME} + restart: unless-stopped + ipc: host + ports: + - "${QWEN_PORT}:8000" + volumes: + - /tank/aimodels/huggingface:/hfcache + environment: + - HF_HOME=/hfcache + - HF_HUB_CACHE=/hfcache/hub + - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} + - VLLM_API_KEY=${API_KEY:-} + command: + - ${QWEN_MODEL} + - --served-model-name + - ${QWEN_SERVED_NAME} + - --quantization + - fp8 + - --host + - 0.0.0.0 + - --port + - "8000" + - --gpu-memory-utilization + - ${QWEN_GPU_MEM_UTIL} + - --max-model-len + - ${QWEN_MAX_MODEL_LEN} + - --dtype + - auto + # Prefix caching pinned ON (the nightly defaults it OFF). Free win for the + # text-chat path; marginal for vision (each image is a distinct prefix). + - --enable-prefix-caching + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: + - "${QWEN_GPU_ID}" + capabilities: + - gpu + healthcheck: + test: ["CMD", "curl", "-f", "http://localhost:8000/health"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 300s + networks: + - tnet + labels: + - homepage.group=AI Systems + - homepage.name=Qwen3.5-9B VL (FP8) + - homepage.icon=mdi-image-search + - homepage.description=Qwen3.5-9B vision-language (FP8) via vLLM (ana-ml2) + - homepage.href=http://10.250.50.54:${QWEN_PORT}/docs + +networks: + tnet: + name: traefik-net + external: true + >>> /opt/docker/compose/vibevoice/compose.yaml services: vibevoice: @@ -513,14 +664,20 @@ networks: name: traefik-net external: true ->>> /opt/docker/compose/vllm-qwen3/compose.yaml -# vLLM — Qwen3 Embedding + Reranker (one stack, two services). +>>> /opt/docker/compose/vllm/compose.yaml +# vLLM — Qwen3 Embedding + Reranker + Skywork Reward-V2 classifier. # -# Replaces the unmaintained Infinity stack. vLLM runs one model per process, -# so this stack brings up two containers sharing a single GPU: +# Originally created to replace the unmaintained Infinity stack (embed + +# rerank); generalized 2026-05-13 to host any vLLM-served model on ana-ml2, +# starting with the Skywork-Reward-V2-Llama-3.1-8B reward classifier +# (AWQ-quantized locally, served from /tank/aimodels/llm/). +# +# vLLM runs one model per process, so this stack brings up three containers +# sharing a single GPU: # # vllm-embed — Qwen3-Embedding served as an OpenAI /v1/embeddings server # vllm-rerank — Qwen3-Reranker served as a /rerank + /score server +# vllm-reward — Skywork-Reward-V2-Llama-3.1-8B-AWQ served as a /classify scorer # # The reranker is a causal-LM checkpoint; --hf-overrides re-maps it to # Qwen3ForSequenceClassification so vLLM's reranking endpoints work and the @@ -529,8 +686,14 @@ networks: # All tunables live in .env — edit that, not this file. # # Pre-download models to avoid first-run delay: -# HF_HOME=/tank/aimodels/huggingface hf download Qwen/Qwen3-Embedding-0.6B -# HF_HOME=/tank/aimodels/huggingface hf download Qwen/Qwen3-Reranker-0.6B +# scripts/elway ana-ml2 --playbook playbooks/pull-hf-repo.yaml \ +# --var hf_repo=Qwen/Qwen3-Embedding-0.6B +# scripts/elway ana-ml2 --playbook playbooks/pull-hf-repo.yaml \ +# --var hf_repo=Qwen/Qwen3-Reranker-0.6B +# +# Skywork-Reward-V2-Llama-3.1-8B-AWQ is a locally-quantized model — lives at +# /tank/aimodels/llm/Skywork-Reward-V2-Llama-3.1-8B-AWQ on ana-ml2 and is +# bind-mounted into the reward service at /local-models. Not from HF Hub. services: vllm-embed: @@ -539,7 +702,7 @@ services: restart: unless-stopped ipc: host ports: - - ${EMBED_PORT}:8000 + - "${EMBED_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache environment: @@ -569,15 +732,11 @@ services: devices: - driver: nvidia device_ids: - - ${GPU_ID} + - "${GPU_ID}" capabilities: - gpu healthcheck: - test: - - CMD - - curl - - -f - - http://localhost:8000/health + test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 @@ -590,13 +749,14 @@ services: - homepage.icon=mdi-vector-arrange-below - homepage.description=Qwen3 Embedding via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${EMBED_PORT}/docs + vllm-rerank: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-rerank restart: unless-stopped ipc: host ports: - - ${RERANK_PORT}:8000 + - "${RERANK_PORT}:8000" volumes: - /tank/aimodels/huggingface:/hfcache environment: @@ -628,15 +788,11 @@ services: devices: - driver: nvidia device_ids: - - ${GPU_ID} + - "${GPU_ID}" capabilities: - gpu healthcheck: - test: - - CMD - - curl - - -f - - http://localhost:8000/health + test: ["CMD", "curl", "-f", "http://localhost:8000/health"] interval: 30s timeout: 10s retries: 3 @@ -649,6 +805,140 @@ services: - homepage.icon=mdi-sort-variant - homepage.description=Qwen3 Reranker via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${RERANK_PORT}/docs + + vllm-reward: + image: vllm/vllm-openai:${VLLM_VERSION} + container_name: vllm-reward + restart: unless-stopped + ipc: host + ports: + - "${REWARD_PORT}:8000" + volumes: + # AWQ output lives in the legacy llama-swap models tree, not the HF cache + # — bind-mount the LLM models dir read-only so the reward service can + # load it as a local-path HF-format model. + - /tank/aimodels/llm:/local-models:ro + environment: + - VLLM_API_KEY=${API_KEY:-} + command: + - /local-models/Skywork-Reward-V2-Llama-3.1-8B-AWQ + - --served-model-name + - Skywork/Skywork-Reward-V2-Llama-3.1-8B-AWQ + # vLLM 0.19.1 deprecated --task in favor of --runner. The model's + # config.json declares `LlamaForSequenceClassification` so the + # pooling runner uses it as a classifier (single-label reward score) + # without needing an explicit task flag. + - --runner + - pooling + - --host + - 0.0.0.0 + - --port + - "8000" + - --gpu-memory-utilization + - ${REWARD_GPU_MEM_UTIL} + - --max-model-len + - ${REWARD_MAX_MODEL_LEN} + - --dtype + - auto + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: + - "${GPU_ID}" + capabilities: + - gpu + healthcheck: + test: ["CMD", "curl", "-f", "http://localhost:8000/health"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 240s + networks: + - tnet + labels: + - homepage.group=AI Systems + - homepage.name=vLLM Reward (Skywork) + - homepage.icon=mdi-scale-balance + - homepage.description=Skywork-Reward-V2 8B classifier via vLLM (ana-ml2) + - homepage.href=http://10.250.50.54:${REWARD_PORT}/docs + + # Phi-4-mini (FP8) — summarizer + "dreaming" agent. Supersedes the + # llama-swap granite-4-small pin. Generative chat model (OpenAI + # /v1/chat/completions), so NO --runner pooling. FP8 on RTX 6000 Ada + # (cc 8.9): near-lossless, ~1.2x, ~6 GB. + vllm-granite: + image: vllm/vllm-openai:${VLLM_VERSION} + container_name: vllm-granite + restart: unless-stopped + ipc: host + ports: + - "${GRANITE_PORT}:8000" + volumes: + - /tank/aimodels/huggingface:/hfcache + environment: + - HF_HOME=/hfcache + - HF_HUB_CACHE=/hfcache/hub + - HUGGING_FACE_HUB_TOKEN=${HF_TOKEN:-} + - VLLM_API_KEY=${API_KEY:-} + command: + # Production summarizer (replaced phi4-mini 2026-06-05). Default = official + # IBM pre-quantized FP8 (compressed-tensors), loaded directly; FP8 is native + # on the RTX 6000 Ada (cc 8.9). Fallback to vLLM-native dynamic FP8 from + # BF16: GRANITE_MODEL=ibm-granite/granite-4.1-8b + GRANITE_QUANT=fp8. + - ${GRANITE_MODEL} + - --served-model-name + - ${GRANITE_SERVED_NAME} + - --quantization + - ${GRANITE_QUANT} + - --host + - 0.0.0.0 + - --port + - "8000" + - --gpu-memory-utilization + - ${GRANITE_GPU_MEM_UTIL} + - --max-model-len + - ${GRANITE_MAX_MODEL_LEN} + - --dtype + - auto + # CUDA graphs ENABLED (no --enforce-eager) for decode throughput. Made + # room 2026-06-05 by right-sizing the embed/rerank/reward trio's KV pools + # (they were over-provisioned at 5.9x/2.0x/3.9x concurrency); GPU 1 now has + # ~17 GB free after granite, so graph-capture buffers fit. If the trio + # ever grows back, granite may need --enforce-eager again on this card. + # FP8 KV cache — halves KV memory; near-lossless on Ada (cc 8.9). + - --kv-cache-dtype + - ${GRANITE_KV_CACHE_DTYPE} + # Prefix caching pinned EXPLICIT (vLLM v1 defaults it on, but pin so a + # version flip can't silently disable it). Benched 2026-06-13: ~6.5x faster + # TTFT (45ms vs 292ms) on a shared ~4.5k-token summarizer template; soft/ + # evictable KV, neutral when prefixes don't repeat — pure win for granite. + - --enable-prefix-caching + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: + - "${GRANITE_GPU_ID}" + capabilities: + - gpu + healthcheck: + test: ["CMD", "curl", "-f", "http://localhost:8000/health"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 180s + networks: + - tnet + labels: + - homepage.group=AI Systems + - homepage.name=vLLM Granite 4.1 8B (summarizer) + - homepage.icon=mdi-text-box-outline + - homepage.description=Granite 4.1 8B FP8 via vLLM (ana-ml2) + - homepage.href=http://10.250.50.54:${GRANITE_PORT}/docs + networks: tnet: name: traefik-net @@ -658,63 +948,75 @@ networks: /opt/docker/conf /opt/docker/conf/llama-swap -/opt/docker/conf/llama-swap/config.oldyaml /opt/docker/conf/llama-swap/config.yaml +/opt/docker/conf/vllm ===== LISTENING PORTS ===== +0.0.0.0:111 0.0.0.0:22 0.0.0.0:5001 0.0.0.0:7007 0.0.0.0:8001 0.0.0.0:8002 +0.0.0.0:8003 +0.0.0.0:8004 +0.0.0.0:8007 0.0.0.0:9292 +[::]:111 [::]:22 *:2375 [::]:5001 [::]:8001 [::]:8002 +[::]:8003 +[::]:8004 +[::]:8007 [::]:9292 ===== MODEL / HUGGINGFACE CACHES ===== -/tank/aimodels/huggingface (108G) +/tank/aimodels/huggingface (342G) hub entries: CACHEDIR.TAG datasets--HuggingFaceH4--ultrachat_200k datasets--mlabonne--harmful_behaviors datasets--mlabonne--harmless_alpaca + datasets--Skywork--Skywork-Reward-Preference-80K-v0.2 + datasets--wikitext + models--AxionML--Qwen3.5-9B-NVFP4 models--bartowski--Meta-Llama-3.1-8B-Instruct-GGUF models--bartowski--NousResearch_Hermes-4-14B-GGUF models--bartowski--TheDrummer_GLM-Steam-106B-A12B-v1-GGUF models--bartowski--TheDrummer_Skyfall-31B-v4-GGUF + models--BeaverAI--Artemis-31B-v1i-GGUF models--BeaverAI--Skyfall-R1-31B-v4a-GGUF + models--drawais--Granite-4.1-30B-NVFP4 models--ibm-granite--granite-4.0-h-small-GGUF models--ibm-granite--granite-4.0-h-tiny-GGUF models--ibm-granite--granite-4.0-micro-GGUF + models--ibm-granite--granite-4.1-8b + models--ibm-granite--granite-4.1-8b-fp8 + models--llmfan46--Qwen3.6-35B-A3B-uncensored-heretic-GGUF + models--microsoft--Phi-4-mini-instruct models--mradermacher--Daredevil-8B-abliterated-dpomix-GGUF models--mradermacher--Qwen3-30B-A3B-abliterated-erotic-i1-GGUF models--mradermacher--Qwen3.6-35B-A3B-abliterated-i1-GGUF + models--mradermacher--Selene-1-Mini-Llama-3.1-8B-i1-GGUF + models--murilonwt--granite-4.1-8b-NVFP4 models--newsletter--VibeVoice-Large-pt models--Qwen--Qwen2.5-0.5B-Instruct + models--Qwen--Qwen3.5-9B models--Qwen--Qwen3.6-35B-A3B - models--Qwen--Qwen3-Embedding-0.6B - models--Qwen--Qwen3-Omni-30B-A3B-Instruct - models--Qwen--Qwen3-Reranker-0.6B - models--Qwen--Qwen-Image-Edit-2509 - models--unsloth--embeddinggemma-300m-GGUF - models--unsloth--gemma-4-26B-A4B-it-GGUF - models--unsloth--GLM-4.6-GGUF - models--unsloth--GLM-4.7-Flash-GGUF - models--unsloth--granite-4.0-h-micro-GGUF - models--unsloth--granite-4.0-h-small-GGUF - models--unsloth--granite-4.0-h-tiny-GGUF - models--unsloth--Kimi-K2-Instruct-0905-GGUF -/tank/aimodels/llm (790G) +/tank/aimodels/llm (794G) -/home/lkraven/.cache/huggingface (2.1M) +/home/lkraven/.cache/huggingface (15G) hub entries: + CACHEDIR.TAG + datasets--Salesforce--wikitext + datasets--Skywork--Skywork-Reward-Preference-80K-v0.2 + datasets--wikitext models--Astralyra--bge-reranker-large-Q8_0-GGUF models--ggml-org--embeddinggemma-300M-GGUF models--ggml-org--Qwen3-Reranker-0.6B-Q8_0-GGUF @@ -724,6 +1026,7 @@ networks: models--Mungert--Qwen3-Reranker-0.6B-GGUF models--Qwen--Qwen3-Embedding-0.6B-GGUF models--Qwen--Qwen3-Reranker-0.6B + models--Skywork--Skywork-Reward-V2-Llama-3.1-8B models--unsloth--gemma-4-26B-A4B-it models--unsloth--gemma-4-26B-A4B-it-GGUF models--unsloth--gemma-4-31B-it-GGUF @@ -733,6 +1036,7 @@ networks: containerd.service running docker.service running +nvidia-persistenced.service running ===== DONE ===== diff --git a/stacks/llama-swap/conf/config.yaml b/stacks/llama-swap/conf/config.yaml index a6d3c96..eb3ae4b 100644 --- a/stacks/llama-swap/conf/config.yaml +++ b/stacks/llama-swap/conf/config.yaml @@ -636,9 +636,9 @@ groups: # # Current pins: # qwen3.5-9b — ~6 GB at Q4 + KV. General-purpose chat baseline. - # VRAM budget: ~6 GB persistent in the pin slot. Single RTX 6000 Ada - # is 48 GB, so this leaves ~40 GB for whichever non-pinned model the - # user invokes alongside. + # VRAM budget: ~6 GB persistent in the pin slot. llama-swap is pinned to + # GPU 0 (a single RTX PRO 6000 Blackwell, 96 GB), so this leaves ~90 GB + # for whichever non-pinned model the user invokes alongside. # # granite-4-small WAS pinned here; removed 2026-06-04 — superseded by # phi4-mini (vLLM FP8, stacks/vllm → vllm-phi4). Freed ~24 GB (120K KV). diff --git a/stacks/vllm/.env.example b/stacks/vllm/.env.example index 1f0f9e4..dbe5696 100644 --- a/stacks/vllm/.env.example +++ b/stacks/vllm/.env.example @@ -78,7 +78,7 @@ GRANITE_PORT=8004 # pin llama-swap to GPU 0 for clean separation (follow-up). GRANITE_GPU_ID=1 # Official IBM pre-quantized FP8 (compressed-tensors) — calibrated, ~9.6 GB, -# loaded directly (FP8 native on Ada cc 8.9). Fallback to vLLM-native dynamic FP8 +# loaded directly (FP8 native on Blackwell cc 12.0). Fallback to vLLM-native dynamic FP8 # from BF16: GRANITE_MODEL=ibm-granite/granite-4.1-8b + GRANITE_QUANT=fp8. GRANITE_MODEL=ibm-granite/granite-4.1-8b-fp8 GRANITE_QUANT=compressed-tensors @@ -90,7 +90,7 @@ GRANITE_SERVED_NAME=granite-4.1-8b # summarize turn uses ~1k tokens, so the pool holds ~300 concurrently; the "2.33x" # headline is worst-case (every request maxing 131k). Granite 4.1 supports 131072. GRANITE_MAX_MODEL_LEN=131072 -# FP8 KV cache (native on Ada cc 8.9). At 50K ≈ ~4.2 GB (vs ~8.4 GB at fp16). +# FP8 KV cache (native on Blackwell cc 12.0). At 50K ≈ ~4.2 GB (vs ~8.4 GB at fp16). GRANITE_KV_CACHE_DTYPE=fp8 # util 0.35 (~33.6 GB) — tuned 2026-06-13 to leave ~3.5 GB free on GPU 1 alongside # the trio + qwen co-tenants. On this shared card vLLM needs free >= util*total at diff --git a/stacks/vllm/compose.yaml b/stacks/vllm/compose.yaml index 186ee4b..f6dd7f9 100644 --- a/stacks/vllm/compose.yaml +++ b/stacks/vllm/compose.yaml @@ -197,10 +197,10 @@ services: - homepage.description=Skywork-Reward-V2 8B classifier via vLLM (ana-ml2) - homepage.href=http://10.250.50.54:${REWARD_PORT}/docs - # Phi-4-mini (FP8) — summarizer + "dreaming" agent. Supersedes the - # llama-swap granite-4-small pin. Generative chat model (OpenAI - # /v1/chat/completions), so NO --runner pooling. FP8 on RTX 6000 Ada - # (cc 8.9): near-lossless, ~1.2x, ~6 GB. + # Granite 4.1 8B (FP8) — production summarizer (replaced phi4-mini + # 2026-06-05, which had superseded the llama-swap granite-4-small pin). + # Generative chat model (OpenAI /v1/chat/completions), so NO --runner + # pooling. FP8 on RTX PRO 6000 Blackwell (cc 12.0): near-lossless, ~1.2x, ~6 GB. vllm-granite: image: vllm/vllm-openai:${VLLM_VERSION} container_name: vllm-granite @@ -218,7 +218,7 @@ services: command: # Production summarizer (replaced phi4-mini 2026-06-05). Default = official # IBM pre-quantized FP8 (compressed-tensors), loaded directly; FP8 is native - # on the RTX 6000 Ada (cc 8.9). Fallback to vLLM-native dynamic FP8 from + # on the RTX PRO 6000 Blackwell (cc 12.0). Fallback to vLLM-native dynamic FP8 from # BF16: GRANITE_MODEL=ibm-granite/granite-4.1-8b + GRANITE_QUANT=fp8. - ${GRANITE_MODEL} - --served-model-name @@ -240,7 +240,7 @@ services: # (they were over-provisioned at 5.9x/2.0x/3.9x concurrency); GPU 1 now has # ~17 GB free after granite, so graph-capture buffers fit. If the trio # ever grows back, granite may need --enforce-eager again on this card. - # FP8 KV cache — halves KV memory; near-lossless on Ada (cc 8.9). + # FP8 KV cache — halves KV memory; near-lossless on Blackwell (cc 12.0). - --kv-cache-dtype - ${GRANITE_KV_CACHE_DTYPE} # Prefix caching pinned EXPLICIT (vLLM v1 defaults it on, but pin so a