diff --git a/playbooks/gx10-outfit-privileged.yaml b/playbooks/gx10-outfit-privileged.yaml new file mode 100644 index 0000000..fd825ca --- /dev/null +++ b/playbooks/gx10-outfit-privileged.yaml @@ -0,0 +1,107 @@ +# pfi-gx10 — the privileged half of outfitting the box as an inference + training +# machine. The userspace half (uv, ~/ml/.venv, torch 2.14.0+cu130, the whole +# transformers/peft/trl stack) is ALREADY DONE and needed no root; this playbook +# is only the part that does. +# +# scripts/elway infra-ops@10.100.10.226 --playbook playbooks/gx10-outfit-privileged.yaml +# +# ───────────────────────────────────────────────────────────────────────────── +# ⚠ PREREQUISITE — infra-ops has NO NOPASSWD SUDO ON THIS BOX +# +# pfi-gx10 is the fleet exception: infra-ops is in the `sudo` group with +# `(ALL : ALL) ALL`, but every invocation wants a password, and no gx10 +# credential is in the vault. elway prompts once via getpass, so an interactive +# operator can just run this. A non-interactive agent cannot. +# +# Step 1 below closes that permanently and makes gx10 consistent with the rest +# of the fleet. Run this playbook once interactively and the exception is gone. +# ───────────────────────────────────────────────────────────────────────────── +# +# WHY BARE METAL AND NOT A VM (settled 2026-09-01, do not re-litigate): +# Proxmox VE has no aarch64 build, and the GB10's GPU sits on an on-package +# NVIDIA root complex cache-coherent with the CPU over NVLink-C2C sharing one +# pool of LPDDR5X. Passing it to a guest partitions the unified memory that is +# the entire reason for the box. See persistent-memory.d/2026-09-01-pfi-gx10-*. + +vars: + user: infra-ops + sudoers_file: /etc/sudoers.d/infra-ops-nopasswd + cuda_keyring_url: https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2404/arm64/cuda-keyring_1.1-1_all.deb + +steps: + - name: Grant infra-ops NOPASSWD sudo (ends the gx10 fleet exception) + # visudo -cf validates BEFORE the file is put in place. A malformed + # sudoers drop-in locks every sudo user out of the box, and this one is + # on a desk with no iDRAC. + sudo: true + shell: | + printf '%s ALL=(ALL) NOPASSWD:ALL\n' {{ user }} > /tmp/.sudoers-candidate + visudo -cf /tmp/.sudoers-candidate + install -m 0440 -o root -g root /tmp/.sudoers-candidate {{ sudoers_file }} + rm -f /tmp/.sudoers-candidate + when: "! sudo -n true 2>/dev/null" + + - name: Verify NOPASSWD actually took before relying on it + # Asserting the effective behaviour, not the presence of a file. + shell: sudo -n true + changed_when: "false" + + - name: Add infra-ops to the docker group + # The daemon already runs; infra-ops just gets EACCES on the socket. + sudo: true + shell: usermod -aG docker {{ user }} + when: "! id -nG {{ user }} | tr ' ' '\\n' | grep -qx docker" + + - name: Install the NVIDIA CUDA apt repository keyring + sudo: true + shell: | + cd /tmp + curl -fsSLO {{ cuda_keyring_url }} + dpkg -i cuda-keyring_1.1-1_all.deb + rm -f cuda-keyring_1.1-1_all.deb + when: "! test -f /etc/apt/sources.list.d/cuda-ubuntu2404-arm64.list" + + - name: apt update + sudo: true + shell: apt-get update -qq + + - name: Install the CUDA toolkit (nvcc and friends) + # NOT `cuda` or `cuda-drivers` -- the driver is already installed and + # working (580.173.02) and pulling the metapackage risks replacing it. + # cuda-toolkit is the compiler + libraries only. + sudo: true + shell: DEBIAN_FRONTEND=noninteractive apt-get install -y -qq cuda-toolkit + when: "! test -x /usr/local/cuda/bin/nvcc" + + - name: Install the NVIDIA Container Toolkit + sudo: true + shell: DEBIAN_FRONTEND=noninteractive apt-get install -y -qq nvidia-container-toolkit + when: "! command -v nvidia-ctk >/dev/null 2>&1" + + - name: Wire the container runtime into dockerd + sudo: true + shell: nvidia-ctk runtime configure --runtime=docker && systemctl restart docker + when: "! grep -q nvidia /etc/docker/daemon.json 2>/dev/null" + +verify: + - name: NOPASSWD sudo works + shell: sudo -n true + changed_when: "false" + + - name: nvcc is on PATH and reports a version + shell: /usr/local/cuda/bin/nvcc --version | grep -q "release" + changed_when: "false" + + - name: infra-ops can reach the docker socket + # `sg docker` picks up the new group without needing a fresh login. + shell: sg docker -c 'docker info >/dev/null' + changed_when: "false" + + - name: A CUDA container can see the GB10 + shell: sg docker -c 'docker run --rm --gpus all nvidia/cuda:13.0.0-base-ubuntu24.04 nvidia-smi -L' | grep -q GB10 + changed_when: "false" + + - name: The userspace torch stack still sees the GPU + # Guards against a toolkit install disturbing the working driver. + shell: /home/infra-ops/ml/.venv/bin/python -c 'import torch;assert torch.cuda.is_available()' + changed_when: "false"