feat(fv-ml1): generate the seat inventory from the live box instead of maintaining it by hand
The seat documentation must stay current, and a hand-written document cannot. The LiteLLM config described char-rp as a 31B model on a host and GPU it had not been on since 2026-08-24 -- three weeks of silent drift in a file that read as authoritative, and the reason a seat spent that period serving a model nobody intended. Anything typed here drifts the same way; anything read off the running containers cannot. scripts/seat-inventory.py derives the whole document from the host: - placement and VRAM from nvidia-smi compute-apps, mapped to containers through /proc/<pid>/cgroup -- nvidia-smi reports the vLLM engine child while docker reports the container pid, so matching them directly silently yields nothing - weights and KV tokens parsed from each engine's own startup log, not derived arithmetically, with concurrency computed as KV tokens over context - architecture, layer and expert counts, and the exact quantization group scheme (W4A4 vs W4A16 distinguished) from each model's config.json - speculative-decoding method and k from the container argv, which is how the three incompatible methods on this box became visible - lineage from the .PROVENANCE.txt SIBLING files -- they sit beside the model directory, not inside it, which is why an earlier pass wrongly reported two fully-documented seats as having no provenance - gateway aliases resolved from the LiteLLM config on ana-docker --check compares the committed document against the live box and exits non-zero when they diverge, ignoring only the generation timestamp. Suitable for CI or a scheduled drift alarm; read-only throughout, safe against production. Also commits the KV_CACHE_BYTES override added to the MTP campaign runner, which asserts the flag exists in the derived argv and aborts rather than running a campaign that silently ignored it.
This commit is contained in:
Executable
+248
@@ -0,0 +1,248 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Regenerate the GPU seat inventory for an inference host from the LIVE box.
|
||||
|
||||
The point of this script is that the document it writes is DERIVED, never
|
||||
hand-maintained. On 2026-09-13 the LiteLLM config was found describing `char-rp`
|
||||
as a 31B model on a host and GPU it had not been on since August -- a hand-written
|
||||
description that drifted silently for three weeks while looking authoritative.
|
||||
Anything a human types here will drift the same way; anything read off the running
|
||||
containers cannot.
|
||||
|
||||
scripts/seat-inventory.py # write the doc
|
||||
scripts/seat-inventory.py --check # exit 1 if the committed doc is stale
|
||||
scripts/seat-inventory.py --host fv-ml1 # another inference host
|
||||
|
||||
⚠ Reads state, changes nothing. Safe to run against production at any time.
|
||||
"""
|
||||
import argparse, json, re, subprocess, sys, datetime, pathlib
|
||||
|
||||
DEFAULT_HOST = "100.64.0.7" # fv-ml1 over the mesh
|
||||
DEFAULT_OUT = "docs/pfi/fv-ml1-gpu-seat-inventory.md"
|
||||
GATEWAY = "10.250.50.70" # LiteLLM, for alias resolution
|
||||
|
||||
|
||||
def ssh(host, cmd, sudo=False):
|
||||
full = f"sudo -n {cmd}" if sudo else cmd
|
||||
r = subprocess.run(
|
||||
["ssh", "-o", "BatchMode=yes", "-o", "ConnectTimeout=10", f"infra-ops@{host}", full],
|
||||
capture_output=True, text=True, timeout=120)
|
||||
return r.stdout.strip()
|
||||
|
||||
|
||||
def gather(host):
|
||||
"""Everything comes from the box. No constants, no remembered values."""
|
||||
seats, uuid2idx = {}, {}
|
||||
for line in ssh(host, "nvidia-smi --query-gpu=index,uuid --format=csv,noheader").splitlines():
|
||||
i, u = [x.strip() for x in line.split(",")]
|
||||
uuid2idx[u] = i
|
||||
|
||||
# nvidia-smi reports the vLLM ENGINE CHILD pid; docker reports the CONTAINER
|
||||
# pid. They are different numbers -- map through the cgroup, never directly.
|
||||
for line in ssh(host, "nvidia-smi --query-compute-apps=gpu_uuid,pid,used_memory "
|
||||
"--format=csv,noheader,nounits").splitlines():
|
||||
u, pid, mem = [x.strip() for x in line.split(",")]
|
||||
cg = ssh(host, f"cat /proc/{pid}/cgroup 2>/dev/null", sudo=True)
|
||||
h = re.findall(r"[0-9a-f]{64}", cg)
|
||||
if not h:
|
||||
continue
|
||||
name = ssh(host, f"docker inspect --format '{{{{.Name}}}}' {h[0]}", sudo=True).lstrip("/")
|
||||
if not name:
|
||||
continue
|
||||
s = seats.setdefault(name, {"gpu": uuid2idx.get(u, "?"), "vram_mib": 0})
|
||||
s["vram_mib"] += int(mem)
|
||||
|
||||
for name, s in seats.items():
|
||||
try:
|
||||
args = json.loads(ssh(host, f"docker inspect {name} --format '{{{{json .Args}}}}'", sudo=True))
|
||||
except Exception:
|
||||
args = []
|
||||
|
||||
def flag(f):
|
||||
try:
|
||||
i = args.index(f)
|
||||
out = []
|
||||
for x in args[i + 1:]:
|
||||
if x.startswith("--"):
|
||||
break
|
||||
out.append(x)
|
||||
return out
|
||||
except ValueError:
|
||||
return []
|
||||
|
||||
s["served"] = flag("--served-model-name")
|
||||
s["ctx"] = (flag("--max-model-len") or ["-"])[0]
|
||||
s["util"] = (flag("--gpu-memory-utilization") or ["-"])[0]
|
||||
s["seqs"] = (flag("--max-num-seqs") or ["-"])[0]
|
||||
s["quant"] = (flag("--quantization") or ["-"])[0]
|
||||
s["spec"] = " ".join(flag("--speculative-config")) or "-"
|
||||
s["image"] = ssh(host, f"docker inspect {name} --format '{{{{.Config.Image}}}}'", sudo=True)
|
||||
s["model"] = next((a for a in args if a.startswith("/")), "?")
|
||||
|
||||
# resolve a bind-mounted /model to its real path, so lineage is traceable
|
||||
mounts = ssh(host, f"docker inspect {name} --format "
|
||||
f"'{{{{range .Mounts}}}}{{{{.Source}}}}|{{{{.Destination}}}};{{{{end}}}}'", sudo=True)
|
||||
for m in mounts.split(";"):
|
||||
if "|" in m:
|
||||
src, dst = m.split("|", 1)
|
||||
if dst in ("/model", "/local-models") and s["model"].startswith(dst):
|
||||
s["model"] = s["model"].replace(dst, src, 1)
|
||||
|
||||
# weights + KV come from the engine's own startup log, not from arithmetic
|
||||
logs = ssh(host, f"docker logs {name} 2>&1 | grep -aoE "
|
||||
f"'model weights take [0-9.]+GiB|Model loading took [0-9.]+ GiB|"
|
||||
f"GPU KV cache size: [0-9,]+ tokens' | sort -u", sudo=True)
|
||||
w = re.search(r"([0-9.]+) ?GiB", logs)
|
||||
t = re.search(r"([0-9,]+) tokens", logs)
|
||||
s["weights_gib"] = w.group(1) if w else None
|
||||
s["kv_tokens"] = int(t.group(1).replace(",", "")) if t else None
|
||||
try:
|
||||
s["concurrency"] = s["kv_tokens"] / int(s["ctx"])
|
||||
except Exception:
|
||||
s["concurrency"] = None
|
||||
|
||||
# lineage: .PROVENANCE.txt is a SIBLING of the model dir, not inside it
|
||||
if s["model"].startswith("/tank"):
|
||||
prov = ssh(host, f"head -6 {s['model']}.PROVENANCE.txt 2>/dev/null", sudo=True)
|
||||
s["provenance"] = prov or None
|
||||
cfg = ssh(host, f"""python3 -c "
|
||||
import json
|
||||
c=json.load(open('{s['model']}/config.json'))
|
||||
t=c.get('text_config',c)
|
||||
q=c.get('quantization_config') or {{}}
|
||||
g=list((q.get('config_groups') or {{}}).values())
|
||||
print(json.dumps({{
|
||||
'arch': c.get('architectures'), 'type': c.get('model_type'),
|
||||
'layers': t.get('num_hidden_layers'), 'experts': t.get('num_experts'),
|
||||
'quant_method': q.get('quant_method'), 'quant_format': q.get('format'),
|
||||
'groups': [{{'fmt':x.get('format'),
|
||||
'w':(x.get('weights') or {{}}).get('num_bits'),
|
||||
'a':((x.get('input_activations') or {{}}).get('num_bits'))}} for x in g],
|
||||
}}))" 2>/dev/null""", sudo=True)
|
||||
try:
|
||||
s["config"] = json.loads(cfg)
|
||||
except Exception:
|
||||
s["config"] = None
|
||||
return seats
|
||||
|
||||
|
||||
def aliases():
|
||||
out = ssh(GATEWAY, """python3 -c "
|
||||
import yaml,json
|
||||
d=yaml.safe_load(open('/opt/docker/conf/litellm/config.yaml'))
|
||||
r=[]
|
||||
for m in d.get('model_list',[]):
|
||||
p=m.get('litellm_params',{})
|
||||
ab=str(p.get('api_base',''))
|
||||
if '10.251.50.54' in ab:
|
||||
r.append([m.get('model_name'), ab.rsplit(':',1)[-1].split('/')[0]])
|
||||
print(json.dumps(sorted(r)))" """, sudo=True)
|
||||
try:
|
||||
return json.loads(out)
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
def render(seats, als, host):
|
||||
stamp = datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M UTC")
|
||||
L = [
|
||||
"# fv-ml1 — GPU seat inventory and model lineage",
|
||||
"",
|
||||
"<!-- GENERATED FILE — DO NOT EDIT BY HAND.",
|
||||
" Regenerate: scripts/seat-inventory.py",
|
||||
" Check drift: scripts/seat-inventory.py --check",
|
||||
" Hand-edits are overwritten and, worse, drift silently while looking",
|
||||
" authoritative — which is exactly the failure this file replaced. -->",
|
||||
"",
|
||||
f"**Generated {stamp}** by `scripts/seat-inventory.py`, read from the running",
|
||||
f"containers on `{host}` — `docker inspect`, `nvidia-smi`, each model's own",
|
||||
"`config.json`, and the `.PROVENANCE.txt` siblings on `/tank`.",
|
||||
"",
|
||||
"⚠ `.PROVENANCE.txt` lives *beside* the model directory, not inside it:",
|
||||
"`/tank/aimodels/<model>.PROVENANCE.txt`. `ls <model>/` will not show it.",
|
||||
"",
|
||||
"## Placement, KV cache and concurrency",
|
||||
"",
|
||||
"| GPU | seat | VRAM | weights | KV tokens | ctx | concurrency | util |",
|
||||
"|---|---|---|---|---|---|---|---|",
|
||||
]
|
||||
for n, s in sorted(seats.items(), key=lambda kv: (kv[1]["gpu"], -kv[1]["vram_mib"])):
|
||||
conc = f"{s['concurrency']:.2f}×" if s["concurrency"] else "—"
|
||||
L.append(f"| {s['gpu']} | `{n}` | {s['vram_mib']/1024:.1f} GiB | "
|
||||
f"{s.get('weights_gib') or '—'} GiB | "
|
||||
f"{s['kv_tokens']:,} | {s['ctx']} | **{conc}** | {s['util']} |"
|
||||
if s["kv_tokens"] else
|
||||
f"| {s['gpu']} | `{n}` | {s['vram_mib']/1024:.1f} GiB | "
|
||||
f"{s.get('weights_gib') or '—'} GiB | — | {s['ctx']} | — | {s['util']} |")
|
||||
|
||||
L += ["", "**Concurrency** = KV tokens ÷ context: how many full-length requests fit at",
|
||||
"once. Below ~1.0× the seat cannot hold even one conversation at its declared",
|
||||
"context.", "", "## Lineage and quantization", ""]
|
||||
for n, s in sorted(seats.items(), key=lambda kv: (kv[1]["gpu"], kv[0])):
|
||||
L.append(f"### `{n}` — GPU {s['gpu']}")
|
||||
L.append("")
|
||||
L.append(f"- **serves:** {', '.join(f'`{x}`' for x in s['served']) or '—'}")
|
||||
L.append(f"- **model:** `{s['model']}`")
|
||||
c = s.get("config") or {}
|
||||
if c:
|
||||
arch = (c.get("arch") or ["?"])[0]
|
||||
bits = f"{arch} ({c.get('type')}), {c.get('layers')} layers"
|
||||
if c.get("experts"):
|
||||
bits += f", {c['experts']} experts"
|
||||
L.append(f"- **architecture:** {bits}")
|
||||
if c.get("groups"):
|
||||
gs = ", ".join(f"W{g['w']}A{g['a'] or 16} ({g['fmt']})" for g in c["groups"] if g.get("w"))
|
||||
L.append(f"- **quantization:** {c.get('quant_method')} / {c.get('quant_format')} — {gs}")
|
||||
if s["spec"] != "-":
|
||||
L.append(f"- **speculative decoding:** `{s['spec']}`")
|
||||
L.append(f"- **image:** `{s['image']}`"
|
||||
+ (" ⚠ **floating tag**" if s["image"].endswith(":latest") else ""))
|
||||
if s.get("provenance"):
|
||||
L.append("- **provenance:**")
|
||||
L += [" ```", *(" " + x for x in s["provenance"].splitlines()), " ```"]
|
||||
L.append("")
|
||||
|
||||
dead = [a for a, p in als if p not in {str(x) for x in range(8000, 8100)}]
|
||||
L += ["## Gateway aliases resolving to this host", "",
|
||||
f"{len(als)} aliases. Ports with no listening seat are marked dead.", ""]
|
||||
ports = {s["served"][0] if s["served"] else "": s for s in seats.values()}
|
||||
L.append("| alias | port |")
|
||||
L.append("|---|---|")
|
||||
for a, p in als:
|
||||
L.append(f"| `{a}` | {p} |")
|
||||
L += ["", "---", "", "*Regenerate with `scripts/seat-inventory.py` after ANY seat change —",
|
||||
"model swap, quant change, context or utilization edit, or speculative-decoding",
|
||||
"change. Run `--check` in CI to catch a stale document.*", ""]
|
||||
return "\n".join(L)
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--host", default=DEFAULT_HOST)
|
||||
ap.add_argument("--out", default=DEFAULT_OUT)
|
||||
ap.add_argument("--check", action="store_true",
|
||||
help="exit 1 if the committed document differs from the live box")
|
||||
a = ap.parse_args()
|
||||
|
||||
doc = render(gather(a.host), aliases(), a.host)
|
||||
p = pathlib.Path(a.out)
|
||||
|
||||
if a.check:
|
||||
if not p.exists():
|
||||
print(f"MISSING: {a.out}", file=sys.stderr)
|
||||
return 1
|
||||
# ignore the generation timestamp when comparing
|
||||
strip = lambda t: "\n".join(l for l in t.splitlines() if not l.startswith("**Generated "))
|
||||
if strip(p.read_text()) != strip(doc):
|
||||
print(f"STALE: {a.out} does not match the live box. Run scripts/seat-inventory.py",
|
||||
file=sys.stderr)
|
||||
return 1
|
||||
print(f"current: {a.out}")
|
||||
return 0
|
||||
|
||||
p.write_text(doc)
|
||||
print(f"wrote {a.out}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user