Files
esh-pfi-infrastructure/scripts/seat-inventory.py
T
vh dfa91a8eaf docs(fv-ml1): add curated LLM seat catalog (lineage/provenance/cards/speed) + bench script
Adds docs/pfi/llm-seat-catalog.md, the durable hand-curated record of what each
seat IS -- lineage, provenance, model-card facts, quantization, speculative
decoding, licenses, and measured warm tok/s + deep-prefill depth results with
their harness and date. It complements the auto-generated
fv-ml1-gpu-seat-inventory.md (live placement/KV/concurrency): the two split by
volatility, and the catalog defers to the inventory for any live number.

Adds scripts/seat-bench.py so the catalog's speed/depth numbers are reproducible
(warm decode tok/s, n=3, conc=1, median; deep prefill at ~0.97x max-model-len
with an allocator-log OOM scan). Serial by design -- concurrent deep prefills
would confound both OOM and tok/s.

Captures the 2026-09-14 measurements: all six generative seats prefill to ~255K
(coder ~16K) with zero OOM/CUBLAS/illegal-memory; warm decode 62.7-337.3 tok/s;
per-seat VRAM. seat-inventory.py now cross-links the catalog in its footer.
2026-09-14 10:45:31 -07:00

252 lines
11 KiB
Python
Executable File
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Regenerate the GPU seat inventory for an inference host from the LIVE box.
The point of this script is that the document it writes is DERIVED, never
hand-maintained. On 2026-09-13 the LiteLLM config was found describing `char-rp`
as a 31B model on a host and GPU it had not been on since August -- a hand-written
description that drifted silently for three weeks while looking authoritative.
Anything a human types here will drift the same way; anything read off the running
containers cannot.
scripts/seat-inventory.py # write the doc
scripts/seat-inventory.py --check # exit 1 if the committed doc is stale
scripts/seat-inventory.py --host fv-ml1 # another inference host
⚠ Reads state, changes nothing. Safe to run against production at any time.
"""
import argparse, json, re, subprocess, sys, datetime, pathlib
DEFAULT_HOST = "100.64.0.7" # fv-ml1 over the mesh
DEFAULT_OUT = "docs/pfi/fv-ml1-gpu-seat-inventory.md"
GATEWAY = "10.250.50.70" # LiteLLM, for alias resolution
def ssh(host, cmd, sudo=False):
full = f"sudo -n {cmd}" if sudo else cmd
r = subprocess.run(
["ssh", "-o", "BatchMode=yes", "-o", "ConnectTimeout=10", f"infra-ops@{host}", full],
capture_output=True, text=True, timeout=120)
return r.stdout.strip()
def gather(host):
"""Everything comes from the box. No constants, no remembered values."""
seats, uuid2idx = {}, {}
for line in ssh(host, "nvidia-smi --query-gpu=index,uuid --format=csv,noheader").splitlines():
i, u = [x.strip() for x in line.split(",")]
uuid2idx[u] = i
# nvidia-smi reports the vLLM ENGINE CHILD pid; docker reports the CONTAINER
# pid. They are different numbers -- map through the cgroup, never directly.
for line in ssh(host, "nvidia-smi --query-compute-apps=gpu_uuid,pid,used_memory "
"--format=csv,noheader,nounits").splitlines():
u, pid, mem = [x.strip() for x in line.split(",")]
cg = ssh(host, f"cat /proc/{pid}/cgroup 2>/dev/null", sudo=True)
h = re.findall(r"[0-9a-f]{64}", cg)
if not h:
continue
name = ssh(host, f"docker inspect --format '{{{{.Name}}}}' {h[0]}", sudo=True).lstrip("/")
if not name:
continue
s = seats.setdefault(name, {"gpu": uuid2idx.get(u, "?"), "vram_mib": 0})
s["vram_mib"] += int(mem)
for name, s in seats.items():
try:
args = json.loads(ssh(host, f"docker inspect {name} --format '{{{{json .Args}}}}'", sudo=True))
except Exception:
args = []
def flag(f):
try:
i = args.index(f)
out = []
for x in args[i + 1:]:
if x.startswith("--"):
break
out.append(x)
return out
except ValueError:
return []
s["served"] = flag("--served-model-name")
s["ctx"] = (flag("--max-model-len") or ["-"])[0]
s["util"] = (flag("--gpu-memory-utilization") or ["-"])[0]
s["seqs"] = (flag("--max-num-seqs") or ["-"])[0]
s["quant"] = (flag("--quantization") or ["-"])[0]
s["spec"] = " ".join(flag("--speculative-config")) or "-"
s["image"] = ssh(host, f"docker inspect {name} --format '{{{{.Config.Image}}}}'", sudo=True)
s["model"] = next((a for a in args if a.startswith("/")), "?")
# resolve a bind-mounted /model to its real path, so lineage is traceable
mounts = ssh(host, f"docker inspect {name} --format "
f"'{{{{range .Mounts}}}}{{{{.Source}}}}|{{{{.Destination}}}};{{{{end}}}}'", sudo=True)
for m in mounts.split(";"):
if "|" in m:
src, dst = m.split("|", 1)
if dst in ("/model", "/local-models") and s["model"].startswith(dst):
s["model"] = s["model"].replace(dst, src, 1)
# weights + KV come from the engine's own startup log, not from arithmetic
logs = ssh(host, f"docker logs {name} 2>&1 | grep -aoE "
f"'model weights take [0-9.]+GiB|Model loading took [0-9.]+ GiB|"
f"GPU KV cache size: [0-9,]+ tokens' | sort -u", sudo=True)
w = re.search(r"([0-9.]+) ?GiB", logs)
t = re.search(r"([0-9,]+) tokens", logs)
s["weights_gib"] = w.group(1) if w else None
s["kv_tokens"] = int(t.group(1).replace(",", "")) if t else None
try:
s["concurrency"] = s["kv_tokens"] / int(s["ctx"])
except Exception:
s["concurrency"] = None
# lineage: .PROVENANCE.txt is a SIBLING of the model dir, not inside it
if s["model"].startswith("/tank"):
prov = ssh(host, f"head -6 {s['model']}.PROVENANCE.txt 2>/dev/null", sudo=True)
s["provenance"] = prov or None
cfg = ssh(host, f"""python3 -c "
import json
c=json.load(open('{s['model']}/config.json'))
t=c.get('text_config',c)
q=c.get('quantization_config') or {{}}
g=list((q.get('config_groups') or {{}}).values())
print(json.dumps({{
'arch': c.get('architectures'), 'type': c.get('model_type'),
'layers': t.get('num_hidden_layers'), 'experts': t.get('num_experts'),
'quant_method': q.get('quant_method'), 'quant_format': q.get('format'),
'groups': [{{'fmt':x.get('format'),
'w':(x.get('weights') or {{}}).get('num_bits'),
'a':((x.get('input_activations') or {{}}).get('num_bits'))}} for x in g],
}}))" 2>/dev/null""", sudo=True)
try:
s["config"] = json.loads(cfg)
except Exception:
s["config"] = None
return seats
def aliases():
out = ssh(GATEWAY, """python3 -c "
import yaml,json
d=yaml.safe_load(open('/opt/docker/conf/litellm/config.yaml'))
r=[]
for m in d.get('model_list',[]):
p=m.get('litellm_params',{})
ab=str(p.get('api_base',''))
if '10.251.50.54' in ab:
r.append([m.get('model_name'), ab.rsplit(':',1)[-1].split('/')[0]])
print(json.dumps(sorted(r)))" """, sudo=True)
try:
return json.loads(out)
except Exception:
return []
def render(seats, als, host):
stamp = datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M UTC")
L = [
"# fv-ml1 — GPU seat inventory and model lineage",
"",
"<!-- GENERATED FILE — DO NOT EDIT BY HAND.",
" Regenerate: scripts/seat-inventory.py",
" Check drift: scripts/seat-inventory.py --check",
" Hand-edits are overwritten and, worse, drift silently while looking",
" authoritative — which is exactly the failure this file replaced. -->",
"",
f"**Generated {stamp}** by `scripts/seat-inventory.py`, read from the running",
f"containers on `{host}` — `docker inspect`, `nvidia-smi`, each model's own",
"`config.json`, and the `.PROVENANCE.txt` siblings on `/tank`.",
"",
"⚠ `.PROVENANCE.txt` lives *beside* the model directory, not inside it:",
"`/tank/aimodels/<model>.PROVENANCE.txt`. `ls <model>/` will not show it.",
"",
"## Placement, KV cache and concurrency",
"",
"| GPU | seat | VRAM | weights | KV tokens | ctx | concurrency | util |",
"|---|---|---|---|---|---|---|---|",
]
for n, s in sorted(seats.items(), key=lambda kv: (kv[1]["gpu"], -kv[1]["vram_mib"])):
conc = f"{s['concurrency']:.2f}×" if s["concurrency"] else ""
L.append(f"| {s['gpu']} | `{n}` | {s['vram_mib']/1024:.1f} GiB | "
f"{s.get('weights_gib') or ''} GiB | "
f"{s['kv_tokens']:,} | {s['ctx']} | **{conc}** | {s['util']} |"
if s["kv_tokens"] else
f"| {s['gpu']} | `{n}` | {s['vram_mib']/1024:.1f} GiB | "
f"{s.get('weights_gib') or ''} GiB | — | {s['ctx']} | — | {s['util']} |")
L += ["", "**Concurrency** = KV tokens ÷ context: how many full-length requests fit at",
"once. Below ~1.0× the seat cannot hold even one conversation at its declared",
"context.", "", "## Lineage and quantization", ""]
for n, s in sorted(seats.items(), key=lambda kv: (kv[1]["gpu"], kv[0])):
L.append(f"### `{n}` — GPU {s['gpu']}")
L.append("")
L.append(f"- **serves:** {', '.join(f'`{x}`' for x in s['served']) or ''}")
L.append(f"- **model:** `{s['model']}`")
c = s.get("config") or {}
if c:
arch = (c.get("arch") or ["?"])[0]
bits = f"{arch} ({c.get('type')}), {c.get('layers')} layers"
if c.get("experts"):
bits += f", {c['experts']} experts"
L.append(f"- **architecture:** {bits}")
if c.get("groups"):
gs = ", ".join(f"W{g['w']}A{g['a'] or 16} ({g['fmt']})" for g in c["groups"] if g.get("w"))
L.append(f"- **quantization:** {c.get('quant_method')} / {c.get('quant_format')}{gs}")
if s["spec"] != "-":
L.append(f"- **speculative decoding:** `{s['spec']}`")
L.append(f"- **image:** `{s['image']}`"
+ (" ⚠ **floating tag**" if s["image"].endswith(":latest") else ""))
if s.get("provenance"):
L.append("- **provenance:**")
L += [" ```", *(" " + x for x in s["provenance"].splitlines()), " ```"]
L.append("")
dead = [a for a, p in als if p not in {str(x) for x in range(8000, 8100)}]
L += ["## Gateway aliases resolving to this host", "",
f"{len(als)} aliases. Ports with no listening seat are marked dead.", ""]
ports = {s["served"][0] if s["served"] else "": s for s in seats.values()}
L.append("| alias | port |")
L.append("|---|---|")
for a, p in als:
L.append(f"| `{a}` | {p} |")
L += ["", "---", "",
"*Lineage, provenance, model cards, measured tok/s and depth results live in the "
"hand-curated companion [`llm-seat-catalog.md`](llm-seat-catalog.md).*", "",
"*Regenerate with `scripts/seat-inventory.py` after ANY seat change —",
"model swap, quant change, context or utilization edit, or speculative-decoding",
"change. Run `--check` in CI to catch a stale document.*", ""]
return "\n".join(L)
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--host", default=DEFAULT_HOST)
ap.add_argument("--out", default=DEFAULT_OUT)
ap.add_argument("--check", action="store_true",
help="exit 1 if the committed document differs from the live box")
a = ap.parse_args()
doc = render(gather(a.host), aliases(), a.host)
p = pathlib.Path(a.out)
if a.check:
if not p.exists():
print(f"MISSING: {a.out}", file=sys.stderr)
return 1
# ignore the generation timestamp when comparing
strip = lambda t: "\n".join(l for l in t.splitlines() if not l.startswith("**Generated "))
if strip(p.read_text()) != strip(doc):
print(f"STALE: {a.out} does not match the live box. Run scripts/seat-inventory.py",
file=sys.stderr)
return 1
print(f"current: {a.out}")
return 0
p.write_text(doc)
print(f"wrote {a.out}")
return 0
if __name__ == "__main__":
sys.exit(main())