stacks/intern-decision: compose (GPU 1, :8033, hard VRAM cap as the single .env knob, healthcheck, Homepage group 'AI - Eval & Retrieval'), .env.example and README. dns: intern-decision.fv.internal -> fv-ml1 (synced to ana/esh/nh3). acceptance on fv-ml1 GPU 3, 3 fresh processes: bit-identical to the Jev bench's native rows (pooled 240/259, Wyrd 79/84, 0/560 flips, Δp 0), negative control 10/122/14, 0 flips across restarts; largest accepted request 200 at a 10,134 MiB card peak under a 9.25 GiB cap; 503 and recovery proven at a tight cap. GPU 1 deploy held: nvidia-smi Free on GPU 1 is 15,442 MiB.
38 lines
2.1 KiB
Python
38 lines
2.1 KiB
Python
"""GPU memory outside torch's allocator (whole card minus 2 MiB idle minus torch reserved), at quiet points
|
|
between loads. GPU 3 carries only this process."""
|
|
import json, subprocess, sys, threading, time, urllib.request
|
|
URL = "http://127.0.0.1:18033"
|
|
TOKEN = open("token").read().strip()
|
|
H = {"Authorization": "Bearer " + TOKEN, "Content-Type": "application/json"}
|
|
def card():
|
|
return int(subprocess.check_output(["nvidia-smi", "-i", "3", "--query-gpu=memory.used", "--format=csv,noheader,nounits"]).strip())
|
|
def health():
|
|
return json.load(urllib.request.urlopen(URL + "/health"))["model"]
|
|
def post(path, body):
|
|
r = urllib.request.Request(URL + path, data=json.dumps(body).encode(), headers=H)
|
|
try:
|
|
return urllib.request.urlopen(r).status
|
|
except urllib.error.HTTPError as e:
|
|
return e.code
|
|
SMALL = {"id": "x", "state": "The deploy passed.", "question": "Did it pass?",
|
|
"options": [{"id": "yes", "description": "Yes"}, {"id": "no", "description": "No"}]}
|
|
out = []
|
|
def snap(tag):
|
|
time.sleep(1.5)
|
|
c, h = card(), health()
|
|
r = h["reserved_gib"] * 1024
|
|
row = {"tag": tag, "card_mib": c, "reserved_mib": round(r), "outside_mib": round(c - 2 - r), "max_reserved_mib": round(h["max_reserved_gib"] * 1024)}
|
|
out.append(row); print(row, flush=True)
|
|
snap("rest")
|
|
for burst in range(2):
|
|
ts = [threading.Thread(target=post, args=("/decide", SMALL)) for _ in range(30)]
|
|
[t.start() for t in ts]; [t.join() for t in ts]
|
|
snap(f"after concurrent burst {burst + 1}")
|
|
subprocess.run([sys.executable, "bench_shape.py", "--backend", "semif", "--url", URL, "--token-file", "token",
|
|
"--long-state-file", "long_state.txt", "--label", "probe", "--out", sys.argv[1] + "/shape.json", "--runs", "1", "--per", "5"], check=True)
|
|
snap("after latency shapes")
|
|
subprocess.run([sys.executable, "checks.py", "--url", URL, "--token-file", "token", "--out", sys.argv[1] + "/checks.json",
|
|
"--checks", "maxreq,oom", "--repeats", "2"], check=True)
|
|
snap("after largest requests")
|
|
json.dump(out, open(sys.argv[1] + "/overhead.json", "w"), indent=1)
|