"""GPU memory outside torch's allocator (whole card minus 2 MiB idle minus torch reserved), at quiet points between loads. GPU 3 carries only this process.""" import json, subprocess, sys, threading, time, urllib.request URL = "http://127.0.0.1:18033" TOKEN = open("token").read().strip() H = {"Authorization": "Bearer " + TOKEN, "Content-Type": "application/json"} def card(): return int(subprocess.check_output(["nvidia-smi", "-i", "3", "--query-gpu=memory.used", "--format=csv,noheader,nounits"]).strip()) def health(): return json.load(urllib.request.urlopen(URL + "/health"))["model"] def post(path, body): r = urllib.request.Request(URL + path, data=json.dumps(body).encode(), headers=H) try: return urllib.request.urlopen(r).status except urllib.error.HTTPError as e: return e.code SMALL = {"id": "x", "state": "The deploy passed.", "question": "Did it pass?", "options": [{"id": "yes", "description": "Yes"}, {"id": "no", "description": "No"}]} out = [] def snap(tag): time.sleep(1.5) c, h = card(), health() r = h["reserved_gib"] * 1024 row = {"tag": tag, "card_mib": c, "reserved_mib": round(r), "outside_mib": round(c - 2 - r), "max_reserved_mib": round(h["max_reserved_gib"] * 1024)} out.append(row); print(row, flush=True) snap("rest") for burst in range(2): ts = [threading.Thread(target=post, args=("/decide", SMALL)) for _ in range(30)] [t.start() for t in ts]; [t.join() for t in ts] snap(f"after concurrent burst {burst + 1}") subprocess.run([sys.executable, "bench_shape.py", "--backend", "semif", "--url", URL, "--token-file", "token", "--long-state-file", "long_state.txt", "--label", "probe", "--out", sys.argv[1] + "/shape.json", "--runs", "1", "--per", "5"], check=True) snap("after latency shapes") subprocess.run([sys.executable, "checks.py", "--url", URL, "--token-file", "token", "--out", sys.argv[1] + "/checks.json", "--checks", "maxreq,oom", "--repeats", "2"], check=True) snap("after largest requests") json.dump(out, open(sys.argv[1] + "/overhead.json", "w"), indent=1)