"""Does GPU memory outside torch's allocator grow with the number of distinct threads that run a call?""" import json, subprocess, sys, threading, time, urllib.request URL = "http://127.0.0.1:18033" TOKEN = open("token").read().strip() BODY = json.dumps({"id": "x", "state": "The deploy passed.", "question": "Did it pass?", "options": [{"id": "yes", "description": "Yes"}, {"id": "no", "description": "No"}]}).encode() def card(): return int(subprocess.check_output(["nvidia-smi", "-i", "3", "--query-gpu=memory.used", "--format=csv,noheader,nounits"]).strip()) def reserved(): return json.load(urllib.request.urlopen(URL + "/health"))["model"]["reserved_gib"] * 1024 def post(): r = urllib.request.Request(URL + "/decide", data=BODY, headers={"Authorization": "Bearer " + TOKEN, "Content-Type": "application/json"}) urllib.request.urlopen(r).read() def snap(tag): time.sleep(1.5); c, r = card(), reserved(); print(f"{tag}: card {c} MiB, reserved {r:.0f} MiB, outside allocator {c - 2 - r:.0f} MiB", flush=True) snap("start") for i in range(20): post() snap("after 20 sequential requests") for burst in range(3): ts = [threading.Thread(target=post) for _ in range(40)] [t.start() for t in ts]; [t.join() for t in ts] snap(f"after concurrent burst {burst + 1} (40 requests)")