"""Run a Python module or script under a hard per-process VRAM cap, the way semif-serve applies SEMIF_VRAM_CAP_GIB (torch.cuda.set_per_process_memory_fraction BEFORE any weights load). BENCH_VRAM_CAP_GIB=12 python capped.py [args...] Unset or 0 = no cap. The cap covers torch's allocator only, as in semif-serve; the CUDA context (~0.5-0.9 GiB) sits outside it, so nvidia-smi reads higher than the cap would suggest.""" import os import runpy import sys import torch cap = float(os.environ.get("BENCH_VRAM_CAP_GIB") or 0) if cap: total = torch.cuda.get_device_properties(0).total_memory torch.cuda.set_per_process_memory_fraction(cap * 2**30 / total, 0) print(f"[capped] per-process VRAM cap {cap} GiB of {total / 2**30:.1f} GiB", flush=True) target, sys.argv = sys.argv[1], sys.argv[1:] if target.endswith(".py"): sys.path.insert(0, os.path.dirname(os.path.abspath(target))) runpy.run_path(target, run_name="__main__") else: runpy.run_module(target, run_name="__main__", alter_sys=True)