services/semif-serve is a FastAPI wrapper around SemIf's direct and shared torch scorers (SemIf-OpenJev @ 23cf1f39, MIT). Upstream ships only a batch CLI. The wrapper loads the pinned Qwen3.5-4B (851bf6e8, BF16) once from the offline HF cache and returns SemIf's result dicts unchanged, with an optional per-workload temperature-calibrated view. Contract: semif-serve.contract.md. Built with a short contract, TDD (39 tests, fake engine and fake torch, no GPU) and a heid bug-hunt panel (pending). On the card: - torch 2.10.0+cu128 with sm_120 kernels, which is SemIf's own stack; - a hard 12 GiB VRAM cap. Two defects surfaced only on the card, and each fix is covered by a test: - 0.1.1: an OOM raised as a chained exception kept the failed request's tensors alive (11.9 GiB after the 503). It is now raised unchained, after gc. - 0.1.2: a large request left 12.6 GB reserved on the shared card. After each call, reserved memory over the baseline + 512 MiB is now released. Acceptance against SemIf's committed torch predictions (authored144): - 142/144 same top choice; both misses are exact bf16 ties; - 144/144 identical prompt hashes; - deterministic A-vs-A; - negative control 14/144; - shared vs direct 72/72. 21 binary criteria over one state take 159 ms. The shared-mode capacity table under the cap is in stacks/semif/README.md. The Dockerfile installs dependencies from a manifest with the project version blanked, so a version bump reuses the ~4 GB torch layer. Verified: 41 s rebuild, dependency layer CACHED. DNS: semif.fv.internal. Token: vault semif/api-token.
84 lines
3.3 KiB
Python
84 lines
3.3 KiB
Python
"""TorchEngine's OOM path against a fake torch (INV-4). Found on the card 2026-09-27: after a
|
|
503 the failed call's tensors stayed alive (11.92 GiB allocated) because the raised
|
|
OutOfMemory chained back to the torch exception, whose traceback held the scorer's frames."""
|
|
import weakref
|
|
|
|
import pytest
|
|
|
|
from semif_serve.config import Settings
|
|
from semif_serve.engine import TorchEngine
|
|
from semif_serve.errors import OutOfMemory
|
|
|
|
|
|
class FakeTorch:
|
|
class cuda:
|
|
class OutOfMemoryError(RuntimeError):
|
|
pass
|
|
|
|
empties = [] # for each empty_cache() call: was the failed call's tensor already freed?
|
|
watched = []
|
|
|
|
@classmethod
|
|
def empty_cache(cls):
|
|
cls.empties.append(all(ref() is None for ref in cls.watched))
|
|
|
|
|
|
class Tensor:
|
|
pass
|
|
|
|
|
|
def failing_scorer(*_args):
|
|
kv_cache = Tensor() # stands in for the replicated prefix cache
|
|
FakeTorch.cuda.watched.append(weakref.ref(kv_cache))
|
|
raise FakeTorch.cuda.OutOfMemoryError("CUDA out of memory. Tried to allocate 490.00 MiB.\nGPU 0 has ...")
|
|
|
|
|
|
@pytest.mark.parametrize("call", ["direct", "shared"])
|
|
def test_oom_frees_the_failed_call_before_emptying_the_cache_and_keeps_nothing_alive(call):
|
|
FakeTorch.cuda.empties.clear(), FakeTorch.cuda.watched.clear()
|
|
engine = TorchEngine(FakeTorch, model=None, tokenizer=None, metadata={}, settings=Settings(api_token="t" * 40),
|
|
direct_fn=failing_scorer, shared_fn=failing_scorer)
|
|
with pytest.raises(OutOfMemory) as info:
|
|
getattr(engine, call)({})
|
|
assert str(info.value) == "CUDA out of memory. Tried to allocate 490.00 MiB."
|
|
assert info.value.__cause__ is None and info.value.__context__ is None # no chain back to the frames
|
|
assert FakeTorch.cuda.watched[0]() is None # nothing keeps the tensor alive
|
|
assert FakeTorch.cuda.empties == [True] # emptied once, after it was freed
|
|
|
|
|
|
def test_other_scorer_errors_pass_through_unchanged():
|
|
def bad(*_args):
|
|
raise ValueError("Row r: 5000 input tokens exceed limit 4096")
|
|
engine = TorchEngine(FakeTorch, None, None, {}, Settings(api_token="t" * 40), direct_fn=bad, shared_fn=bad)
|
|
with pytest.raises(ValueError, match="exceed limit"):
|
|
engine.direct({})
|
|
|
|
|
|
class ReservingTorch(FakeTorch):
|
|
class cuda(FakeTorch.cuda):
|
|
reserved = 0
|
|
emptied = 0
|
|
|
|
@classmethod
|
|
def memory_reserved(cls, _device=0):
|
|
return cls.reserved
|
|
|
|
@classmethod
|
|
def empty_cache(cls):
|
|
cls.emptied += 1
|
|
|
|
|
|
@pytest.mark.parametrize("reserved_after, released", [(8 * 2**30, 0), (8 * 2**30 + 512 * 2**20, 0),
|
|
(8 * 2**30 + 513 * 2**20, 1), (12 * 2**30, 1)])
|
|
def test_a_burst_is_returned_to_the_driver_after_the_call(reserved_after, released):
|
|
ReservingTorch.cuda.emptied = 0
|
|
|
|
def scorer(*_args):
|
|
ReservingTorch.cuda.reserved = reserved_after
|
|
return {"ok": True}
|
|
|
|
engine = TorchEngine(ReservingTorch, None, None, {}, Settings(api_token="t" * 40),
|
|
direct_fn=scorer, shared_fn=scorer, release_above_bytes=8 * 2**30 + 512 * 2**20)
|
|
assert engine.direct({}) == {"ok": True}
|
|
assert ReservingTorch.cuda.emptied == released
|