feat(semif): SemIf option-logit decisions on fv-ml1 GPU 1 (Prime)

services/semif-serve is a FastAPI wrapper around SemIf's direct and shared torch
scorers (SemIf-OpenJev @ 23cf1f39, MIT). Upstream ships only a batch CLI. The
wrapper loads the pinned Qwen3.5-4B (851bf6e8, BF16) once from the offline HF
cache and returns SemIf's result dicts unchanged, with an optional per-workload
temperature-calibrated view. Contract: semif-serve.contract.md. Built with a
short contract, TDD (39 tests, fake engine and fake torch, no GPU) and a heid
bug-hunt panel (pending).

On the card:
- torch 2.10.0+cu128 with sm_120 kernels, which is SemIf's own stack;
- a hard 12 GiB VRAM cap.
Two defects surfaced only on the card, and each fix is covered by a test:
- 0.1.1: an OOM raised as a chained exception kept the failed request's tensors
  alive (11.9 GiB after the 503). It is now raised unchained, after gc.
- 0.1.2: a large request left 12.6 GB reserved on the shared card. After each
  call, reserved memory over the baseline + 512 MiB is now released.

Acceptance against SemIf's committed torch predictions (authored144):
- 142/144 same top choice; both misses are exact bf16 ties;
- 144/144 identical prompt hashes;
- deterministic A-vs-A;
- negative control 14/144;
- shared vs direct 72/72.
21 binary criteria over one state take 159 ms. The shared-mode capacity table
under the cap is in stacks/semif/README.md.

The Dockerfile installs dependencies from a manifest with the project version
blanked, so a version bump reuses the ~4 GB torch layer. Verified: 41 s rebuild,
dependency layer CACHED.

DNS: semif.fv.internal. Token: vault semif/api-token.
This commit is contained in:
vh
2026-09-27 02:36:56 -07:00
parent 30f2c977b7
commit 069725c4b3
24 changed files with 2487 additions and 1 deletions
@@ -0,0 +1,101 @@
"""The real engine: SemIf's torch scorers over one resident model. Needs the `model` extra.
Contract: semif-serve.contract.md, INV-3 (fail-closed startup), INV-4 (VRAM cap + OOM),
INV-5 (offline weights). load() is checked on the card at acceptance; the OOM path is
unit-tested against a fake torch (tests/test_engine.py).
"""
from __future__ import annotations
import gc
from typing import Any, Callable
from .config import Settings
from .errors import OutOfMemory
RELEASE_SLACK_BYTES = 512 * 2**20
WARMUP_ROW = {
"id": "semif-serve-warmup",
"state": "The deployment completed at 14:02 UTC. Health checks passed in all three zones.",
"question": "Is there evidence that the deployment succeeded?",
"options": [
{"id": "yes", "description": "The deployment succeeded."},
{"id": "no", "description": "The deployment did not succeed."},
],
}
class TorchEngine:
def __init__(self, torch: Any, model: Any, tokenizer: Any, metadata: dict, settings: Settings,
direct_fn: Callable, shared_fn: Callable, release_above_bytes: int | None = None):
self._torch, self._model, self._tokenizer = torch, model, tokenizer
self._metadata, self._settings = metadata, settings
self._direct, self._shared = direct_fn, shared_fn
self._release_above = release_above_bytes
@classmethod
def load(cls, settings: Settings) -> "TorchEngine":
import torch
from semif_phase1.core import load_causal_model
from semif_phase1.direct import score
from semif_phase1.shared import score_shared
if settings.device == "cuda":
if not torch.cuda.is_available():
raise RuntimeError("SEMIF_DEVICE=cuda but torch sees no CUDA device")
major, minor = torch.cuda.get_device_capability(0)
arch = f"sm_{major}{minor}"
if arch not in torch.cuda.get_arch_list(): # INV-3: no silent PTX/CPU fallback
raise RuntimeError(f"torch {torch.__version__} has no kernels for {arch}: {torch.cuda.get_arch_list()}")
if settings.vram_cap_gib: # INV-4: cap BEFORE the weights land
total = torch.cuda.get_device_properties(0).total_memory
fraction = settings.vram_cap_gib * 2**30 / total
if not 0 < fraction <= 1:
raise ValueError(f"SEMIF_VRAM_CAP_GIB={settings.vram_cap_gib} does not fit a {total / 2**30:.1f} GiB card")
torch.cuda.set_per_process_memory_fraction(fraction, 0)
elif settings.device != "cpu":
raise ValueError(f"SEMIF_DEVICE must be cuda or cpu, not {settings.device!r}")
model, tokenizer, metadata = load_causal_model(settings.model, settings.revision, settings.device, "bfloat16")
placed = next(model.parameters()).device.type
if placed != settings.device: # INV-3
raise RuntimeError(f"model landed on {placed}, expected {settings.device}")
engine = cls(torch, model, tokenizer, metadata, settings, direct_fn=score, shared_fn=score_shared)
engine.direct(WARMUP_ROW) # INV-3: one decision must score
if settings.device == "cuda": # INV-4: the resting footprint
engine._release_above = torch.cuda.memory_reserved(0) + RELEASE_SLACK_BYTES
return engine
def health(self) -> dict:
info = dict(self._metadata)
if self._settings.device == "cuda":
info["device_name"] = self._torch.cuda.get_device_name(0)
info["allocated_gib"] = round(self._torch.cuda.memory_allocated(0) / 2**30, 2)
info["reserved_gib"] = round(self._torch.cuda.memory_reserved(0) / 2**30, 2)
return info
def _release_burst(self) -> None:
"""INV-4: hand a burst back to the driver so the card's shared headroom (scriberr, the
vLLM seats) returns after a big request, instead of sitting in torch's cache."""
if self._release_above is not None and self._torch.cuda.memory_reserved(0) > self._release_above:
self._torch.cuda.empty_cache()
def _guard(self, fn, *args):
try:
result = fn(*args)
except self._torch.cuda.OutOfMemoryError as exc:
message = str(exc).splitlines()[0]
else:
self._release_burst()
return result
# INV-4, outside the except block on purpose: the torch exception's traceback holds the
# failed scorer's frames, and with them its tensors (the replicated prefix cache). Raising
# inside the block, or `from exc`, would chain to it and keep GiBs allocated after the 503.
gc.collect()
self._torch.cuda.empty_cache()
raise OutOfMemory(message)
def direct(self, row: dict) -> dict:
return self._guard(self._direct, self._model, self._tokenizer, row, self._metadata, self._settings.max_tokens)
def shared(self, rows: list[dict]) -> tuple[list[dict], dict]:
return self._guard(self._shared, self._model, self._tokenizer, rows, self._metadata, self._settings.max_tokens)