"""The real engine: SemIf's torch scorers over one resident model. Needs the `model` extra. Contract: semif-serve.contract.md, INV-3 (fail-closed startup), INV-4 (VRAM cap + OOM), INV-5 (offline weights). load() is checked on the card at acceptance; the OOM path is unit-tested against a fake torch (tests/test_engine.py). """ from __future__ import annotations import gc import logging import traceback from typing import Any, Callable from .config import Settings from .errors import OutOfMemory, ScoringFailed log = logging.getLogger("semif_serve.engine") RELEASE_SLACK_BYTES = 512 * 2**20 WARMUP_ROW = { "id": "semif-serve-warmup", "state": "The deployment completed at 14:02 UTC. Health checks passed in all three zones.", "question": "Is there evidence that the deployment succeeded?", "options": [ {"id": "yes", "description": "The deployment succeeded."}, {"id": "no", "description": "The deployment did not succeed."}, ], } def _first_line(exc: BaseException) -> str: lines = str(exc).splitlines() return lines[0] if lines else "" class TorchEngine: def __init__(self, torch: Any, model: Any, tokenizer: Any, metadata: dict, settings: Settings, direct_fn: Callable, shared_fn: Callable, release_above_bytes: int | None = None): self._torch, self._model, self._tokenizer = torch, model, tokenizer self._metadata, self._settings = metadata, settings self._direct, self._shared = direct_fn, shared_fn self._release_above = release_above_bytes @classmethod def load(cls, settings: Settings) -> "TorchEngine": import torch from semif_phase1.core import load_causal_model from semif_phase1.direct import score from semif_phase1.shared import score_shared if settings.device == "cuda": if not torch.cuda.is_available(): raise RuntimeError("SEMIF_DEVICE=cuda but torch sees no CUDA device") major, minor = torch.cuda.get_device_capability(0) arch = f"sm_{major}{minor}" if arch not in torch.cuda.get_arch_list(): # INV-3: no silent PTX/CPU fallback raise RuntimeError(f"torch {torch.__version__} has no kernels for {arch}: {torch.cuda.get_arch_list()}") if settings.vram_cap_gib is not None: # INV-4: cap BEFORE the weights land total = torch.cuda.get_device_properties(0).total_memory fraction = settings.vram_cap_gib * 2**30 / total if not 0 < fraction <= 1: raise ValueError(f"SEMIF_VRAM_CAP_GIB={settings.vram_cap_gib} does not fit a {total / 2**30:.1f} GiB card") torch.cuda.set_per_process_memory_fraction(fraction, 0) elif settings.device != "cpu": raise ValueError(f"SEMIF_DEVICE must be cuda or cpu, not {settings.device!r}") model, tokenizer, metadata = load_causal_model(settings.model, settings.revision, settings.device, "bfloat16") placed = next(model.parameters()).device.type if placed != settings.device: # INV-3 raise RuntimeError(f"model landed on {placed}, expected {settings.device}") engine = cls(torch, model, tokenizer, metadata, settings, direct_fn=score, shared_fn=score_shared) engine.direct(WARMUP_ROW) # INV-3: one decision must score if settings.device == "cuda": # INV-4: the resting footprint engine._release_above = torch.cuda.memory_reserved(0) + RELEASE_SLACK_BYTES return engine def health(self) -> dict: info = dict(self._metadata) if self._settings.device == "cuda": info["device_name"] = self._torch.cuda.get_device_name(0) info["allocated_gib"] = round(self._torch.cuda.memory_allocated(0) / 2**30, 2) info["reserved_gib"] = round(self._torch.cuda.memory_reserved(0) / 2**30, 2) return info def _release_burst(self) -> None: """INV-4: hand a burst back to the driver so the card's shared headroom (scriberr, the vLLM seats) returns after a big request, instead of sitting in torch's cache.""" if self._release_above is not None and self._torch.cuda.memory_reserved(0) > self._release_above: self._torch.cuda.empty_cache() def _guard(self, fn, *args): try: result = fn(*args) except ValueError: raise # validation: SemIf raises it before any GPU work except self._torch.cuda.OutOfMemoryError as exc: failure, message = OutOfMemory, _first_line(exc) or "CUDA out of memory" except Exception as exc: # noqa: BLE001 — every other failure is released and reported below message = _first_line(exc) if "out of memory" in message.lower(): # cuBLAS/cuDNN allocation failures failure = OutOfMemory else: failure, message = ScoringFailed, f"{type(exc).__name__}: {message}" # Formatted text, not exc_info: a log record that keeps the traceback object alive # (pytest's capture handler does; so would any buffering handler) pins the tensors. log.error("scorer failed:\n%s", traceback.format_exc()) else: self._release_burst() return result # INV-4, outside the except block on purpose: the exception's traceback holds the failed # scorer's frames, and with them its tensors (the replicated prefix cache). Raising inside # the block, or `from exc`, would chain to it and keep GiBs allocated after the response. gc.collect() self._torch.cuda.empty_cache() raise failure(message) def direct(self, row: dict) -> dict: return self._guard(self._direct, self._model, self._tokenizer, row, self._metadata, self._settings.max_tokens) def shared(self, rows: list[dict]) -> tuple[list[dict], dict]: return self._guard(self._shared, self._model, self._tokenizer, rows, self._metadata, self._settings.max_tokens)