Order averaging (Prime, after the 739aa03 spike):
- A decision may set orderings: rotations|all (all only for <= 4 options). Every
ordering goes to the engine in one shared batch.
- The reply keeps each native result and adds combined {probabilities (log-mean),
top, agreement, spread}.
- Through the service on SemIf's labelled sets (252 rows): 78.6% -> 88.1%
(group-bootstrap 95% CI +5.1..+14.3). Unanimous agreement is 94.5% accurate.
Fast kernels: flash-linear-attention 0.5.2 and causal-conv1d 1.7.0 are now the
default build. A/B on the empty GPU 3:
- parity with upstream went from 142/144 to 144/144;
- a ~2k-token /decide went from 169 to 92 ms server-side;
- short 3-rotation batches cost ~3-6 ms more.
triton builds a C shim at runtime, so the image carries gcc. Without it the
warm-up failed and startup failed closed.
Heid bug-hunt panel (4/4 arms, thread 01M3H3F4RR7XBP90KQ3A39H4SX), folded:
- Startup validation: VRAM cap 0 no longer means uncapped (C1); limits must be
>= 1 (S1); the token must be visible ASCII (S2); the calibration file must
exist and parse, with T in [0.05, 20] (S8, and C3's NaN leg).
- The body limit is checked before a chunk is kept, and a Unicode-digit
Content-Length no longer crashes (C2, S3).
- Failures while building the response now get the 500 envelope (C3).
- 429 busy past SEMIF_MAX_QUEUE requests in progress (C6).
- The engine releases memory on every non-validation failure, unchained after
gc; an empty OOM message is handled; 'out of memory' RuntimeErrors map to 503
(C4, C5, S9).
- The entry point forces HF_HUB_OFFLINE (S10). README wording fixed (S5, S6).
- New guard tests close the gaps the arms' mutation grids exposed: early stop of
the body read, a shared-route lock, calibration pass-through, the gc cycle,
the exact caps, TorchEngine.load's arch and device checks, and the offline
entry point.
86 tests.
Deployed on fv-ml1 GPU 1: parity 144/144, OOM and burst release verified, shared
capacity 63/51/26/16 rows at ~140/520/1960/3900 prefix tokens.
122 lines
6.1 KiB
Python
122 lines
6.1 KiB
Python
"""The real engine: SemIf's torch scorers over one resident model. Needs the `model` extra.
|
|
|
|
Contract: semif-serve.contract.md, INV-3 (fail-closed startup), INV-4 (VRAM cap + OOM),
|
|
INV-5 (offline weights). load() is checked on the card at acceptance; the OOM path is
|
|
unit-tested against a fake torch (tests/test_engine.py).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import gc
|
|
import logging
|
|
import traceback
|
|
from typing import Any, Callable
|
|
|
|
from .config import Settings
|
|
from .errors import OutOfMemory, ScoringFailed
|
|
|
|
log = logging.getLogger("semif_serve.engine")
|
|
|
|
RELEASE_SLACK_BYTES = 512 * 2**20
|
|
WARMUP_ROW = {
|
|
"id": "semif-serve-warmup",
|
|
"state": "The deployment completed at 14:02 UTC. Health checks passed in all three zones.",
|
|
"question": "Is there evidence that the deployment succeeded?",
|
|
"options": [
|
|
{"id": "yes", "description": "The deployment succeeded."},
|
|
{"id": "no", "description": "The deployment did not succeed."},
|
|
],
|
|
}
|
|
|
|
|
|
def _first_line(exc: BaseException) -> str:
|
|
lines = str(exc).splitlines()
|
|
return lines[0] if lines else ""
|
|
|
|
|
|
class TorchEngine:
|
|
def __init__(self, torch: Any, model: Any, tokenizer: Any, metadata: dict, settings: Settings,
|
|
direct_fn: Callable, shared_fn: Callable, release_above_bytes: int | None = None):
|
|
self._torch, self._model, self._tokenizer = torch, model, tokenizer
|
|
self._metadata, self._settings = metadata, settings
|
|
self._direct, self._shared = direct_fn, shared_fn
|
|
self._release_above = release_above_bytes
|
|
|
|
@classmethod
|
|
def load(cls, settings: Settings) -> "TorchEngine":
|
|
import torch
|
|
from semif_phase1.core import load_causal_model
|
|
from semif_phase1.direct import score
|
|
from semif_phase1.shared import score_shared
|
|
|
|
if settings.device == "cuda":
|
|
if not torch.cuda.is_available():
|
|
raise RuntimeError("SEMIF_DEVICE=cuda but torch sees no CUDA device")
|
|
major, minor = torch.cuda.get_device_capability(0)
|
|
arch = f"sm_{major}{minor}"
|
|
if arch not in torch.cuda.get_arch_list(): # INV-3: no silent PTX/CPU fallback
|
|
raise RuntimeError(f"torch {torch.__version__} has no kernels for {arch}: {torch.cuda.get_arch_list()}")
|
|
if settings.vram_cap_gib is not None: # INV-4: cap BEFORE the weights land
|
|
total = torch.cuda.get_device_properties(0).total_memory
|
|
fraction = settings.vram_cap_gib * 2**30 / total
|
|
if not 0 < fraction <= 1:
|
|
raise ValueError(f"SEMIF_VRAM_CAP_GIB={settings.vram_cap_gib} does not fit a {total / 2**30:.1f} GiB card")
|
|
torch.cuda.set_per_process_memory_fraction(fraction, 0)
|
|
elif settings.device != "cpu":
|
|
raise ValueError(f"SEMIF_DEVICE must be cuda or cpu, not {settings.device!r}")
|
|
|
|
model, tokenizer, metadata = load_causal_model(settings.model, settings.revision, settings.device, "bfloat16")
|
|
placed = next(model.parameters()).device.type
|
|
if placed != settings.device: # INV-3
|
|
raise RuntimeError(f"model landed on {placed}, expected {settings.device}")
|
|
engine = cls(torch, model, tokenizer, metadata, settings, direct_fn=score, shared_fn=score_shared)
|
|
engine.direct(WARMUP_ROW) # INV-3: one decision must score
|
|
if settings.device == "cuda": # INV-4: the resting footprint
|
|
engine._release_above = torch.cuda.memory_reserved(0) + RELEASE_SLACK_BYTES
|
|
return engine
|
|
|
|
def health(self) -> dict:
|
|
info = dict(self._metadata)
|
|
if self._settings.device == "cuda":
|
|
info["device_name"] = self._torch.cuda.get_device_name(0)
|
|
info["allocated_gib"] = round(self._torch.cuda.memory_allocated(0) / 2**30, 2)
|
|
info["reserved_gib"] = round(self._torch.cuda.memory_reserved(0) / 2**30, 2)
|
|
return info
|
|
|
|
def _release_burst(self) -> None:
|
|
"""INV-4: hand a burst back to the driver so the card's shared headroom (scriberr, the
|
|
vLLM seats) returns after a big request, instead of sitting in torch's cache."""
|
|
if self._release_above is not None and self._torch.cuda.memory_reserved(0) > self._release_above:
|
|
self._torch.cuda.empty_cache()
|
|
|
|
def _guard(self, fn, *args):
|
|
try:
|
|
result = fn(*args)
|
|
except ValueError:
|
|
raise # validation: SemIf raises it before any GPU work
|
|
except self._torch.cuda.OutOfMemoryError as exc:
|
|
failure, message = OutOfMemory, _first_line(exc) or "CUDA out of memory"
|
|
except Exception as exc: # noqa: BLE001 — every other failure is released and reported below
|
|
message = _first_line(exc)
|
|
if "out of memory" in message.lower(): # cuBLAS/cuDNN allocation failures
|
|
failure = OutOfMemory
|
|
else:
|
|
failure, message = ScoringFailed, f"{type(exc).__name__}: {message}"
|
|
# Formatted text, not exc_info: a log record that keeps the traceback object alive
|
|
# (pytest's capture handler does; so would any buffering handler) pins the tensors.
|
|
log.error("scorer failed:\n%s", traceback.format_exc())
|
|
else:
|
|
self._release_burst()
|
|
return result
|
|
# INV-4, outside the except block on purpose: the exception's traceback holds the failed
|
|
# scorer's frames, and with them its tensors (the replicated prefix cache). Raising inside
|
|
# the block, or `from exc`, would chain to it and keep GiBs allocated after the response.
|
|
gc.collect()
|
|
self._torch.cuda.empty_cache()
|
|
raise failure(message)
|
|
|
|
def direct(self, row: dict) -> dict:
|
|
return self._guard(self._direct, self._model, self._tokenizer, row, self._metadata, self._settings.max_tokens)
|
|
|
|
def shared(self, rows: list[dict]) -> tuple[list[dict], dict]:
|
|
return self._guard(self._shared, self._model, self._tokenizer, rows, self._metadata, self._settings.max_tokens)
|