feat(semif): SemIf option-logit decisions on fv-ml1 GPU 1 (Prime)
services/semif-serve is a FastAPI wrapper around SemIf's direct and shared torch scorers (SemIf-OpenJev @ 23cf1f39, MIT). Upstream ships only a batch CLI. The wrapper loads the pinned Qwen3.5-4B (851bf6e8, BF16) once from the offline HF cache and returns SemIf's result dicts unchanged, with an optional per-workload temperature-calibrated view. Contract: semif-serve.contract.md. Built with a short contract, TDD (39 tests, fake engine and fake torch, no GPU) and a heid bug-hunt panel (pending). On the card: - torch 2.10.0+cu128 with sm_120 kernels, which is SemIf's own stack; - a hard 12 GiB VRAM cap. Two defects surfaced only on the card, and each fix is covered by a test: - 0.1.1: an OOM raised as a chained exception kept the failed request's tensors alive (11.9 GiB after the 503). It is now raised unchained, after gc. - 0.1.2: a large request left 12.6 GB reserved on the shared card. After each call, reserved memory over the baseline + 512 MiB is now released. Acceptance against SemIf's committed torch predictions (authored144): - 142/144 same top choice; both misses are exact bf16 ties; - 144/144 identical prompt hashes; - deterministic A-vs-A; - negative control 14/144; - shared vs direct 72/72. 21 binary criteria over one state take 159 ms. The shared-mode capacity table under the cap is in stacks/semif/README.md. The Dockerfile installs dependencies from a manifest with the project version blanked, so a version bump reuses the ~4 GB torch layer. Verified: 41 s rebuild, dependency layer CACHED. DNS: semif.fv.internal. Token: vault semif/api-token.
This commit is contained in:
@@ -0,0 +1,57 @@
|
||||
"""Settings for semif-serve. Contract: semif-serve.contract.md § Configuration."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
MIN_TOKEN_CHARS = 32
|
||||
SEMIF_COMMIT = "23cf1f39fc9534fe81437200959b6dfc7106e45a"
|
||||
DEFAULT_MODEL = "Qwen/Qwen3.5-4B"
|
||||
DEFAULT_REVISION = "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Settings:
|
||||
api_token: str
|
||||
model: str = DEFAULT_MODEL
|
||||
revision: str = DEFAULT_REVISION
|
||||
device: str = "cuda"
|
||||
vram_cap_gib: float | None = None
|
||||
max_tokens: int = 4096
|
||||
max_decisions: int = 64
|
||||
max_body_bytes: int = 1024 * 1024
|
||||
calibration: dict[str, float] = field(default_factory=dict)
|
||||
|
||||
@classmethod
|
||||
def from_env(cls, env: Mapping[str, str]) -> "Settings":
|
||||
token = env.get("SEMIF_API_TOKEN", "")
|
||||
if len(token) < MIN_TOKEN_CHARS: # INV-6
|
||||
raise ValueError(f"SEMIF_API_TOKEN must be at least {MIN_TOKEN_CHARS} characters")
|
||||
cap = env.get("SEMIF_VRAM_CAP_GIB")
|
||||
return cls(
|
||||
api_token=token,
|
||||
model=env.get("SEMIF_MODEL", DEFAULT_MODEL),
|
||||
revision=env.get("SEMIF_REVISION", DEFAULT_REVISION),
|
||||
device=env.get("SEMIF_DEVICE", "cuda"),
|
||||
vram_cap_gib=float(cap) if cap else None,
|
||||
max_tokens=int(env.get("SEMIF_MAX_TOKENS", 4096)),
|
||||
max_decisions=int(env.get("SEMIF_MAX_DECISIONS", 64)),
|
||||
max_body_bytes=int(env.get("SEMIF_MAX_BODY_BYTES", 1024 * 1024)),
|
||||
calibration=_load_calibration(env.get("SEMIF_CALIBRATION")),
|
||||
)
|
||||
|
||||
|
||||
def _load_calibration(path: str | None) -> dict[str, float]:
|
||||
"""{workload: T}, every T a finite number > 0 (T scales option logits before softmax)."""
|
||||
if not path:
|
||||
return {}
|
||||
table = json.loads(Path(path).read_text())
|
||||
if not isinstance(table, dict) or not all(
|
||||
isinstance(t, (int, float)) and not isinstance(t, bool) and math.isfinite(t) and t > 0
|
||||
for t in table.values()
|
||||
):
|
||||
raise ValueError("SEMIF_CALIBRATION must be a JSON object of workload -> finite temperature > 0")
|
||||
return {str(k): float(v) for k, v in table.items()}
|
||||
Reference in New Issue
Block a user