feat(semif): 0.1.3 — order averaging, fast kernels, bug-hunt hardening (Prime)
Order averaging (Prime, after the 739aa03 spike):
- A decision may set orderings: rotations|all (all only for <= 4 options). Every
ordering goes to the engine in one shared batch.
- The reply keeps each native result and adds combined {probabilities (log-mean),
top, agreement, spread}.
- Through the service on SemIf's labelled sets (252 rows): 78.6% -> 88.1%
(group-bootstrap 95% CI +5.1..+14.3). Unanimous agreement is 94.5% accurate.
Fast kernels: flash-linear-attention 0.5.2 and causal-conv1d 1.7.0 are now the
default build. A/B on the empty GPU 3:
- parity with upstream went from 142/144 to 144/144;
- a ~2k-token /decide went from 169 to 92 ms server-side;
- short 3-rotation batches cost ~3-6 ms more.
triton builds a C shim at runtime, so the image carries gcc. Without it the
warm-up failed and startup failed closed.
Heid bug-hunt panel (4/4 arms, thread 01M3H3F4RR7XBP90KQ3A39H4SX), folded:
- Startup validation: VRAM cap 0 no longer means uncapped (C1); limits must be
>= 1 (S1); the token must be visible ASCII (S2); the calibration file must
exist and parse, with T in [0.05, 20] (S8, and C3's NaN leg).
- The body limit is checked before a chunk is kept, and a Unicode-digit
Content-Length no longer crashes (C2, S3).
- Failures while building the response now get the 500 envelope (C3).
- 429 busy past SEMIF_MAX_QUEUE requests in progress (C6).
- The engine releases memory on every non-validation failure, unchained after
gc; an empty OOM message is handled; 'out of memory' RuntimeErrors map to 503
(C4, C5, S9).
- The entry point forces HF_HUB_OFFLINE (S10). README wording fixed (S5, S6).
- New guard tests close the gaps the arms' mutation grids exposed: early stop of
the body read, a shared-route lock, calibration pass-through, the gc cycle,
the exact caps, TorchEngine.load's arch and device checks, and the offline
entry point.
86 tests.
Deployed on fv-ml1 GPU 1: parity 144/144, OOM and burst release verified, shared
capacity 63/51/26/16 rows at ~140/520/1960/3900 prefix tokens.
This commit is contained in:
@@ -1,4 +1,9 @@
|
||||
"""Settings for semif-serve. Contract: semif-serve.contract.md § Configuration."""
|
||||
"""Settings for semif-serve. Contract: semif-serve.contract.md § Configuration.
|
||||
|
||||
Every value is validated at startup and a bad one is refused with a ValueError naming the
|
||||
variable: a service that starts and then rejects every request (or runs uncapped) is worse
|
||||
than one that does not start (bug hunt 2026-09-27: C1, S1, S2, S8).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
@@ -8,6 +13,7 @@ from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
MIN_TOKEN_CHARS = 32
|
||||
MIN_TEMPERATURE, MAX_TEMPERATURE = 0.05, 20.0
|
||||
SEMIF_COMMIT = "23cf1f39fc9534fe81437200959b6dfc7106e45a"
|
||||
DEFAULT_MODEL = "Qwen/Qwen3.5-4B"
|
||||
DEFAULT_REVISION = "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a"
|
||||
@@ -23,35 +29,71 @@ class Settings:
|
||||
max_tokens: int = 4096
|
||||
max_decisions: int = 64
|
||||
max_body_bytes: int = 1024 * 1024
|
||||
max_queue: int = 32
|
||||
calibration: dict[str, float] = field(default_factory=dict)
|
||||
|
||||
@classmethod
|
||||
def from_env(cls, env: Mapping[str, str]) -> "Settings":
|
||||
token = env.get("SEMIF_API_TOKEN", "")
|
||||
if len(token) < MIN_TOKEN_CHARS: # INV-6
|
||||
raise ValueError(f"SEMIF_API_TOKEN must be at least {MIN_TOKEN_CHARS} characters")
|
||||
cap = env.get("SEMIF_VRAM_CAP_GIB")
|
||||
# INV-6: visible ASCII only. A CR, LF or NUL can never arrive in a header, so a token
|
||||
# carrying one would lock every caller out while /health still said ok.
|
||||
if len(token) < MIN_TOKEN_CHARS or not all(33 <= ord(c) <= 126 for c in token):
|
||||
raise ValueError(f"SEMIF_API_TOKEN must be at least {MIN_TOKEN_CHARS} visible ASCII characters")
|
||||
return cls(
|
||||
api_token=token,
|
||||
model=env.get("SEMIF_MODEL", DEFAULT_MODEL),
|
||||
revision=env.get("SEMIF_REVISION", DEFAULT_REVISION),
|
||||
device=env.get("SEMIF_DEVICE", "cuda"),
|
||||
vram_cap_gib=float(cap) if cap else None,
|
||||
max_tokens=int(env.get("SEMIF_MAX_TOKENS", 4096)),
|
||||
max_decisions=int(env.get("SEMIF_MAX_DECISIONS", 64)),
|
||||
max_body_bytes=int(env.get("SEMIF_MAX_BODY_BYTES", 1024 * 1024)),
|
||||
vram_cap_gib=_positive_float(env, "SEMIF_VRAM_CAP_GIB"),
|
||||
max_tokens=_positive_int(env, "SEMIF_MAX_TOKENS", 4096),
|
||||
max_decisions=_positive_int(env, "SEMIF_MAX_DECISIONS", 64),
|
||||
max_body_bytes=_positive_int(env, "SEMIF_MAX_BODY_BYTES", 1024 * 1024),
|
||||
max_queue=_positive_int(env, "SEMIF_MAX_QUEUE", 32),
|
||||
calibration=_load_calibration(env.get("SEMIF_CALIBRATION")),
|
||||
)
|
||||
|
||||
|
||||
def _positive_int(env: Mapping[str, str], name: str, default: int) -> int:
|
||||
raw = env.get(name)
|
||||
if raw is None:
|
||||
return default
|
||||
try:
|
||||
value = int(raw)
|
||||
except ValueError:
|
||||
raise ValueError(f"{name} must be an integer, got {raw!r}") from None
|
||||
if value < 1:
|
||||
raise ValueError(f"{name} must be >= 1, got {value}")
|
||||
return value
|
||||
|
||||
|
||||
def _positive_float(env: Mapping[str, str], name: str) -> float | None:
|
||||
"""Unset means no cap. When set it must be finite and > 0: `0` used to slip through as 'no cap'."""
|
||||
raw = env.get(name)
|
||||
if raw is None or raw == "":
|
||||
return None
|
||||
try:
|
||||
value = float(raw)
|
||||
except ValueError:
|
||||
raise ValueError(f"{name} must be a number, got {raw!r}") from None
|
||||
if not math.isfinite(value) or value <= 0:
|
||||
raise ValueError(f"{name} must be a finite number > 0, got {raw!r}")
|
||||
return value
|
||||
|
||||
|
||||
def _load_calibration(path: str | None) -> dict[str, float]:
|
||||
"""{workload: T}, every T a finite number > 0 (T scales option logits before softmax)."""
|
||||
"""{workload: T}; T scales option logits before softmax, so it is kept in a sane range
|
||||
(a tiny T overflows to NaN and the response then fails to render)."""
|
||||
if not path:
|
||||
return {}
|
||||
table = json.loads(Path(path).read_text())
|
||||
try:
|
||||
table = json.loads(Path(path).read_text())
|
||||
except (OSError, ValueError) as exc:
|
||||
raise ValueError(f"SEMIF_CALIBRATION {path!r} could not be read as JSON: {exc}") from None
|
||||
if not isinstance(table, dict) or not all(
|
||||
isinstance(t, (int, float)) and not isinstance(t, bool) and math.isfinite(t) and t > 0
|
||||
isinstance(t, (int, float)) and not isinstance(t, bool) and math.isfinite(t)
|
||||
and MIN_TEMPERATURE <= t <= MAX_TEMPERATURE
|
||||
for t in table.values()
|
||||
):
|
||||
raise ValueError("SEMIF_CALIBRATION must be a JSON object of workload -> finite temperature > 0")
|
||||
raise ValueError(f"SEMIF_CALIBRATION must be a JSON object of workload -> temperature in "
|
||||
f"[{MIN_TEMPERATURE}, {MAX_TEMPERATURE}]")
|
||||
return {str(k): float(v) for k, v in table.items()}
|
||||
|
||||
Reference in New Issue
Block a user