services/semif-serve is a FastAPI wrapper around SemIf's direct and shared torch scorers (SemIf-OpenJev @ 23cf1f39, MIT). Upstream ships only a batch CLI. The wrapper loads the pinned Qwen3.5-4B (851bf6e8, BF16) once from the offline HF cache and returns SemIf's result dicts unchanged, with an optional per-workload temperature-calibrated view. Contract: semif-serve.contract.md. Built with a short contract, TDD (39 tests, fake engine and fake torch, no GPU) and a heid bug-hunt panel (pending). On the card: - torch 2.10.0+cu128 with sm_120 kernels, which is SemIf's own stack; - a hard 12 GiB VRAM cap. Two defects surfaced only on the card, and each fix is covered by a test: - 0.1.1: an OOM raised as a chained exception kept the failed request's tensors alive (11.9 GiB after the 503). It is now raised unchained, after gc. - 0.1.2: a large request left 12.6 GB reserved on the shared card. After each call, reserved memory over the baseline + 512 MiB is now released. Acceptance against SemIf's committed torch predictions (authored144): - 142/144 same top choice; both misses are exact bf16 ties; - 144/144 identical prompt hashes; - deterministic A-vs-A; - negative control 14/144; - shared vs direct 72/72. 21 binary criteria over one state take 159 ms. The shared-mode capacity table under the cap is in stacks/semif/README.md. The Dockerfile installs dependencies from a manifest with the project version blanked, so a version bump reuses the ~4 GB torch layer. Verified: 41 s rebuild, dependency layer CACHED. DNS: semif.fv.internal. Token: vault semif/api-token.
162 lines
6.3 KiB
Python
162 lines
6.3 KiB
Python
"""semif-serve HTTP layer. Contract: semif-serve.contract.md."""
|
|
from __future__ import annotations
|
|
|
|
import hmac
|
|
import math
|
|
import threading
|
|
from typing import Any
|
|
|
|
from fastapi import FastAPI, Request
|
|
from fastapi.concurrency import run_in_threadpool
|
|
from fastapi.responses import JSONResponse
|
|
from pydantic import BaseModel, ConfigDict, ValidationError
|
|
|
|
from .config import SEMIF_COMMIT, Settings
|
|
from .errors import OutOfMemory
|
|
|
|
OPEN_PATHS = frozenset({"/health"})
|
|
State = str | dict[str, Any] | list[Any]
|
|
|
|
|
|
class Option(BaseModel):
|
|
model_config = ConfigDict(extra="ignore")
|
|
id: str
|
|
description: str
|
|
|
|
|
|
class Decision(BaseModel):
|
|
model_config = ConfigDict(extra="ignore")
|
|
id: str
|
|
question: str
|
|
options: list[Option]
|
|
|
|
def row(self, state: State) -> dict:
|
|
"""The SemIf row shape: exactly id, state, question, options."""
|
|
return {"id": self.id, "state": state, "question": self.question,
|
|
"options": [o.model_dump() for o in self.options]}
|
|
|
|
|
|
class DecideBody(Decision):
|
|
state: State
|
|
workload: str | None = None
|
|
|
|
|
|
class SharedBody(BaseModel):
|
|
model_config = ConfigDict(extra="ignore")
|
|
state: State
|
|
decisions: list[Decision]
|
|
workload: str | None = None
|
|
|
|
|
|
class ApiError(Exception):
|
|
def __init__(self, status: int, code: str, message: str):
|
|
super().__init__(message)
|
|
self.status, self.code, self.message = status, code, message
|
|
|
|
|
|
def error(status: int, code: str, message: str) -> JSONResponse:
|
|
return JSONResponse(status_code=status, content={"error": {"code": code, "message": message}})
|
|
|
|
|
|
def _first_error(exc: ValidationError) -> str:
|
|
first = exc.errors()[0]
|
|
where = ".".join(str(p) for p in first.get("loc", ())) or "body"
|
|
return f"{where}: {first.get('msg', 'invalid')}"
|
|
|
|
|
|
def calibrated_view(result: dict, workload: str, temperature: float) -> dict:
|
|
"""softmax(option_logits / T): the native fields are left exactly as SemIf returned them (INV-1)."""
|
|
scaled = [x / temperature for x in result["option_logits"]]
|
|
top = max(scaled)
|
|
weights = [math.exp(x - top) for x in scaled]
|
|
total = sum(weights)
|
|
return {"workload": workload, "temperature": temperature, "probabilities": [w / total for w in weights]}
|
|
|
|
|
|
def create_app(settings: Settings, engine: Any) -> FastAPI:
|
|
app = FastAPI(title="semif-serve")
|
|
expected = f"Bearer {settings.api_token}".encode()
|
|
inference = threading.Lock() # INV-2: one scorer call at a time, off the event loop
|
|
|
|
def locked(fn, *args):
|
|
with inference:
|
|
return fn(*args)
|
|
|
|
@app.middleware("http")
|
|
async def require_bearer(request: Request, call_next):
|
|
if request.url.path not in OPEN_PATHS:
|
|
supplied = request.headers.get("authorization", "").encode()
|
|
if not hmac.compare_digest(supplied, expected): # INV-6
|
|
return error(401, "unauthorized", "missing or wrong bearer token")
|
|
return await call_next(request)
|
|
|
|
@app.exception_handler(ApiError)
|
|
async def api_error(_request: Request, exc: ApiError):
|
|
return error(exc.status, exc.code, exc.message)
|
|
|
|
async def read_limited(request: Request) -> bytes:
|
|
limit = settings.max_body_bytes
|
|
too_large = ApiError(413, "request_too_large", f"request body exceeds {limit} bytes")
|
|
declared = request.headers.get("content-length")
|
|
if declared is not None and declared.isdigit() and int(declared) > limit:
|
|
raise too_large
|
|
body = bytearray()
|
|
async for chunk in request.stream(): # also caps bodies that declare no length
|
|
body.extend(chunk)
|
|
if len(body) > limit:
|
|
raise too_large
|
|
return bytes(body)
|
|
|
|
async def parse(request: Request, model: type[BaseModel]):
|
|
try:
|
|
return model.model_validate_json(await read_limited(request))
|
|
except ValidationError as exc:
|
|
raise ApiError(422, "invalid_request", _first_error(exc)) from exc
|
|
|
|
async def score(fn, *args):
|
|
"""Run one scorer call in a worker thread under the lock; map its failures to contract codes."""
|
|
try:
|
|
return await run_in_threadpool(locked, fn, *args)
|
|
except ValueError as exc: # SemIf validation, token limit, tokenisation
|
|
raise ApiError(422, "invalid_request", str(exc)) from exc
|
|
except OutOfMemory as exc:
|
|
raise ApiError(503, "out_of_memory", str(exc)) from exc
|
|
except Exception as exc: # noqa: BLE001 — any other scorer failure
|
|
raise ApiError(500, "scoring_failed", f"{type(exc).__name__}: {exc}") from exc
|
|
|
|
def temperature_for(workload: str | None) -> float | None:
|
|
if workload is None:
|
|
return None
|
|
if workload not in settings.calibration:
|
|
raise ApiError(422, "invalid_request", f"unknown workload {workload!r}")
|
|
return settings.calibration[workload]
|
|
|
|
def with_calibration(result: dict, workload: str | None, temperature: float | None) -> dict:
|
|
if temperature is None:
|
|
return result
|
|
return {**result, "calibrated": calibrated_view(result, workload, temperature)}
|
|
|
|
@app.get("/health")
|
|
async def health():
|
|
return {"status": "ok", "semif_commit": SEMIF_COMMIT, "model": engine.health(),
|
|
"vram_cap_gib": settings.vram_cap_gib, "max_tokens": settings.max_tokens,
|
|
"max_decisions": settings.max_decisions, "workloads": sorted(settings.calibration)}
|
|
|
|
@app.post("/decide")
|
|
async def decide(request: Request):
|
|
body = await parse(request, DecideBody)
|
|
temperature = temperature_for(body.workload)
|
|
return with_calibration(await score(engine.direct, body.row(body.state)), body.workload, temperature)
|
|
|
|
@app.post("/decide/shared")
|
|
async def decide_shared(request: Request):
|
|
body = await parse(request, SharedBody)
|
|
if not 1 <= len(body.decisions) <= settings.max_decisions:
|
|
raise ApiError(422, "invalid_request",
|
|
f"decisions must hold 1..{settings.max_decisions} entries, got {len(body.decisions)}")
|
|
temperature = temperature_for(body.workload)
|
|
results, timing = await score(engine.shared, [d.row(body.state) for d in body.decisions])
|
|
return {"results": [with_calibration(r, body.workload, temperature) for r in results], "timing": timing}
|
|
|
|
return app
|