feat(semif): SemIf option-logit decisions on fv-ml1 GPU 1 (Prime)

services/semif-serve is a FastAPI wrapper around SemIf's direct and shared torch
scorers (SemIf-OpenJev @ 23cf1f39, MIT). Upstream ships only a batch CLI. The
wrapper loads the pinned Qwen3.5-4B (851bf6e8, BF16) once from the offline HF
cache and returns SemIf's result dicts unchanged, with an optional per-workload
temperature-calibrated view. Contract: semif-serve.contract.md. Built with a
short contract, TDD (39 tests, fake engine and fake torch, no GPU) and a heid
bug-hunt panel (pending).

On the card:
- torch 2.10.0+cu128 with sm_120 kernels, which is SemIf's own stack;
- a hard 12 GiB VRAM cap.
Two defects surfaced only on the card, and each fix is covered by a test:
- 0.1.1: an OOM raised as a chained exception kept the failed request's tensors
  alive (11.9 GiB after the 503). It is now raised unchained, after gc.
- 0.1.2: a large request left 12.6 GB reserved on the shared card. After each
  call, reserved memory over the baseline + 512 MiB is now released.

Acceptance against SemIf's committed torch predictions (authored144):
- 142/144 same top choice; both misses are exact bf16 ties;
- 144/144 identical prompt hashes;
- deterministic A-vs-A;
- negative control 14/144;
- shared vs direct 72/72.
21 binary criteria over one state take 159 ms. The shared-mode capacity table
under the cap is in stacks/semif/README.md.

The Dockerfile installs dependencies from a manifest with the project version
blanked, so a version bump reuses the ~4 GB torch layer. Verified: 41 s rebuild,
dependency layer CACHED.

DNS: semif.fv.internal. Token: vault semif/api-token.
This commit is contained in:
vh
2026-09-27 02:36:56 -07:00
parent 30f2c977b7
commit 069725c4b3
24 changed files with 2487 additions and 1 deletions
+6
View File
@@ -0,0 +1,6 @@
*
!pyproject.toml
!uv.lock
!src/
**/__pycache__
src/*.egg-info
+4
View File
@@ -0,0 +1,4 @@
.venv/
.pytest_cache/
__pycache__/
*.egg-info/
+47
View File
@@ -0,0 +1,47 @@
# syntax=docker/dockerfile:1
# semif-serve: SemIf (pinned commit) behind a small FastAPI service. Contract: semif-serve.contract.md.
# docker build -t semif-serve:<version> .
# Weights are NOT in the image: the pinned Qwen3.5-4B revision is read from the mounted
# HF cache, offline (INV-5).
# The dependency manifest with semif-serve's own version blanked to 0.0.0. A version bump
# then leaves these two files byte-identical, and COPY --from compares CONTENT, so the ~4 GB
# torch/CUDA install below stays cached across releases (the same problem augaman hit).
# `uv sync` keeps uv's per-package index routing (torch from the cu128 index, everything
# else from PyPI). An exported requirements.txt loses that, and then fetches triton from the
# wrong index and fails its hash check.
FROM python:3.12-slim-bookworm AS deps
WORKDIR /deps
COPY pyproject.toml uv.lock ./
RUN python - <<'EOF'
import re, pathlib
p = pathlib.Path("pyproject.toml")
p.write_text(re.sub(r'(?m)^version = "[^"]+"', 'version = "0.0.0"', p.read_text(), count=1))
l = pathlib.Path("uv.lock")
l.write_text(re.sub(r'(name = "semif-serve"\nversion = )"[^"]+"', r'\1"0.0.0"', l.read_text(), count=1))
EOF
FROM python:3.12-slim-bookworm
COPY --from=ghcr.io/astral-sh/uv:0.6.9 /uv /bin/uv
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy UV_PYTHON_DOWNLOADS=never
# git: semif-phase1 installs from a pinned GitHub commit.
RUN apt-get update && apt-get install -y --no-install-recommends git ca-certificates \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
COPY --from=deps /deps/pyproject.toml /deps/uv.lock ./
RUN --mount=type=cache,target=/root/.cache/uv \
uv sync --frozen --no-dev --extra model --no-install-project
COPY pyproject.toml uv.lock ./
COPY src ./src
RUN uv sync --frozen --no-dev --extra model --no-editable --no-cache
RUN groupadd --system --gid 10001 semif \
&& useradd --system --uid 10001 --gid 10001 --no-create-home --shell /usr/sbin/nologin semif
USER semif
ENV PATH=/app/.venv/bin:$PATH \
HF_HOME=/hf \
HF_HUB_OFFLINE=1 \
HF_HUB_DISABLE_TELEMETRY=1 \
NVIDIA_DRIVER_CAPABILITIES=compute,utility
EXPOSE 8000
# One worker (INV-2): the model and the inference lock live in this one process.
CMD ["uvicorn", "semif_serve.main:app_from_env", "--factory", "--host", "0.0.0.0", "--port", "8000", "--workers", "1"]
+101
View File
@@ -0,0 +1,101 @@
"""semif-serve acceptance on the real card. Contract § Acceptance.
Needs SemIf's authored144 rows and their committed torch predictions (same model revision,
torch 2.10.0+cu128) from a SemIf checkout at the pinned commit:
SEMIF_DIR=<checkout> SEMIF_URL=http://10.251.50.54:8032 SEMIF_TOKEN=... \
uv run --with httpx python accept.py out.json
1 parity our /decide vs their direct-authored144 predictions: top choice, max |Δp|, prompt hash
2 noise floor the same 144 rows again (A-vs-A)
3 negative option descriptions rotated one place: agreement with the reference MUST drop
4 shared one state, many criteria through /decide/shared vs /decide on the same rows
5 speed 21 binary criteria over one state, shared vs 21 sequential /decide, 3 runs, after warm-up
"""
import json
import os
import statistics
import sys
import time
from pathlib import Path
import httpx
URL, TOKEN, SEMIF = os.environ["SEMIF_URL"], os.environ["SEMIF_TOKEN"], Path(os.environ["SEMIF_DIR"])
H = {"Authorization": f"Bearer {TOKEN}"}
client = httpx.Client(timeout=300)
def row(r):
return {k: r[k] for k in ("id", "state", "question", "options")}
def decide(r):
resp = client.post(f"{URL}/decide", json=row(r), headers=H)
resp.raise_for_status()
return resp.json()
def argmax(p):
return max(range(len(p)), key=p.__getitem__)
def compare(ours, ref):
agree = sum(argmax(ours[i]["probabilities"]) == argmax(ref[i]["probabilities"]) for i in ref)
gap = max(max(abs(a - b) for a, b in zip(ours[i]["probabilities"], ref[i]["probabilities"])) for i in ref)
return {"rows": len(ref), "top_choice_agree": agree, "max_abs_prob_gap": gap}
rows = [json.loads(l) for l in (SEMIF / "benchmarks/data/authored144.jsonl").read_text().splitlines() if l.strip()]
ref = {p["id"]: p for p in map(json.loads, (SEMIF / "results/raw/predictions/direct-authored144.jsonl").read_text().splitlines()) if p}
report = {"url": URL, "health": client.get(f"{URL}/health").json()}
run_a = {r["id"]: decide(r) for r in rows}
run_b = {r["id"]: decide(r) for r in rows}
report["1_parity_vs_upstream"] = compare(run_a, ref)
report["1_prompt_sha256_equal"] = sum(run_a[i]["prompt_sha256"] == ref[i]["prompt_sha256"] for i in ref)
report["2_noise_floor_a_vs_b"] = compare(run_a, run_b)
rotated = []
for r in rows:
descs = [o["description"] for o in r["options"]]
descs = descs[1:] + descs[:1]
rotated.append({**r, "options": [{**o, "description": d} for o, d in zip(r["options"], descs)]})
report["3_negative_rotated_options"] = compare({r["id"]: decide(r) for r in rotated}, ref)
# 4: group authored144 by identical state; score every multi-row group both ways.
groups = {}
for r in rows:
groups.setdefault(json.dumps(r["state"], sort_keys=True), []).append(r)
multi = [g for g in groups.values() if len(g) > 1]
shared_out = {}
for g in multi:
resp = client.post(f"{URL}/decide/shared", headers=H, json={
"state": g[0]["state"], "decisions": [{k: r[k] for k in ("id", "question", "options")} for r in g]})
resp.raise_for_status()
shared_out.update({res["id"]: res for res in resp.json()["results"]})
report["4_shared_vs_direct"] = {"groups": len(multi), **compare(shared_out, {i: run_a[i] for i in shared_out})}
# 5: 21 binary criteria over one state
state = rows[0]["state"]
crit = [{"id": f"c{i}", "question": f"Does the evidence mention item number {i}?",
"options": [{"id": "yes", "description": "Yes"}, {"id": "no", "description": "No"}]} for i in range(21)]
for _ in range(2): # warm-up
client.post(f"{URL}/decide/shared", headers=H, json={"state": state, "decisions": crit}).raise_for_status()
shared_s, seq_s, prefix = [], [], None
for _ in range(3):
t = time.perf_counter()
resp = client.post(f"{URL}/decide/shared", headers=H, json={"state": state, "decisions": crit})
resp.raise_for_status()
shared_s.append(time.perf_counter() - t)
prefix = resp.json()["timing"]["prefix_tokens"]
t = time.perf_counter()
for c in crit:
decide({**c, "state": state})
seq_s.append(time.perf_counter() - t)
report["5_speed_21_binary"] = {
"prefix_tokens": prefix,
"shared_s": {"runs": shared_s, "median": statistics.median(shared_s)},
"sequential_decide_s": {"runs": seq_s, "median": statistics.median(seq_s)},
}
json.dump(report, open(sys.argv[1], "w"), indent=1)
print(json.dumps({k: v for k, v in report.items() if k != "health"}, indent=1))
@@ -0,0 +1,63 @@
{
"url": "http://10.251.50.54:8032",
"health": {
"status": "ok",
"semif_commit": "23cf1f39fc9534fe81437200959b6dfc7106e45a",
"model": {
"source": "Qwen/Qwen3.5-4B",
"revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a",
"dtype": "bfloat16",
"device": "cuda:0",
"torch_version": "2.10.0+cu128",
"transformers_version": "5.17.0",
"device_name": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition",
"allocated_gib": 7.85,
"reserved_gib": 7.86
},
"vram_cap_gib": 12.0,
"max_tokens": 4096,
"max_decisions": 64,
"workloads": []
},
"1_parity_vs_upstream": {
"rows": 144,
"top_choice_agree": 142,
"max_abs_prob_gap": 0.09262644585476243
},
"1_prompt_sha256_equal": 144,
"2_noise_floor_a_vs_b": {
"rows": 144,
"top_choice_agree": 144,
"max_abs_prob_gap": 0.0
},
"3_negative_rotated_options": {
"rows": 144,
"top_choice_agree": 14,
"max_abs_prob_gap": 0.9987451392432198
},
"4_shared_vs_direct": {
"groups": 36,
"rows": 72,
"top_choice_agree": 72,
"max_abs_prob_gap": 0.044578713319407104
},
"5_speed_21_binary": {
"prefix_tokens": 62,
"shared_s": {
"runs": [
0.15981742000440136,
0.159110098000383,
0.15852549100236502
],
"median": 0.159110098000383
},
"sequential_decide_s": {
"runs": [
0.978545692996704,
0.9808067879930604,
0.9893363219889579
],
"median": 0.9808067879930604
}
}
}
+39
View File
@@ -0,0 +1,39 @@
[project]
name = "semif-serve"
version = "0.1.2"
description = "HTTP wrapper around SemIf's direct and shared option-logit scorers"
requires-python = ">=3.12"
dependencies = [
"fastapi==0.118.0",
"uvicorn==0.37.0",
]
[project.optional-dependencies]
# The real engine. Pulls torch 2.10.0 (cu128) and transformers 5.17.0 through SemIf's exact pins.
model = [
"semif-phase1 @ git+https://github.com/TheoLeeCJ/SemIf-OpenJev@23cf1f39fc9534fe81437200959b6dfc7106e45a",
# Same pin SemIf declares, taken from the cu128 index: SemIf's committed predictions report
# torch 2.10.0+cu128, and cu128 carries sm_120 kernels for the Blackwell cards.
"torch==2.10.0",
]
[dependency-groups]
dev = ["pytest==8.4.2", "httpx==0.28.1"]
[build-system]
requires = ["setuptools>=68"]
build-backend = "setuptools.build_meta"
[tool.setuptools.packages.find]
where = ["src"]
[tool.pytest.ini_options]
testpaths = ["tests"]
[[tool.uv.index]]
name = "pytorch-cu128"
url = "https://download.pytorch.org/whl/cu128"
explicit = true
[tool.uv.sources]
torch = { index = "pytorch-cu128" }
@@ -0,0 +1,122 @@
---
title: semif-serve
kind: module-contract
status: draft
owner: infra-ops
created: 2026-09-27
depends_on:
- SemIf-OpenJev (MIT) at commit 23cf1f39fc9534fe81437200959b6dfc7106e45a, package semif_phase1
- Qwen/Qwen3.5-4B at revision 851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a, BF16
---
# semif-serve: an HTTP wrapper around SemIf's direct and shared scorers
## Purpose
SemIf decides by reading the logits of the option letters after one forward pass.
It ships as a batch CLI only. semif-serve loads the model **once** and exposes the
two torch scorers over HTTP, so fleet callers can ask typed questions without
decoding. It adds nothing to the scoring itself: every score it returns is what
`semif_phase1.direct.score` or `semif_phase1.shared.score_shared` returned,
unchanged, and it adds an optional calibrated view next to it.
Operator decisions (Prime, 2026-09-27): runs on fv-ml1 GPU 1 under a hard VRAM
cap; built by infra-ops with a light process (this contract → TDD → heid bug
hunt); no consumer is named yet.
## Endpoints
Every POST takes and returns JSON. Every endpoint except `GET /health` requires
`Authorization: Bearer <token>`.
| method | path | body | success |
|---|---|---|---|
| GET | `/health` | — | 200 `{status: "ok", semif_commit, model, vram_cap_gib, max_tokens, max_decisions, workloads}` |
| POST | `/decide` | one SemIf row `{id, state, question, options[2..16]}` + optional `workload` | 200 the `direct.score` dict + optional `calibrated` |
| POST | `/decide/shared` | `{state, decisions: [{id, question, options}], workload?}` | 200 `{results: [...], timing: {...}}` from `score_shared` |
`options` items are `{id, description}`, as in SemIf. `state` is a nonempty
string, object or array. For `/decide/shared`, each decision becomes a SemIf row
by adding the shared `state`.
**Calibration.** `workload` is optional. When it names an entry in the
calibration table, each result gains `calibrated: {workload, temperature,
probabilities}` with `softmax(option_logits / T)`. The argmax never changes. The
native `probabilities` and `probability_status` stay untouched. An unknown
`workload` is a 422. With no `workload`, no `calibrated` key appears.
## Invariants
- **INV-1 pass-through.** `option_ids`, `probabilities`, `option_logits`,
`prompt_sha256`, `prompt_version`, `model` and `probability_status` are exactly
what SemIf returned. The wrapper never rewrites a score.
- **INV-2 one model, one inference at a time.** The model is loaded at startup,
and a process-wide lock serialises every scorer call. The app runs as one worker.
Scorer calls run off the event loop, so `/health` answers while one is in progress.
- **INV-3 fail-closed startup.** Startup refuses to serve unless the model sits
on a CUDA device, torch's arch list includes the card's `sm_XY`, and one
warm-up decision scores. `device=cpu` is allowed only when set explicitly.
- **INV-4 VRAM cap.** When `SEMIF_VRAM_CAP_GIB` is set, the process is capped at
that share of the card (`torch.cuda.set_per_process_memory_fraction`) before
the model loads. An out-of-memory error during a request is a 503
`out_of_memory`, followed by `torch.cuda.empty_cache()`. The process stays up.
**After every scorer call**, when the reserved memory exceeds the post-warm-up
baseline by more than 512 MiB, the engine calls `empty_cache()`. A burst must not
keep the card's shared headroom: on 2026-09-27 a 64-decision request left the
process holding 12.6 GB, leaving scriberr 3.5 GB.
- **INV-5 no network at runtime.** Weights come from the mounted HF cache at the
pinned revision (`HF_HUB_OFFLINE=1`).
- **INV-6 constant-time auth.** Token comparison uses `hmac.compare_digest`. The
token is ≥ 32 characters, and startup refuses a shorter one.
## Limits and errors
- `SEMIF_MAX_TOKENS` (default 4096) is passed to the scorers. A longer prompt is a
422, never truncated (SemIf raises).
- `SEMIF_MAX_DECISIONS` (default 64) caps `decisions` per shared request. It must
hold 1..max entries, else 422.
- Request body ≤ `SEMIF_MAX_BODY_BYTES` (default 1 MiB), else 413.
| status | code | when |
|---|---|---|
| 401 | `unauthorized` | missing or wrong bearer |
| 413 | `request_too_large` | body over the limit |
| 422 | `invalid_request` | bad JSON shape, a SemIf `ValueError` (validation, token limit, tokenisation), unknown workload, too many decisions |
| 503 | `out_of_memory` | CUDA OOM during scoring |
| 500 | `scoring_failed` | any other scorer exception |
The error body is `{error: {code, message}}`.
## Configuration (env)
`SEMIF_API_TOKEN` (required), `SEMIF_MODEL` (default `Qwen/Qwen3.5-4B`),
`SEMIF_REVISION` (default the pinned SHA), `SEMIF_DEVICE` (default `cuda`),
`SEMIF_VRAM_CAP_GIB`, `SEMIF_MAX_TOKENS`, `SEMIF_MAX_DECISIONS`,
`SEMIF_MAX_BODY_BYTES`, `SEMIF_CALIBRATION` (path to a JSON `{workload: T}`; T > 0).
## Tests (TDD, fake scorer: no torch, no model)
auth required on POSTs and not on /health; a short token is refused at startup;
`/decide` passes the row through and returns the scorer dict unchanged;
`/decide/shared` builds rows with the shared state and returns results + timing;
calibration adds `calibrated` and keeps the argmax; an unknown workload → 422; a
scorer `ValueError` → 422; the engine's `OutOfMemory` → 503; the torch engine,
against a fake torch, turns `torch.cuda.OutOfMemoryError` into an **unchained**
`OutOfMemory` and calls `empty_cache()` only after the failed call's tensors are freed
(found on the card: a chained exception kept 11.9 GiB allocated after the 503); after
a call, reserved memory over baseline + 512 MiB is released and at or under it is left
alone; any
other exception → 500; malformed
JSON or a wrong body shape → 422; too many decisions → 422; an oversized body → 413; requests
are serialised (two concurrent calls never overlap inside the scorer); `/health`
answers while a scorer call is blocked.
## Acceptance (on fv-ml1, real model; not unit tests)
1. **Parity:** our `/decide` over SemIf's `authored144` against their committed
torch predictions (top choice and max probability gap).
2. **Noise floor:** the same run twice (A-vs-A).
3. **Negative control:** shuffled option descriptions must break agreement.
4. **Shared vs direct:** the same rows agree within the A-vs-A floor.
5. **Speed:** 21 binary criteria over one state, N ≥ 3, p50 + spread.
6. **VRAM:** the peak at a 4096-token input sets `SEMIF_VRAM_CAP_GIB`.
+161
View File
@@ -0,0 +1,161 @@
"""semif-serve HTTP layer. Contract: semif-serve.contract.md."""
from __future__ import annotations
import hmac
import math
import threading
from typing import Any
from fastapi import FastAPI, Request
from fastapi.concurrency import run_in_threadpool
from fastapi.responses import JSONResponse
from pydantic import BaseModel, ConfigDict, ValidationError
from .config import SEMIF_COMMIT, Settings
from .errors import OutOfMemory
OPEN_PATHS = frozenset({"/health"})
State = str | dict[str, Any] | list[Any]
class Option(BaseModel):
model_config = ConfigDict(extra="ignore")
id: str
description: str
class Decision(BaseModel):
model_config = ConfigDict(extra="ignore")
id: str
question: str
options: list[Option]
def row(self, state: State) -> dict:
"""The SemIf row shape: exactly id, state, question, options."""
return {"id": self.id, "state": state, "question": self.question,
"options": [o.model_dump() for o in self.options]}
class DecideBody(Decision):
state: State
workload: str | None = None
class SharedBody(BaseModel):
model_config = ConfigDict(extra="ignore")
state: State
decisions: list[Decision]
workload: str | None = None
class ApiError(Exception):
def __init__(self, status: int, code: str, message: str):
super().__init__(message)
self.status, self.code, self.message = status, code, message
def error(status: int, code: str, message: str) -> JSONResponse:
return JSONResponse(status_code=status, content={"error": {"code": code, "message": message}})
def _first_error(exc: ValidationError) -> str:
first = exc.errors()[0]
where = ".".join(str(p) for p in first.get("loc", ())) or "body"
return f"{where}: {first.get('msg', 'invalid')}"
def calibrated_view(result: dict, workload: str, temperature: float) -> dict:
"""softmax(option_logits / T): the native fields are left exactly as SemIf returned them (INV-1)."""
scaled = [x / temperature for x in result["option_logits"]]
top = max(scaled)
weights = [math.exp(x - top) for x in scaled]
total = sum(weights)
return {"workload": workload, "temperature": temperature, "probabilities": [w / total for w in weights]}
def create_app(settings: Settings, engine: Any) -> FastAPI:
app = FastAPI(title="semif-serve")
expected = f"Bearer {settings.api_token}".encode()
inference = threading.Lock() # INV-2: one scorer call at a time, off the event loop
def locked(fn, *args):
with inference:
return fn(*args)
@app.middleware("http")
async def require_bearer(request: Request, call_next):
if request.url.path not in OPEN_PATHS:
supplied = request.headers.get("authorization", "").encode()
if not hmac.compare_digest(supplied, expected): # INV-6
return error(401, "unauthorized", "missing or wrong bearer token")
return await call_next(request)
@app.exception_handler(ApiError)
async def api_error(_request: Request, exc: ApiError):
return error(exc.status, exc.code, exc.message)
async def read_limited(request: Request) -> bytes:
limit = settings.max_body_bytes
too_large = ApiError(413, "request_too_large", f"request body exceeds {limit} bytes")
declared = request.headers.get("content-length")
if declared is not None and declared.isdigit() and int(declared) > limit:
raise too_large
body = bytearray()
async for chunk in request.stream(): # also caps bodies that declare no length
body.extend(chunk)
if len(body) > limit:
raise too_large
return bytes(body)
async def parse(request: Request, model: type[BaseModel]):
try:
return model.model_validate_json(await read_limited(request))
except ValidationError as exc:
raise ApiError(422, "invalid_request", _first_error(exc)) from exc
async def score(fn, *args):
"""Run one scorer call in a worker thread under the lock; map its failures to contract codes."""
try:
return await run_in_threadpool(locked, fn, *args)
except ValueError as exc: # SemIf validation, token limit, tokenisation
raise ApiError(422, "invalid_request", str(exc)) from exc
except OutOfMemory as exc:
raise ApiError(503, "out_of_memory", str(exc)) from exc
except Exception as exc: # noqa: BLE001 — any other scorer failure
raise ApiError(500, "scoring_failed", f"{type(exc).__name__}: {exc}") from exc
def temperature_for(workload: str | None) -> float | None:
if workload is None:
return None
if workload not in settings.calibration:
raise ApiError(422, "invalid_request", f"unknown workload {workload!r}")
return settings.calibration[workload]
def with_calibration(result: dict, workload: str | None, temperature: float | None) -> dict:
if temperature is None:
return result
return {**result, "calibrated": calibrated_view(result, workload, temperature)}
@app.get("/health")
async def health():
return {"status": "ok", "semif_commit": SEMIF_COMMIT, "model": engine.health(),
"vram_cap_gib": settings.vram_cap_gib, "max_tokens": settings.max_tokens,
"max_decisions": settings.max_decisions, "workloads": sorted(settings.calibration)}
@app.post("/decide")
async def decide(request: Request):
body = await parse(request, DecideBody)
temperature = temperature_for(body.workload)
return with_calibration(await score(engine.direct, body.row(body.state)), body.workload, temperature)
@app.post("/decide/shared")
async def decide_shared(request: Request):
body = await parse(request, SharedBody)
if not 1 <= len(body.decisions) <= settings.max_decisions:
raise ApiError(422, "invalid_request",
f"decisions must hold 1..{settings.max_decisions} entries, got {len(body.decisions)}")
temperature = temperature_for(body.workload)
results, timing = await score(engine.shared, [d.row(body.state) for d in body.decisions])
return {"results": [with_calibration(r, body.workload, temperature) for r in results], "timing": timing}
return app
@@ -0,0 +1,57 @@
"""Settings for semif-serve. Contract: semif-serve.contract.md § Configuration."""
from __future__ import annotations
import json
import math
from collections.abc import Mapping
from dataclasses import dataclass, field
from pathlib import Path
MIN_TOKEN_CHARS = 32
SEMIF_COMMIT = "23cf1f39fc9534fe81437200959b6dfc7106e45a"
DEFAULT_MODEL = "Qwen/Qwen3.5-4B"
DEFAULT_REVISION = "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a"
@dataclass(frozen=True)
class Settings:
api_token: str
model: str = DEFAULT_MODEL
revision: str = DEFAULT_REVISION
device: str = "cuda"
vram_cap_gib: float | None = None
max_tokens: int = 4096
max_decisions: int = 64
max_body_bytes: int = 1024 * 1024
calibration: dict[str, float] = field(default_factory=dict)
@classmethod
def from_env(cls, env: Mapping[str, str]) -> "Settings":
token = env.get("SEMIF_API_TOKEN", "")
if len(token) < MIN_TOKEN_CHARS: # INV-6
raise ValueError(f"SEMIF_API_TOKEN must be at least {MIN_TOKEN_CHARS} characters")
cap = env.get("SEMIF_VRAM_CAP_GIB")
return cls(
api_token=token,
model=env.get("SEMIF_MODEL", DEFAULT_MODEL),
revision=env.get("SEMIF_REVISION", DEFAULT_REVISION),
device=env.get("SEMIF_DEVICE", "cuda"),
vram_cap_gib=float(cap) if cap else None,
max_tokens=int(env.get("SEMIF_MAX_TOKENS", 4096)),
max_decisions=int(env.get("SEMIF_MAX_DECISIONS", 64)),
max_body_bytes=int(env.get("SEMIF_MAX_BODY_BYTES", 1024 * 1024)),
calibration=_load_calibration(env.get("SEMIF_CALIBRATION")),
)
def _load_calibration(path: str | None) -> dict[str, float]:
"""{workload: T}, every T a finite number > 0 (T scales option logits before softmax)."""
if not path:
return {}
table = json.loads(Path(path).read_text())
if not isinstance(table, dict) or not all(
isinstance(t, (int, float)) and not isinstance(t, bool) and math.isfinite(t) and t > 0
for t in table.values()
):
raise ValueError("SEMIF_CALIBRATION must be a JSON object of workload -> finite temperature > 0")
return {str(k): float(v) for k, v in table.items()}
@@ -0,0 +1,101 @@
"""The real engine: SemIf's torch scorers over one resident model. Needs the `model` extra.
Contract: semif-serve.contract.md, INV-3 (fail-closed startup), INV-4 (VRAM cap + OOM),
INV-5 (offline weights). load() is checked on the card at acceptance; the OOM path is
unit-tested against a fake torch (tests/test_engine.py).
"""
from __future__ import annotations
import gc
from typing import Any, Callable
from .config import Settings
from .errors import OutOfMemory
RELEASE_SLACK_BYTES = 512 * 2**20
WARMUP_ROW = {
"id": "semif-serve-warmup",
"state": "The deployment completed at 14:02 UTC. Health checks passed in all three zones.",
"question": "Is there evidence that the deployment succeeded?",
"options": [
{"id": "yes", "description": "The deployment succeeded."},
{"id": "no", "description": "The deployment did not succeed."},
],
}
class TorchEngine:
def __init__(self, torch: Any, model: Any, tokenizer: Any, metadata: dict, settings: Settings,
direct_fn: Callable, shared_fn: Callable, release_above_bytes: int | None = None):
self._torch, self._model, self._tokenizer = torch, model, tokenizer
self._metadata, self._settings = metadata, settings
self._direct, self._shared = direct_fn, shared_fn
self._release_above = release_above_bytes
@classmethod
def load(cls, settings: Settings) -> "TorchEngine":
import torch
from semif_phase1.core import load_causal_model
from semif_phase1.direct import score
from semif_phase1.shared import score_shared
if settings.device == "cuda":
if not torch.cuda.is_available():
raise RuntimeError("SEMIF_DEVICE=cuda but torch sees no CUDA device")
major, minor = torch.cuda.get_device_capability(0)
arch = f"sm_{major}{minor}"
if arch not in torch.cuda.get_arch_list(): # INV-3: no silent PTX/CPU fallback
raise RuntimeError(f"torch {torch.__version__} has no kernels for {arch}: {torch.cuda.get_arch_list()}")
if settings.vram_cap_gib: # INV-4: cap BEFORE the weights land
total = torch.cuda.get_device_properties(0).total_memory
fraction = settings.vram_cap_gib * 2**30 / total
if not 0 < fraction <= 1:
raise ValueError(f"SEMIF_VRAM_CAP_GIB={settings.vram_cap_gib} does not fit a {total / 2**30:.1f} GiB card")
torch.cuda.set_per_process_memory_fraction(fraction, 0)
elif settings.device != "cpu":
raise ValueError(f"SEMIF_DEVICE must be cuda or cpu, not {settings.device!r}")
model, tokenizer, metadata = load_causal_model(settings.model, settings.revision, settings.device, "bfloat16")
placed = next(model.parameters()).device.type
if placed != settings.device: # INV-3
raise RuntimeError(f"model landed on {placed}, expected {settings.device}")
engine = cls(torch, model, tokenizer, metadata, settings, direct_fn=score, shared_fn=score_shared)
engine.direct(WARMUP_ROW) # INV-3: one decision must score
if settings.device == "cuda": # INV-4: the resting footprint
engine._release_above = torch.cuda.memory_reserved(0) + RELEASE_SLACK_BYTES
return engine
def health(self) -> dict:
info = dict(self._metadata)
if self._settings.device == "cuda":
info["device_name"] = self._torch.cuda.get_device_name(0)
info["allocated_gib"] = round(self._torch.cuda.memory_allocated(0) / 2**30, 2)
info["reserved_gib"] = round(self._torch.cuda.memory_reserved(0) / 2**30, 2)
return info
def _release_burst(self) -> None:
"""INV-4: hand a burst back to the driver so the card's shared headroom (scriberr, the
vLLM seats) returns after a big request, instead of sitting in torch's cache."""
if self._release_above is not None and self._torch.cuda.memory_reserved(0) > self._release_above:
self._torch.cuda.empty_cache()
def _guard(self, fn, *args):
try:
result = fn(*args)
except self._torch.cuda.OutOfMemoryError as exc:
message = str(exc).splitlines()[0]
else:
self._release_burst()
return result
# INV-4, outside the except block on purpose: the torch exception's traceback holds the
# failed scorer's frames, and with them its tensors (the replicated prefix cache). Raising
# inside the block, or `from exc`, would chain to it and keep GiBs allocated after the 503.
gc.collect()
self._torch.cuda.empty_cache()
raise OutOfMemory(message)
def direct(self, row: dict) -> dict:
return self._guard(self._direct, self._model, self._tokenizer, row, self._metadata, self._settings.max_tokens)
def shared(self, rows: list[dict]) -> tuple[list[dict], dict]:
return self._guard(self._shared, self._model, self._tokenizer, rows, self._metadata, self._settings.max_tokens)
@@ -0,0 +1,5 @@
"""Torch-free exceptions shared by the HTTP layer and the engine."""
class OutOfMemory(RuntimeError):
"""The engine ran out of GPU memory during a request and has already released its cache (INV-4)."""
@@ -0,0 +1,16 @@
"""uvicorn entry point: `uvicorn semif_serve.main:app_from_env --factory --workers 1`."""
from __future__ import annotations
import os
from fastapi import FastAPI
from .app import create_app
from .config import Settings
def app_from_env() -> FastAPI:
settings = Settings.from_env(os.environ)
from .engine import TorchEngine # torch loads only here, never in the unit tests
return create_app(settings, TorchEngine.load(settings))
+233
View File
@@ -0,0 +1,233 @@
"""semif-serve HTTP behaviour against a fake engine (no torch, no model).
Contract: services/semif-serve/semif-serve.contract.md"""
import pytest
from fastapi.testclient import TestClient
from semif_serve.app import create_app
from semif_serve.config import Settings
from semif_serve.errors import OutOfMemory
TOKEN = "t" * 40
AUTH = {"Authorization": f"Bearer {TOKEN}"}
OPTIONS = [{"id": "yes", "description": "Yes."}, {"id": "no", "description": "No."}]
ROW = {"id": "r1", "state": "The deploy passed.", "question": "Did it pass?", "options": OPTIONS}
class FakeEngine:
"""Returns canned SemIf-shaped dicts and records what it was asked."""
def __init__(self, logits=(2.0, 0.0)):
self.logits = list(logits)
self.direct_calls = []
self.shared_calls = []
def health(self):
return {"source": "fake/model", "revision": "0" * 40}
def direct(self, row):
self.direct_calls.append(row)
return {
"id": row["id"],
"option_ids": [o["id"] for o in row["options"]],
"probabilities": [0.8807970779778823, 0.11920292202211769],
"option_logits": self.logits,
"prompt_sha256": "ab" * 32,
"prompt_version": "direct-options-v1",
"model": {"source": "fake/model"},
"probability_status": "conditional option score; uncalibrated as decision confidence",
}
def shared(self, rows):
self.shared_calls.append(rows)
return [self.direct(r) for r in rows], {"prefix_tokens": 7, "batch_size": len(rows)}
def make_client(engine=None, **overrides):
settings = Settings(api_token=TOKEN, **overrides)
return TestClient(create_app(settings, engine or FakeEngine()))
def test_decide_passes_the_row_through_and_returns_the_scorer_dict_unchanged():
engine = FakeEngine()
client = make_client(engine)
response = client.post("/decide", json=ROW, headers=AUTH)
assert response.status_code == 200
assert response.json() == engine.direct(ROW)
assert engine.direct_calls[0] == ROW
@pytest.mark.parametrize("headers", [{}, {"Authorization": "Bearer wrong"}, {"Authorization": TOKEN}])
def test_posts_without_the_right_bearer_are_401_and_never_reach_the_engine(headers):
engine = FakeEngine()
client = make_client(engine)
for path in ("/decide", "/decide/shared"):
response = client.post(path, json=ROW, headers=headers)
assert response.status_code == 401
assert response.json() == {"error": {"code": "unauthorized", "message": response.json()["error"]["message"]}}
assert engine.direct_calls == []
def test_health_needs_no_auth():
response = make_client().get("/health")
assert response.status_code == 200
assert response.json()["status"] == "ok"
def test_shared_builds_semif_rows_from_the_shared_state_and_returns_results_and_timing():
engine = FakeEngine()
body = {"state": {"deploy": "passed"},
"decisions": [{"id": "a", "question": "Q1?", "options": OPTIONS},
{"id": "b", "question": "Q2?", "options": OPTIONS}]}
response = make_client(engine).post("/decide/shared", json=body, headers=AUTH)
assert response.status_code == 200
assert engine.shared_calls == [[
{"id": "a", "state": {"deploy": "passed"}, "question": "Q1?", "options": OPTIONS},
{"id": "b", "state": {"deploy": "passed"}, "question": "Q2?", "options": OPTIONS},
]]
out = response.json()
assert [r["id"] for r in out["results"]] == ["a", "b"]
assert out["timing"] == {"prefix_tokens": 7, "batch_size": 2}
def test_a_known_workload_adds_a_calibrated_view_and_leaves_the_native_scores_alone():
import math
engine = FakeEngine(logits=(3.0, 1.0))
client = make_client(engine, calibration={"triage": 2.0})
native = engine.direct(ROW)
for path, body, pick in (
("/decide", {**ROW, "workload": "triage"}, lambda j: [j]),
("/decide/shared", {"state": ROW["state"], "workload": "triage",
"decisions": [{"id": "a", "question": "Q?", "options": OPTIONS}]}, lambda j: j["results"]),
):
for result in pick(client.post(path, json=body, headers=AUTH).json()):
cal = result.pop("calibrated")
assert (cal["workload"], cal["temperature"]) == ("triage", 2.0)
e = [math.exp(1.5), math.exp(0.5)]
assert cal["probabilities"] == pytest.approx([e[0] / sum(e), e[1] / sum(e)])
assert cal["probabilities"].index(max(cal["probabilities"])) == 0
assert {k: result[k] for k in ("probabilities", "option_logits", "probability_status")} == \
{k: native[k] for k in ("probabilities", "option_logits", "probability_status")}
def test_no_workload_means_no_calibrated_key():
client = make_client(calibration={"triage": 2.0})
assert "calibrated" not in client.post("/decide", json=ROW, headers=AUTH).json()
def test_an_unknown_workload_is_422_before_the_engine_runs():
engine = FakeEngine()
client = make_client(engine, calibration={"triage": 2.0})
response = client.post("/decide", json={**ROW, "workload": "nope"}, headers=AUTH)
assert response.status_code == 422
assert response.json()["error"]["code"] == "invalid_request"
assert engine.direct_calls == []
class RaisingEngine(FakeEngine):
def __init__(self, exc):
super().__init__()
self.exc = exc
def direct(self, row):
raise self.exc
def shared(self, rows):
raise self.exc
@pytest.mark.parametrize("exc, status, code", [
(ValueError("Row r1: 5000 input tokens exceed limit 4096; no truncation allowed"), 422, "invalid_request"),
(OutOfMemory("CUDA out of memory"), 503, "out_of_memory"),
(RuntimeError("Invalid native prefix cache"), 500, "scoring_failed"),
])
def test_scorer_failures_map_to_the_contract_status_codes(exc, status, code):
client = make_client(RaisingEngine(exc))
for path, body in (("/decide", ROW),
("/decide/shared", {"state": "s", "decisions": [{"id": "a", "question": "Q?", "options": OPTIONS}]})):
response = client.post(path, json=body, headers=AUTH)
assert response.status_code == status
assert response.json()["error"]["code"] == code
assert str(exc) in response.json()["error"]["message"]
@pytest.mark.parametrize("content", [b"{not json", b'{"id": "r1"}', b'{"state": "s", "decisions": "nope"}', b"[1, 2]"])
def test_malformed_json_or_a_wrong_shape_is_422(content):
client = make_client()
for path in ("/decide", "/decide/shared"):
response = client.post(path, content=content, headers={**AUTH, "content-type": "application/json"})
assert response.status_code == 422
assert response.json()["error"]["code"] == "invalid_request"
def test_more_decisions_than_the_cap_is_422_and_the_engine_never_runs():
engine = FakeEngine()
client = make_client(engine, max_decisions=2)
decisions = [{"id": str(i), "question": "Q?", "options": OPTIONS} for i in range(3)]
response = client.post("/decide/shared", json={"state": "s", "decisions": decisions}, headers=AUTH)
assert response.status_code == 422
assert response.json()["error"]["code"] == "invalid_request"
assert engine.shared_calls == []
def test_an_empty_decision_list_is_422():
response = make_client().post("/decide/shared", json={"state": "s", "decisions": []}, headers=AUTH)
assert response.status_code == 422
@pytest.mark.parametrize("chunked", [False, True])
def test_a_body_over_the_limit_is_413_whether_or_not_it_declares_its_length(chunked):
engine = FakeEngine()
client = make_client(engine, max_body_bytes=200)
body = ('{"id": "r1", "state": "' + "x" * 500 + '", "question": "Q?", "options": []}').encode()
content = (chunk for chunk in [body[:100], body[100:]]) if chunked else body
response = client.post("/decide", content=content, headers={**AUTH, "content-type": "application/json"})
assert response.status_code == 413
assert response.json()["error"]["code"] == "request_too_large"
assert engine.direct_calls == []
class SlowEngine(FakeEngine):
"""Holds each scorer call until released, and records the peak number of calls inside at once."""
def __init__(self):
super().__init__()
import threading
self.inside = 0
self.peak = 0
self.guard = threading.Lock()
self.release = threading.Event()
self.entered = threading.Event()
def direct(self, row):
with self.guard:
self.inside += 1
self.peak = max(self.peak, self.inside)
self.entered.set()
self.release.wait(5)
with self.guard:
self.inside -= 1
return super().direct(row)
def test_concurrent_requests_never_overlap_inside_the_scorer_and_health_still_answers():
from concurrent.futures import ThreadPoolExecutor
engine = SlowEngine()
with make_client(engine) as client, ThreadPoolExecutor(4) as pool:
futures = [pool.submit(client.post, "/decide", json={**ROW, "id": f"r{i}"}, headers=AUTH) for i in range(4)]
assert engine.entered.wait(5)
import time
started = time.monotonic()
assert client.get("/health").status_code == 200
assert time.monotonic() - started < 1.0 # answered while the scorer is still held
assert not engine.release.is_set() and engine.inside == 1
engine.release.set()
assert [f.result().status_code for f in futures] == [200] * 4
assert engine.peak == 1
def test_health_reports_the_pins_limits_and_workloads():
from semif_serve.config import SEMIF_COMMIT
client = make_client(vram_cap_gib=12.0, max_decisions=8, calibration={"triage": 2.0, "alerts": 1.3})
body = client.get("/health").json()
assert body == {"status": "ok", "semif_commit": SEMIF_COMMIT, "model": FakeEngine().health(),
"vram_cap_gib": 12.0, "max_tokens": 4096, "max_decisions": 8, "workloads": ["alerts", "triage"]}
+36
View File
@@ -0,0 +1,36 @@
"""Settings.from_env. Contract: semif-serve.contract.md § Configuration, INV-6."""
import json
import pytest
from semif_serve.config import DEFAULT_REVISION, Settings
TOKEN = "t" * 40
def test_defaults_from_a_minimal_env():
s = Settings.from_env({"SEMIF_API_TOKEN": TOKEN})
assert (s.api_token, s.revision, s.device, s.max_tokens, s.max_decisions) == (TOKEN, DEFAULT_REVISION, "cuda", 4096, 64)
assert s.vram_cap_gib is None and s.calibration == {}
@pytest.mark.parametrize("env", [{}, {"SEMIF_API_TOKEN": "short"}, {"SEMIF_API_TOKEN": "x" * 31}])
def test_a_missing_or_short_token_is_refused_at_startup(env):
with pytest.raises(ValueError, match="SEMIF_API_TOKEN"):
Settings.from_env(env)
def test_numbers_and_calibration_file_are_parsed(tmp_path):
cal = tmp_path / "cal.json"
cal.write_text(json.dumps({"triage": 2.5}))
s = Settings.from_env({"SEMIF_API_TOKEN": TOKEN, "SEMIF_VRAM_CAP_GIB": "12", "SEMIF_MAX_DECISIONS": "8",
"SEMIF_CALIBRATION": str(cal)})
assert (s.vram_cap_gib, s.max_decisions, s.calibration) == (12.0, 8, {"triage": 2.5})
@pytest.mark.parametrize("table", [{"w": 0}, {"w": -1.0}, {"w": "2"}, {"w": float("inf")}, ["w", 2.0]])
def test_a_calibration_table_needs_positive_finite_numbers(tmp_path, table):
cal = tmp_path / "cal.json"
cal.write_text(json.dumps(table))
with pytest.raises(ValueError, match="SEMIF_CALIBRATION"):
Settings.from_env({"SEMIF_API_TOKEN": TOKEN, "SEMIF_CALIBRATION": str(cal)})
+83
View File
@@ -0,0 +1,83 @@
"""TorchEngine's OOM path against a fake torch (INV-4). Found on the card 2026-09-27: after a
503 the failed call's tensors stayed alive (11.92 GiB allocated) because the raised
OutOfMemory chained back to the torch exception, whose traceback held the scorer's frames."""
import weakref
import pytest
from semif_serve.config import Settings
from semif_serve.engine import TorchEngine
from semif_serve.errors import OutOfMemory
class FakeTorch:
class cuda:
class OutOfMemoryError(RuntimeError):
pass
empties = [] # for each empty_cache() call: was the failed call's tensor already freed?
watched = []
@classmethod
def empty_cache(cls):
cls.empties.append(all(ref() is None for ref in cls.watched))
class Tensor:
pass
def failing_scorer(*_args):
kv_cache = Tensor() # stands in for the replicated prefix cache
FakeTorch.cuda.watched.append(weakref.ref(kv_cache))
raise FakeTorch.cuda.OutOfMemoryError("CUDA out of memory. Tried to allocate 490.00 MiB.\nGPU 0 has ...")
@pytest.mark.parametrize("call", ["direct", "shared"])
def test_oom_frees_the_failed_call_before_emptying_the_cache_and_keeps_nothing_alive(call):
FakeTorch.cuda.empties.clear(), FakeTorch.cuda.watched.clear()
engine = TorchEngine(FakeTorch, model=None, tokenizer=None, metadata={}, settings=Settings(api_token="t" * 40),
direct_fn=failing_scorer, shared_fn=failing_scorer)
with pytest.raises(OutOfMemory) as info:
getattr(engine, call)({})
assert str(info.value) == "CUDA out of memory. Tried to allocate 490.00 MiB."
assert info.value.__cause__ is None and info.value.__context__ is None # no chain back to the frames
assert FakeTorch.cuda.watched[0]() is None # nothing keeps the tensor alive
assert FakeTorch.cuda.empties == [True] # emptied once, after it was freed
def test_other_scorer_errors_pass_through_unchanged():
def bad(*_args):
raise ValueError("Row r: 5000 input tokens exceed limit 4096")
engine = TorchEngine(FakeTorch, None, None, {}, Settings(api_token="t" * 40), direct_fn=bad, shared_fn=bad)
with pytest.raises(ValueError, match="exceed limit"):
engine.direct({})
class ReservingTorch(FakeTorch):
class cuda(FakeTorch.cuda):
reserved = 0
emptied = 0
@classmethod
def memory_reserved(cls, _device=0):
return cls.reserved
@classmethod
def empty_cache(cls):
cls.emptied += 1
@pytest.mark.parametrize("reserved_after, released", [(8 * 2**30, 0), (8 * 2**30 + 512 * 2**20, 0),
(8 * 2**30 + 513 * 2**20, 1), (12 * 2**30, 1)])
def test_a_burst_is_returned_to_the_driver_after_the_call(reserved_after, released):
ReservingTorch.cuda.emptied = 0
def scorer(*_args):
ReservingTorch.cuda.reserved = reserved_after
return {"ok": True}
engine = TorchEngine(ReservingTorch, None, None, {}, Settings(api_token="t" * 40),
direct_fn=scorer, shared_fn=scorer, release_above_bytes=8 * 2**30 + 512 * 2**20)
assert engine.direct({}) == {"ok": True}
assert ReservingTorch.cuda.emptied == released
+1218
View File
File diff suppressed because it is too large Load Diff