feat(semif): SemIf option-logit decisions on fv-ml1 GPU 1 (Prime)
services/semif-serve is a FastAPI wrapper around SemIf's direct and shared torch scorers (SemIf-OpenJev @ 23cf1f39, MIT). Upstream ships only a batch CLI. The wrapper loads the pinned Qwen3.5-4B (851bf6e8, BF16) once from the offline HF cache and returns SemIf's result dicts unchanged, with an optional per-workload temperature-calibrated view. Contract: semif-serve.contract.md. Built with a short contract, TDD (39 tests, fake engine and fake torch, no GPU) and a heid bug-hunt panel (pending). On the card: - torch 2.10.0+cu128 with sm_120 kernels, which is SemIf's own stack; - a hard 12 GiB VRAM cap. Two defects surfaced only on the card, and each fix is covered by a test: - 0.1.1: an OOM raised as a chained exception kept the failed request's tensors alive (11.9 GiB after the 503). It is now raised unchained, after gc. - 0.1.2: a large request left 12.6 GB reserved on the shared card. After each call, reserved memory over the baseline + 512 MiB is now released. Acceptance against SemIf's committed torch predictions (authored144): - 142/144 same top choice; both misses are exact bf16 ties; - 144/144 identical prompt hashes; - deterministic A-vs-A; - negative control 14/144; - shared vs direct 72/72. 21 binary criteria over one state take 159 ms. The shared-mode capacity table under the cap is in stacks/semif/README.md. The Dockerfile installs dependencies from a manifest with the project version blanked, so a version bump reuses the ~4 GB torch layer. Verified: 41 s rebuild, dependency layer CACHED. DNS: semif.fv.internal. Token: vault semif/api-token.
This commit is contained in:
@@ -0,0 +1,6 @@
|
||||
*
|
||||
!pyproject.toml
|
||||
!uv.lock
|
||||
!src/
|
||||
**/__pycache__
|
||||
src/*.egg-info
|
||||
@@ -0,0 +1,4 @@
|
||||
.venv/
|
||||
.pytest_cache/
|
||||
__pycache__/
|
||||
*.egg-info/
|
||||
@@ -0,0 +1,47 @@
|
||||
# syntax=docker/dockerfile:1
|
||||
# semif-serve: SemIf (pinned commit) behind a small FastAPI service. Contract: semif-serve.contract.md.
|
||||
# docker build -t semif-serve:<version> .
|
||||
# Weights are NOT in the image: the pinned Qwen3.5-4B revision is read from the mounted
|
||||
# HF cache, offline (INV-5).
|
||||
|
||||
# The dependency manifest with semif-serve's own version blanked to 0.0.0. A version bump
|
||||
# then leaves these two files byte-identical, and COPY --from compares CONTENT, so the ~4 GB
|
||||
# torch/CUDA install below stays cached across releases (the same problem augaman hit).
|
||||
# `uv sync` keeps uv's per-package index routing (torch from the cu128 index, everything
|
||||
# else from PyPI). An exported requirements.txt loses that, and then fetches triton from the
|
||||
# wrong index and fails its hash check.
|
||||
FROM python:3.12-slim-bookworm AS deps
|
||||
WORKDIR /deps
|
||||
COPY pyproject.toml uv.lock ./
|
||||
RUN python - <<'EOF'
|
||||
import re, pathlib
|
||||
p = pathlib.Path("pyproject.toml")
|
||||
p.write_text(re.sub(r'(?m)^version = "[^"]+"', 'version = "0.0.0"', p.read_text(), count=1))
|
||||
l = pathlib.Path("uv.lock")
|
||||
l.write_text(re.sub(r'(name = "semif-serve"\nversion = )"[^"]+"', r'\1"0.0.0"', l.read_text(), count=1))
|
||||
EOF
|
||||
|
||||
FROM python:3.12-slim-bookworm
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.6.9 /uv /bin/uv
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy UV_PYTHON_DOWNLOADS=never
|
||||
# git: semif-phase1 installs from a pinned GitHub commit.
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends git ca-certificates \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
WORKDIR /app
|
||||
COPY --from=deps /deps/pyproject.toml /deps/uv.lock ./
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen --no-dev --extra model --no-install-project
|
||||
COPY pyproject.toml uv.lock ./
|
||||
COPY src ./src
|
||||
RUN uv sync --frozen --no-dev --extra model --no-editable --no-cache
|
||||
RUN groupadd --system --gid 10001 semif \
|
||||
&& useradd --system --uid 10001 --gid 10001 --no-create-home --shell /usr/sbin/nologin semif
|
||||
USER semif
|
||||
ENV PATH=/app/.venv/bin:$PATH \
|
||||
HF_HOME=/hf \
|
||||
HF_HUB_OFFLINE=1 \
|
||||
HF_HUB_DISABLE_TELEMETRY=1 \
|
||||
NVIDIA_DRIVER_CAPABILITIES=compute,utility
|
||||
EXPOSE 8000
|
||||
# One worker (INV-2): the model and the inference lock live in this one process.
|
||||
CMD ["uvicorn", "semif_serve.main:app_from_env", "--factory", "--host", "0.0.0.0", "--port", "8000", "--workers", "1"]
|
||||
@@ -0,0 +1,101 @@
|
||||
"""semif-serve acceptance on the real card. Contract § Acceptance.
|
||||
|
||||
Needs SemIf's authored144 rows and their committed torch predictions (same model revision,
|
||||
torch 2.10.0+cu128) from a SemIf checkout at the pinned commit:
|
||||
SEMIF_DIR=<checkout> SEMIF_URL=http://10.251.50.54:8032 SEMIF_TOKEN=... \
|
||||
uv run --with httpx python accept.py out.json
|
||||
|
||||
1 parity our /decide vs their direct-authored144 predictions: top choice, max |Δp|, prompt hash
|
||||
2 noise floor the same 144 rows again (A-vs-A)
|
||||
3 negative option descriptions rotated one place: agreement with the reference MUST drop
|
||||
4 shared one state, many criteria through /decide/shared vs /decide on the same rows
|
||||
5 speed 21 binary criteria over one state, shared vs 21 sequential /decide, 3 runs, after warm-up
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import statistics
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import httpx
|
||||
|
||||
URL, TOKEN, SEMIF = os.environ["SEMIF_URL"], os.environ["SEMIF_TOKEN"], Path(os.environ["SEMIF_DIR"])
|
||||
H = {"Authorization": f"Bearer {TOKEN}"}
|
||||
client = httpx.Client(timeout=300)
|
||||
|
||||
|
||||
def row(r):
|
||||
return {k: r[k] for k in ("id", "state", "question", "options")}
|
||||
|
||||
|
||||
def decide(r):
|
||||
resp = client.post(f"{URL}/decide", json=row(r), headers=H)
|
||||
resp.raise_for_status()
|
||||
return resp.json()
|
||||
|
||||
|
||||
def argmax(p):
|
||||
return max(range(len(p)), key=p.__getitem__)
|
||||
|
||||
|
||||
def compare(ours, ref):
|
||||
agree = sum(argmax(ours[i]["probabilities"]) == argmax(ref[i]["probabilities"]) for i in ref)
|
||||
gap = max(max(abs(a - b) for a, b in zip(ours[i]["probabilities"], ref[i]["probabilities"])) for i in ref)
|
||||
return {"rows": len(ref), "top_choice_agree": agree, "max_abs_prob_gap": gap}
|
||||
|
||||
|
||||
rows = [json.loads(l) for l in (SEMIF / "benchmarks/data/authored144.jsonl").read_text().splitlines() if l.strip()]
|
||||
ref = {p["id"]: p for p in map(json.loads, (SEMIF / "results/raw/predictions/direct-authored144.jsonl").read_text().splitlines()) if p}
|
||||
report = {"url": URL, "health": client.get(f"{URL}/health").json()}
|
||||
|
||||
run_a = {r["id"]: decide(r) for r in rows}
|
||||
run_b = {r["id"]: decide(r) for r in rows}
|
||||
report["1_parity_vs_upstream"] = compare(run_a, ref)
|
||||
report["1_prompt_sha256_equal"] = sum(run_a[i]["prompt_sha256"] == ref[i]["prompt_sha256"] for i in ref)
|
||||
report["2_noise_floor_a_vs_b"] = compare(run_a, run_b)
|
||||
|
||||
rotated = []
|
||||
for r in rows:
|
||||
descs = [o["description"] for o in r["options"]]
|
||||
descs = descs[1:] + descs[:1]
|
||||
rotated.append({**r, "options": [{**o, "description": d} for o, d in zip(r["options"], descs)]})
|
||||
report["3_negative_rotated_options"] = compare({r["id"]: decide(r) for r in rotated}, ref)
|
||||
|
||||
# 4: group authored144 by identical state; score every multi-row group both ways.
|
||||
groups = {}
|
||||
for r in rows:
|
||||
groups.setdefault(json.dumps(r["state"], sort_keys=True), []).append(r)
|
||||
multi = [g for g in groups.values() if len(g) > 1]
|
||||
shared_out = {}
|
||||
for g in multi:
|
||||
resp = client.post(f"{URL}/decide/shared", headers=H, json={
|
||||
"state": g[0]["state"], "decisions": [{k: r[k] for k in ("id", "question", "options")} for r in g]})
|
||||
resp.raise_for_status()
|
||||
shared_out.update({res["id"]: res for res in resp.json()["results"]})
|
||||
report["4_shared_vs_direct"] = {"groups": len(multi), **compare(shared_out, {i: run_a[i] for i in shared_out})}
|
||||
|
||||
# 5: 21 binary criteria over one state
|
||||
state = rows[0]["state"]
|
||||
crit = [{"id": f"c{i}", "question": f"Does the evidence mention item number {i}?",
|
||||
"options": [{"id": "yes", "description": "Yes"}, {"id": "no", "description": "No"}]} for i in range(21)]
|
||||
for _ in range(2): # warm-up
|
||||
client.post(f"{URL}/decide/shared", headers=H, json={"state": state, "decisions": crit}).raise_for_status()
|
||||
shared_s, seq_s, prefix = [], [], None
|
||||
for _ in range(3):
|
||||
t = time.perf_counter()
|
||||
resp = client.post(f"{URL}/decide/shared", headers=H, json={"state": state, "decisions": crit})
|
||||
resp.raise_for_status()
|
||||
shared_s.append(time.perf_counter() - t)
|
||||
prefix = resp.json()["timing"]["prefix_tokens"]
|
||||
t = time.perf_counter()
|
||||
for c in crit:
|
||||
decide({**c, "state": state})
|
||||
seq_s.append(time.perf_counter() - t)
|
||||
report["5_speed_21_binary"] = {
|
||||
"prefix_tokens": prefix,
|
||||
"shared_s": {"runs": shared_s, "median": statistics.median(shared_s)},
|
||||
"sequential_decide_s": {"runs": seq_s, "median": statistics.median(seq_s)},
|
||||
}
|
||||
json.dump(report, open(sys.argv[1], "w"), indent=1)
|
||||
print(json.dumps({k: v for k, v in report.items() if k != "health"}, indent=1))
|
||||
@@ -0,0 +1,63 @@
|
||||
{
|
||||
"url": "http://10.251.50.54:8032",
|
||||
"health": {
|
||||
"status": "ok",
|
||||
"semif_commit": "23cf1f39fc9534fe81437200959b6dfc7106e45a",
|
||||
"model": {
|
||||
"source": "Qwen/Qwen3.5-4B",
|
||||
"revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a",
|
||||
"dtype": "bfloat16",
|
||||
"device": "cuda:0",
|
||||
"torch_version": "2.10.0+cu128",
|
||||
"transformers_version": "5.17.0",
|
||||
"device_name": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition",
|
||||
"allocated_gib": 7.85,
|
||||
"reserved_gib": 7.86
|
||||
},
|
||||
"vram_cap_gib": 12.0,
|
||||
"max_tokens": 4096,
|
||||
"max_decisions": 64,
|
||||
"workloads": []
|
||||
},
|
||||
"1_parity_vs_upstream": {
|
||||
"rows": 144,
|
||||
"top_choice_agree": 142,
|
||||
"max_abs_prob_gap": 0.09262644585476243
|
||||
},
|
||||
"1_prompt_sha256_equal": 144,
|
||||
"2_noise_floor_a_vs_b": {
|
||||
"rows": 144,
|
||||
"top_choice_agree": 144,
|
||||
"max_abs_prob_gap": 0.0
|
||||
},
|
||||
"3_negative_rotated_options": {
|
||||
"rows": 144,
|
||||
"top_choice_agree": 14,
|
||||
"max_abs_prob_gap": 0.9987451392432198
|
||||
},
|
||||
"4_shared_vs_direct": {
|
||||
"groups": 36,
|
||||
"rows": 72,
|
||||
"top_choice_agree": 72,
|
||||
"max_abs_prob_gap": 0.044578713319407104
|
||||
},
|
||||
"5_speed_21_binary": {
|
||||
"prefix_tokens": 62,
|
||||
"shared_s": {
|
||||
"runs": [
|
||||
0.15981742000440136,
|
||||
0.159110098000383,
|
||||
0.15852549100236502
|
||||
],
|
||||
"median": 0.159110098000383
|
||||
},
|
||||
"sequential_decide_s": {
|
||||
"runs": [
|
||||
0.978545692996704,
|
||||
0.9808067879930604,
|
||||
0.9893363219889579
|
||||
],
|
||||
"median": 0.9808067879930604
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
[project]
|
||||
name = "semif-serve"
|
||||
version = "0.1.2"
|
||||
description = "HTTP wrapper around SemIf's direct and shared option-logit scorers"
|
||||
requires-python = ">=3.12"
|
||||
dependencies = [
|
||||
"fastapi==0.118.0",
|
||||
"uvicorn==0.37.0",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
# The real engine. Pulls torch 2.10.0 (cu128) and transformers 5.17.0 through SemIf's exact pins.
|
||||
model = [
|
||||
"semif-phase1 @ git+https://github.com/TheoLeeCJ/SemIf-OpenJev@23cf1f39fc9534fe81437200959b6dfc7106e45a",
|
||||
# Same pin SemIf declares, taken from the cu128 index: SemIf's committed predictions report
|
||||
# torch 2.10.0+cu128, and cu128 carries sm_120 kernels for the Blackwell cards.
|
||||
"torch==2.10.0",
|
||||
]
|
||||
|
||||
[dependency-groups]
|
||||
dev = ["pytest==8.4.2", "httpx==0.28.1"]
|
||||
|
||||
[build-system]
|
||||
requires = ["setuptools>=68"]
|
||||
build-backend = "setuptools.build_meta"
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
where = ["src"]
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
testpaths = ["tests"]
|
||||
|
||||
[[tool.uv.index]]
|
||||
name = "pytorch-cu128"
|
||||
url = "https://download.pytorch.org/whl/cu128"
|
||||
explicit = true
|
||||
|
||||
[tool.uv.sources]
|
||||
torch = { index = "pytorch-cu128" }
|
||||
@@ -0,0 +1,122 @@
|
||||
---
|
||||
title: semif-serve
|
||||
kind: module-contract
|
||||
status: draft
|
||||
owner: infra-ops
|
||||
created: 2026-09-27
|
||||
depends_on:
|
||||
- SemIf-OpenJev (MIT) at commit 23cf1f39fc9534fe81437200959b6dfc7106e45a, package semif_phase1
|
||||
- Qwen/Qwen3.5-4B at revision 851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a, BF16
|
||||
---
|
||||
|
||||
# semif-serve: an HTTP wrapper around SemIf's direct and shared scorers
|
||||
|
||||
## Purpose
|
||||
|
||||
SemIf decides by reading the logits of the option letters after one forward pass.
|
||||
It ships as a batch CLI only. semif-serve loads the model **once** and exposes the
|
||||
two torch scorers over HTTP, so fleet callers can ask typed questions without
|
||||
decoding. It adds nothing to the scoring itself: every score it returns is what
|
||||
`semif_phase1.direct.score` or `semif_phase1.shared.score_shared` returned,
|
||||
unchanged, and it adds an optional calibrated view next to it.
|
||||
|
||||
Operator decisions (Prime, 2026-09-27): runs on fv-ml1 GPU 1 under a hard VRAM
|
||||
cap; built by infra-ops with a light process (this contract → TDD → heid bug
|
||||
hunt); no consumer is named yet.
|
||||
|
||||
## Endpoints
|
||||
|
||||
Every POST takes and returns JSON. Every endpoint except `GET /health` requires
|
||||
`Authorization: Bearer <token>`.
|
||||
|
||||
| method | path | body | success |
|
||||
|---|---|---|---|
|
||||
| GET | `/health` | — | 200 `{status: "ok", semif_commit, model, vram_cap_gib, max_tokens, max_decisions, workloads}` |
|
||||
| POST | `/decide` | one SemIf row `{id, state, question, options[2..16]}` + optional `workload` | 200 the `direct.score` dict + optional `calibrated` |
|
||||
| POST | `/decide/shared` | `{state, decisions: [{id, question, options}], workload?}` | 200 `{results: [...], timing: {...}}` from `score_shared` |
|
||||
|
||||
`options` items are `{id, description}`, as in SemIf. `state` is a nonempty
|
||||
string, object or array. For `/decide/shared`, each decision becomes a SemIf row
|
||||
by adding the shared `state`.
|
||||
|
||||
**Calibration.** `workload` is optional. When it names an entry in the
|
||||
calibration table, each result gains `calibrated: {workload, temperature,
|
||||
probabilities}` with `softmax(option_logits / T)`. The argmax never changes. The
|
||||
native `probabilities` and `probability_status` stay untouched. An unknown
|
||||
`workload` is a 422. With no `workload`, no `calibrated` key appears.
|
||||
|
||||
## Invariants
|
||||
|
||||
- **INV-1 pass-through.** `option_ids`, `probabilities`, `option_logits`,
|
||||
`prompt_sha256`, `prompt_version`, `model` and `probability_status` are exactly
|
||||
what SemIf returned. The wrapper never rewrites a score.
|
||||
- **INV-2 one model, one inference at a time.** The model is loaded at startup,
|
||||
and a process-wide lock serialises every scorer call. The app runs as one worker.
|
||||
Scorer calls run off the event loop, so `/health` answers while one is in progress.
|
||||
- **INV-3 fail-closed startup.** Startup refuses to serve unless the model sits
|
||||
on a CUDA device, torch's arch list includes the card's `sm_XY`, and one
|
||||
warm-up decision scores. `device=cpu` is allowed only when set explicitly.
|
||||
- **INV-4 VRAM cap.** When `SEMIF_VRAM_CAP_GIB` is set, the process is capped at
|
||||
that share of the card (`torch.cuda.set_per_process_memory_fraction`) before
|
||||
the model loads. An out-of-memory error during a request is a 503
|
||||
`out_of_memory`, followed by `torch.cuda.empty_cache()`. The process stays up.
|
||||
**After every scorer call**, when the reserved memory exceeds the post-warm-up
|
||||
baseline by more than 512 MiB, the engine calls `empty_cache()`. A burst must not
|
||||
keep the card's shared headroom: on 2026-09-27 a 64-decision request left the
|
||||
process holding 12.6 GB, leaving scriberr 3.5 GB.
|
||||
- **INV-5 no network at runtime.** Weights come from the mounted HF cache at the
|
||||
pinned revision (`HF_HUB_OFFLINE=1`).
|
||||
- **INV-6 constant-time auth.** Token comparison uses `hmac.compare_digest`. The
|
||||
token is ≥ 32 characters, and startup refuses a shorter one.
|
||||
|
||||
## Limits and errors
|
||||
|
||||
- `SEMIF_MAX_TOKENS` (default 4096) is passed to the scorers. A longer prompt is a
|
||||
422, never truncated (SemIf raises).
|
||||
- `SEMIF_MAX_DECISIONS` (default 64) caps `decisions` per shared request. It must
|
||||
hold 1..max entries, else 422.
|
||||
- Request body ≤ `SEMIF_MAX_BODY_BYTES` (default 1 MiB), else 413.
|
||||
|
||||
| status | code | when |
|
||||
|---|---|---|
|
||||
| 401 | `unauthorized` | missing or wrong bearer |
|
||||
| 413 | `request_too_large` | body over the limit |
|
||||
| 422 | `invalid_request` | bad JSON shape, a SemIf `ValueError` (validation, token limit, tokenisation), unknown workload, too many decisions |
|
||||
| 503 | `out_of_memory` | CUDA OOM during scoring |
|
||||
| 500 | `scoring_failed` | any other scorer exception |
|
||||
|
||||
The error body is `{error: {code, message}}`.
|
||||
|
||||
## Configuration (env)
|
||||
|
||||
`SEMIF_API_TOKEN` (required), `SEMIF_MODEL` (default `Qwen/Qwen3.5-4B`),
|
||||
`SEMIF_REVISION` (default the pinned SHA), `SEMIF_DEVICE` (default `cuda`),
|
||||
`SEMIF_VRAM_CAP_GIB`, `SEMIF_MAX_TOKENS`, `SEMIF_MAX_DECISIONS`,
|
||||
`SEMIF_MAX_BODY_BYTES`, `SEMIF_CALIBRATION` (path to a JSON `{workload: T}`; T > 0).
|
||||
|
||||
## Tests (TDD, fake scorer: no torch, no model)
|
||||
|
||||
auth required on POSTs and not on /health; a short token is refused at startup;
|
||||
`/decide` passes the row through and returns the scorer dict unchanged;
|
||||
`/decide/shared` builds rows with the shared state and returns results + timing;
|
||||
calibration adds `calibrated` and keeps the argmax; an unknown workload → 422; a
|
||||
scorer `ValueError` → 422; the engine's `OutOfMemory` → 503; the torch engine,
|
||||
against a fake torch, turns `torch.cuda.OutOfMemoryError` into an **unchained**
|
||||
`OutOfMemory` and calls `empty_cache()` only after the failed call's tensors are freed
|
||||
(found on the card: a chained exception kept 11.9 GiB allocated after the 503); after
|
||||
a call, reserved memory over baseline + 512 MiB is released and at or under it is left
|
||||
alone; any
|
||||
other exception → 500; malformed
|
||||
JSON or a wrong body shape → 422; too many decisions → 422; an oversized body → 413; requests
|
||||
are serialised (two concurrent calls never overlap inside the scorer); `/health`
|
||||
answers while a scorer call is blocked.
|
||||
|
||||
## Acceptance (on fv-ml1, real model; not unit tests)
|
||||
|
||||
1. **Parity:** our `/decide` over SemIf's `authored144` against their committed
|
||||
torch predictions (top choice and max probability gap).
|
||||
2. **Noise floor:** the same run twice (A-vs-A).
|
||||
3. **Negative control:** shuffled option descriptions must break agreement.
|
||||
4. **Shared vs direct:** the same rows agree within the A-vs-A floor.
|
||||
5. **Speed:** 21 binary criteria over one state, N ≥ 3, p50 + spread.
|
||||
6. **VRAM:** the peak at a 4096-token input sets `SEMIF_VRAM_CAP_GIB`.
|
||||
@@ -0,0 +1,161 @@
|
||||
"""semif-serve HTTP layer. Contract: semif-serve.contract.md."""
|
||||
from __future__ import annotations
|
||||
|
||||
import hmac
|
||||
import math
|
||||
import threading
|
||||
from typing import Any
|
||||
|
||||
from fastapi import FastAPI, Request
|
||||
from fastapi.concurrency import run_in_threadpool
|
||||
from fastapi.responses import JSONResponse
|
||||
from pydantic import BaseModel, ConfigDict, ValidationError
|
||||
|
||||
from .config import SEMIF_COMMIT, Settings
|
||||
from .errors import OutOfMemory
|
||||
|
||||
OPEN_PATHS = frozenset({"/health"})
|
||||
State = str | dict[str, Any] | list[Any]
|
||||
|
||||
|
||||
class Option(BaseModel):
|
||||
model_config = ConfigDict(extra="ignore")
|
||||
id: str
|
||||
description: str
|
||||
|
||||
|
||||
class Decision(BaseModel):
|
||||
model_config = ConfigDict(extra="ignore")
|
||||
id: str
|
||||
question: str
|
||||
options: list[Option]
|
||||
|
||||
def row(self, state: State) -> dict:
|
||||
"""The SemIf row shape: exactly id, state, question, options."""
|
||||
return {"id": self.id, "state": state, "question": self.question,
|
||||
"options": [o.model_dump() for o in self.options]}
|
||||
|
||||
|
||||
class DecideBody(Decision):
|
||||
state: State
|
||||
workload: str | None = None
|
||||
|
||||
|
||||
class SharedBody(BaseModel):
|
||||
model_config = ConfigDict(extra="ignore")
|
||||
state: State
|
||||
decisions: list[Decision]
|
||||
workload: str | None = None
|
||||
|
||||
|
||||
class ApiError(Exception):
|
||||
def __init__(self, status: int, code: str, message: str):
|
||||
super().__init__(message)
|
||||
self.status, self.code, self.message = status, code, message
|
||||
|
||||
|
||||
def error(status: int, code: str, message: str) -> JSONResponse:
|
||||
return JSONResponse(status_code=status, content={"error": {"code": code, "message": message}})
|
||||
|
||||
|
||||
def _first_error(exc: ValidationError) -> str:
|
||||
first = exc.errors()[0]
|
||||
where = ".".join(str(p) for p in first.get("loc", ())) or "body"
|
||||
return f"{where}: {first.get('msg', 'invalid')}"
|
||||
|
||||
|
||||
def calibrated_view(result: dict, workload: str, temperature: float) -> dict:
|
||||
"""softmax(option_logits / T): the native fields are left exactly as SemIf returned them (INV-1)."""
|
||||
scaled = [x / temperature for x in result["option_logits"]]
|
||||
top = max(scaled)
|
||||
weights = [math.exp(x - top) for x in scaled]
|
||||
total = sum(weights)
|
||||
return {"workload": workload, "temperature": temperature, "probabilities": [w / total for w in weights]}
|
||||
|
||||
|
||||
def create_app(settings: Settings, engine: Any) -> FastAPI:
|
||||
app = FastAPI(title="semif-serve")
|
||||
expected = f"Bearer {settings.api_token}".encode()
|
||||
inference = threading.Lock() # INV-2: one scorer call at a time, off the event loop
|
||||
|
||||
def locked(fn, *args):
|
||||
with inference:
|
||||
return fn(*args)
|
||||
|
||||
@app.middleware("http")
|
||||
async def require_bearer(request: Request, call_next):
|
||||
if request.url.path not in OPEN_PATHS:
|
||||
supplied = request.headers.get("authorization", "").encode()
|
||||
if not hmac.compare_digest(supplied, expected): # INV-6
|
||||
return error(401, "unauthorized", "missing or wrong bearer token")
|
||||
return await call_next(request)
|
||||
|
||||
@app.exception_handler(ApiError)
|
||||
async def api_error(_request: Request, exc: ApiError):
|
||||
return error(exc.status, exc.code, exc.message)
|
||||
|
||||
async def read_limited(request: Request) -> bytes:
|
||||
limit = settings.max_body_bytes
|
||||
too_large = ApiError(413, "request_too_large", f"request body exceeds {limit} bytes")
|
||||
declared = request.headers.get("content-length")
|
||||
if declared is not None and declared.isdigit() and int(declared) > limit:
|
||||
raise too_large
|
||||
body = bytearray()
|
||||
async for chunk in request.stream(): # also caps bodies that declare no length
|
||||
body.extend(chunk)
|
||||
if len(body) > limit:
|
||||
raise too_large
|
||||
return bytes(body)
|
||||
|
||||
async def parse(request: Request, model: type[BaseModel]):
|
||||
try:
|
||||
return model.model_validate_json(await read_limited(request))
|
||||
except ValidationError as exc:
|
||||
raise ApiError(422, "invalid_request", _first_error(exc)) from exc
|
||||
|
||||
async def score(fn, *args):
|
||||
"""Run one scorer call in a worker thread under the lock; map its failures to contract codes."""
|
||||
try:
|
||||
return await run_in_threadpool(locked, fn, *args)
|
||||
except ValueError as exc: # SemIf validation, token limit, tokenisation
|
||||
raise ApiError(422, "invalid_request", str(exc)) from exc
|
||||
except OutOfMemory as exc:
|
||||
raise ApiError(503, "out_of_memory", str(exc)) from exc
|
||||
except Exception as exc: # noqa: BLE001 — any other scorer failure
|
||||
raise ApiError(500, "scoring_failed", f"{type(exc).__name__}: {exc}") from exc
|
||||
|
||||
def temperature_for(workload: str | None) -> float | None:
|
||||
if workload is None:
|
||||
return None
|
||||
if workload not in settings.calibration:
|
||||
raise ApiError(422, "invalid_request", f"unknown workload {workload!r}")
|
||||
return settings.calibration[workload]
|
||||
|
||||
def with_calibration(result: dict, workload: str | None, temperature: float | None) -> dict:
|
||||
if temperature is None:
|
||||
return result
|
||||
return {**result, "calibrated": calibrated_view(result, workload, temperature)}
|
||||
|
||||
@app.get("/health")
|
||||
async def health():
|
||||
return {"status": "ok", "semif_commit": SEMIF_COMMIT, "model": engine.health(),
|
||||
"vram_cap_gib": settings.vram_cap_gib, "max_tokens": settings.max_tokens,
|
||||
"max_decisions": settings.max_decisions, "workloads": sorted(settings.calibration)}
|
||||
|
||||
@app.post("/decide")
|
||||
async def decide(request: Request):
|
||||
body = await parse(request, DecideBody)
|
||||
temperature = temperature_for(body.workload)
|
||||
return with_calibration(await score(engine.direct, body.row(body.state)), body.workload, temperature)
|
||||
|
||||
@app.post("/decide/shared")
|
||||
async def decide_shared(request: Request):
|
||||
body = await parse(request, SharedBody)
|
||||
if not 1 <= len(body.decisions) <= settings.max_decisions:
|
||||
raise ApiError(422, "invalid_request",
|
||||
f"decisions must hold 1..{settings.max_decisions} entries, got {len(body.decisions)}")
|
||||
temperature = temperature_for(body.workload)
|
||||
results, timing = await score(engine.shared, [d.row(body.state) for d in body.decisions])
|
||||
return {"results": [with_calibration(r, body.workload, temperature) for r in results], "timing": timing}
|
||||
|
||||
return app
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Settings for semif-serve. Contract: semif-serve.contract.md § Configuration."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
MIN_TOKEN_CHARS = 32
|
||||
SEMIF_COMMIT = "23cf1f39fc9534fe81437200959b6dfc7106e45a"
|
||||
DEFAULT_MODEL = "Qwen/Qwen3.5-4B"
|
||||
DEFAULT_REVISION = "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Settings:
|
||||
api_token: str
|
||||
model: str = DEFAULT_MODEL
|
||||
revision: str = DEFAULT_REVISION
|
||||
device: str = "cuda"
|
||||
vram_cap_gib: float | None = None
|
||||
max_tokens: int = 4096
|
||||
max_decisions: int = 64
|
||||
max_body_bytes: int = 1024 * 1024
|
||||
calibration: dict[str, float] = field(default_factory=dict)
|
||||
|
||||
@classmethod
|
||||
def from_env(cls, env: Mapping[str, str]) -> "Settings":
|
||||
token = env.get("SEMIF_API_TOKEN", "")
|
||||
if len(token) < MIN_TOKEN_CHARS: # INV-6
|
||||
raise ValueError(f"SEMIF_API_TOKEN must be at least {MIN_TOKEN_CHARS} characters")
|
||||
cap = env.get("SEMIF_VRAM_CAP_GIB")
|
||||
return cls(
|
||||
api_token=token,
|
||||
model=env.get("SEMIF_MODEL", DEFAULT_MODEL),
|
||||
revision=env.get("SEMIF_REVISION", DEFAULT_REVISION),
|
||||
device=env.get("SEMIF_DEVICE", "cuda"),
|
||||
vram_cap_gib=float(cap) if cap else None,
|
||||
max_tokens=int(env.get("SEMIF_MAX_TOKENS", 4096)),
|
||||
max_decisions=int(env.get("SEMIF_MAX_DECISIONS", 64)),
|
||||
max_body_bytes=int(env.get("SEMIF_MAX_BODY_BYTES", 1024 * 1024)),
|
||||
calibration=_load_calibration(env.get("SEMIF_CALIBRATION")),
|
||||
)
|
||||
|
||||
|
||||
def _load_calibration(path: str | None) -> dict[str, float]:
|
||||
"""{workload: T}, every T a finite number > 0 (T scales option logits before softmax)."""
|
||||
if not path:
|
||||
return {}
|
||||
table = json.loads(Path(path).read_text())
|
||||
if not isinstance(table, dict) or not all(
|
||||
isinstance(t, (int, float)) and not isinstance(t, bool) and math.isfinite(t) and t > 0
|
||||
for t in table.values()
|
||||
):
|
||||
raise ValueError("SEMIF_CALIBRATION must be a JSON object of workload -> finite temperature > 0")
|
||||
return {str(k): float(v) for k, v in table.items()}
|
||||
@@ -0,0 +1,101 @@
|
||||
"""The real engine: SemIf's torch scorers over one resident model. Needs the `model` extra.
|
||||
|
||||
Contract: semif-serve.contract.md, INV-3 (fail-closed startup), INV-4 (VRAM cap + OOM),
|
||||
INV-5 (offline weights). load() is checked on the card at acceptance; the OOM path is
|
||||
unit-tested against a fake torch (tests/test_engine.py).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import gc
|
||||
from typing import Any, Callable
|
||||
|
||||
from .config import Settings
|
||||
from .errors import OutOfMemory
|
||||
|
||||
RELEASE_SLACK_BYTES = 512 * 2**20
|
||||
WARMUP_ROW = {
|
||||
"id": "semif-serve-warmup",
|
||||
"state": "The deployment completed at 14:02 UTC. Health checks passed in all three zones.",
|
||||
"question": "Is there evidence that the deployment succeeded?",
|
||||
"options": [
|
||||
{"id": "yes", "description": "The deployment succeeded."},
|
||||
{"id": "no", "description": "The deployment did not succeed."},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
class TorchEngine:
|
||||
def __init__(self, torch: Any, model: Any, tokenizer: Any, metadata: dict, settings: Settings,
|
||||
direct_fn: Callable, shared_fn: Callable, release_above_bytes: int | None = None):
|
||||
self._torch, self._model, self._tokenizer = torch, model, tokenizer
|
||||
self._metadata, self._settings = metadata, settings
|
||||
self._direct, self._shared = direct_fn, shared_fn
|
||||
self._release_above = release_above_bytes
|
||||
|
||||
@classmethod
|
||||
def load(cls, settings: Settings) -> "TorchEngine":
|
||||
import torch
|
||||
from semif_phase1.core import load_causal_model
|
||||
from semif_phase1.direct import score
|
||||
from semif_phase1.shared import score_shared
|
||||
|
||||
if settings.device == "cuda":
|
||||
if not torch.cuda.is_available():
|
||||
raise RuntimeError("SEMIF_DEVICE=cuda but torch sees no CUDA device")
|
||||
major, minor = torch.cuda.get_device_capability(0)
|
||||
arch = f"sm_{major}{minor}"
|
||||
if arch not in torch.cuda.get_arch_list(): # INV-3: no silent PTX/CPU fallback
|
||||
raise RuntimeError(f"torch {torch.__version__} has no kernels for {arch}: {torch.cuda.get_arch_list()}")
|
||||
if settings.vram_cap_gib: # INV-4: cap BEFORE the weights land
|
||||
total = torch.cuda.get_device_properties(0).total_memory
|
||||
fraction = settings.vram_cap_gib * 2**30 / total
|
||||
if not 0 < fraction <= 1:
|
||||
raise ValueError(f"SEMIF_VRAM_CAP_GIB={settings.vram_cap_gib} does not fit a {total / 2**30:.1f} GiB card")
|
||||
torch.cuda.set_per_process_memory_fraction(fraction, 0)
|
||||
elif settings.device != "cpu":
|
||||
raise ValueError(f"SEMIF_DEVICE must be cuda or cpu, not {settings.device!r}")
|
||||
|
||||
model, tokenizer, metadata = load_causal_model(settings.model, settings.revision, settings.device, "bfloat16")
|
||||
placed = next(model.parameters()).device.type
|
||||
if placed != settings.device: # INV-3
|
||||
raise RuntimeError(f"model landed on {placed}, expected {settings.device}")
|
||||
engine = cls(torch, model, tokenizer, metadata, settings, direct_fn=score, shared_fn=score_shared)
|
||||
engine.direct(WARMUP_ROW) # INV-3: one decision must score
|
||||
if settings.device == "cuda": # INV-4: the resting footprint
|
||||
engine._release_above = torch.cuda.memory_reserved(0) + RELEASE_SLACK_BYTES
|
||||
return engine
|
||||
|
||||
def health(self) -> dict:
|
||||
info = dict(self._metadata)
|
||||
if self._settings.device == "cuda":
|
||||
info["device_name"] = self._torch.cuda.get_device_name(0)
|
||||
info["allocated_gib"] = round(self._torch.cuda.memory_allocated(0) / 2**30, 2)
|
||||
info["reserved_gib"] = round(self._torch.cuda.memory_reserved(0) / 2**30, 2)
|
||||
return info
|
||||
|
||||
def _release_burst(self) -> None:
|
||||
"""INV-4: hand a burst back to the driver so the card's shared headroom (scriberr, the
|
||||
vLLM seats) returns after a big request, instead of sitting in torch's cache."""
|
||||
if self._release_above is not None and self._torch.cuda.memory_reserved(0) > self._release_above:
|
||||
self._torch.cuda.empty_cache()
|
||||
|
||||
def _guard(self, fn, *args):
|
||||
try:
|
||||
result = fn(*args)
|
||||
except self._torch.cuda.OutOfMemoryError as exc:
|
||||
message = str(exc).splitlines()[0]
|
||||
else:
|
||||
self._release_burst()
|
||||
return result
|
||||
# INV-4, outside the except block on purpose: the torch exception's traceback holds the
|
||||
# failed scorer's frames, and with them its tensors (the replicated prefix cache). Raising
|
||||
# inside the block, or `from exc`, would chain to it and keep GiBs allocated after the 503.
|
||||
gc.collect()
|
||||
self._torch.cuda.empty_cache()
|
||||
raise OutOfMemory(message)
|
||||
|
||||
def direct(self, row: dict) -> dict:
|
||||
return self._guard(self._direct, self._model, self._tokenizer, row, self._metadata, self._settings.max_tokens)
|
||||
|
||||
def shared(self, rows: list[dict]) -> tuple[list[dict], dict]:
|
||||
return self._guard(self._shared, self._model, self._tokenizer, rows, self._metadata, self._settings.max_tokens)
|
||||
@@ -0,0 +1,5 @@
|
||||
"""Torch-free exceptions shared by the HTTP layer and the engine."""
|
||||
|
||||
|
||||
class OutOfMemory(RuntimeError):
|
||||
"""The engine ran out of GPU memory during a request and has already released its cache (INV-4)."""
|
||||
@@ -0,0 +1,16 @@
|
||||
"""uvicorn entry point: `uvicorn semif_serve.main:app_from_env --factory --workers 1`."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
from fastapi import FastAPI
|
||||
|
||||
from .app import create_app
|
||||
from .config import Settings
|
||||
|
||||
|
||||
def app_from_env() -> FastAPI:
|
||||
settings = Settings.from_env(os.environ)
|
||||
from .engine import TorchEngine # torch loads only here, never in the unit tests
|
||||
|
||||
return create_app(settings, TorchEngine.load(settings))
|
||||
@@ -0,0 +1,233 @@
|
||||
"""semif-serve HTTP behaviour against a fake engine (no torch, no model).
|
||||
Contract: services/semif-serve/semif-serve.contract.md"""
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from semif_serve.app import create_app
|
||||
from semif_serve.config import Settings
|
||||
from semif_serve.errors import OutOfMemory
|
||||
|
||||
TOKEN = "t" * 40
|
||||
AUTH = {"Authorization": f"Bearer {TOKEN}"}
|
||||
OPTIONS = [{"id": "yes", "description": "Yes."}, {"id": "no", "description": "No."}]
|
||||
ROW = {"id": "r1", "state": "The deploy passed.", "question": "Did it pass?", "options": OPTIONS}
|
||||
|
||||
|
||||
class FakeEngine:
|
||||
"""Returns canned SemIf-shaped dicts and records what it was asked."""
|
||||
|
||||
def __init__(self, logits=(2.0, 0.0)):
|
||||
self.logits = list(logits)
|
||||
self.direct_calls = []
|
||||
self.shared_calls = []
|
||||
|
||||
def health(self):
|
||||
return {"source": "fake/model", "revision": "0" * 40}
|
||||
|
||||
def direct(self, row):
|
||||
self.direct_calls.append(row)
|
||||
return {
|
||||
"id": row["id"],
|
||||
"option_ids": [o["id"] for o in row["options"]],
|
||||
"probabilities": [0.8807970779778823, 0.11920292202211769],
|
||||
"option_logits": self.logits,
|
||||
"prompt_sha256": "ab" * 32,
|
||||
"prompt_version": "direct-options-v1",
|
||||
"model": {"source": "fake/model"},
|
||||
"probability_status": "conditional option score; uncalibrated as decision confidence",
|
||||
}
|
||||
|
||||
def shared(self, rows):
|
||||
self.shared_calls.append(rows)
|
||||
return [self.direct(r) for r in rows], {"prefix_tokens": 7, "batch_size": len(rows)}
|
||||
|
||||
|
||||
def make_client(engine=None, **overrides):
|
||||
settings = Settings(api_token=TOKEN, **overrides)
|
||||
return TestClient(create_app(settings, engine or FakeEngine()))
|
||||
|
||||
|
||||
def test_decide_passes_the_row_through_and_returns_the_scorer_dict_unchanged():
|
||||
engine = FakeEngine()
|
||||
client = make_client(engine)
|
||||
response = client.post("/decide", json=ROW, headers=AUTH)
|
||||
assert response.status_code == 200
|
||||
assert response.json() == engine.direct(ROW)
|
||||
assert engine.direct_calls[0] == ROW
|
||||
|
||||
|
||||
@pytest.mark.parametrize("headers", [{}, {"Authorization": "Bearer wrong"}, {"Authorization": TOKEN}])
|
||||
def test_posts_without_the_right_bearer_are_401_and_never_reach_the_engine(headers):
|
||||
engine = FakeEngine()
|
||||
client = make_client(engine)
|
||||
for path in ("/decide", "/decide/shared"):
|
||||
response = client.post(path, json=ROW, headers=headers)
|
||||
assert response.status_code == 401
|
||||
assert response.json() == {"error": {"code": "unauthorized", "message": response.json()["error"]["message"]}}
|
||||
assert engine.direct_calls == []
|
||||
|
||||
|
||||
def test_health_needs_no_auth():
|
||||
response = make_client().get("/health")
|
||||
assert response.status_code == 200
|
||||
assert response.json()["status"] == "ok"
|
||||
|
||||
|
||||
def test_shared_builds_semif_rows_from_the_shared_state_and_returns_results_and_timing():
|
||||
engine = FakeEngine()
|
||||
body = {"state": {"deploy": "passed"},
|
||||
"decisions": [{"id": "a", "question": "Q1?", "options": OPTIONS},
|
||||
{"id": "b", "question": "Q2?", "options": OPTIONS}]}
|
||||
response = make_client(engine).post("/decide/shared", json=body, headers=AUTH)
|
||||
assert response.status_code == 200
|
||||
assert engine.shared_calls == [[
|
||||
{"id": "a", "state": {"deploy": "passed"}, "question": "Q1?", "options": OPTIONS},
|
||||
{"id": "b", "state": {"deploy": "passed"}, "question": "Q2?", "options": OPTIONS},
|
||||
]]
|
||||
out = response.json()
|
||||
assert [r["id"] for r in out["results"]] == ["a", "b"]
|
||||
assert out["timing"] == {"prefix_tokens": 7, "batch_size": 2}
|
||||
|
||||
|
||||
def test_a_known_workload_adds_a_calibrated_view_and_leaves_the_native_scores_alone():
|
||||
import math
|
||||
engine = FakeEngine(logits=(3.0, 1.0))
|
||||
client = make_client(engine, calibration={"triage": 2.0})
|
||||
native = engine.direct(ROW)
|
||||
for path, body, pick in (
|
||||
("/decide", {**ROW, "workload": "triage"}, lambda j: [j]),
|
||||
("/decide/shared", {"state": ROW["state"], "workload": "triage",
|
||||
"decisions": [{"id": "a", "question": "Q?", "options": OPTIONS}]}, lambda j: j["results"]),
|
||||
):
|
||||
for result in pick(client.post(path, json=body, headers=AUTH).json()):
|
||||
cal = result.pop("calibrated")
|
||||
assert (cal["workload"], cal["temperature"]) == ("triage", 2.0)
|
||||
e = [math.exp(1.5), math.exp(0.5)]
|
||||
assert cal["probabilities"] == pytest.approx([e[0] / sum(e), e[1] / sum(e)])
|
||||
assert cal["probabilities"].index(max(cal["probabilities"])) == 0
|
||||
assert {k: result[k] for k in ("probabilities", "option_logits", "probability_status")} == \
|
||||
{k: native[k] for k in ("probabilities", "option_logits", "probability_status")}
|
||||
|
||||
|
||||
def test_no_workload_means_no_calibrated_key():
|
||||
client = make_client(calibration={"triage": 2.0})
|
||||
assert "calibrated" not in client.post("/decide", json=ROW, headers=AUTH).json()
|
||||
|
||||
|
||||
def test_an_unknown_workload_is_422_before_the_engine_runs():
|
||||
engine = FakeEngine()
|
||||
client = make_client(engine, calibration={"triage": 2.0})
|
||||
response = client.post("/decide", json={**ROW, "workload": "nope"}, headers=AUTH)
|
||||
assert response.status_code == 422
|
||||
assert response.json()["error"]["code"] == "invalid_request"
|
||||
assert engine.direct_calls == []
|
||||
|
||||
|
||||
class RaisingEngine(FakeEngine):
|
||||
def __init__(self, exc):
|
||||
super().__init__()
|
||||
self.exc = exc
|
||||
|
||||
def direct(self, row):
|
||||
raise self.exc
|
||||
|
||||
def shared(self, rows):
|
||||
raise self.exc
|
||||
|
||||
|
||||
@pytest.mark.parametrize("exc, status, code", [
|
||||
(ValueError("Row r1: 5000 input tokens exceed limit 4096; no truncation allowed"), 422, "invalid_request"),
|
||||
(OutOfMemory("CUDA out of memory"), 503, "out_of_memory"),
|
||||
(RuntimeError("Invalid native prefix cache"), 500, "scoring_failed"),
|
||||
])
|
||||
def test_scorer_failures_map_to_the_contract_status_codes(exc, status, code):
|
||||
client = make_client(RaisingEngine(exc))
|
||||
for path, body in (("/decide", ROW),
|
||||
("/decide/shared", {"state": "s", "decisions": [{"id": "a", "question": "Q?", "options": OPTIONS}]})):
|
||||
response = client.post(path, json=body, headers=AUTH)
|
||||
assert response.status_code == status
|
||||
assert response.json()["error"]["code"] == code
|
||||
assert str(exc) in response.json()["error"]["message"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("content", [b"{not json", b'{"id": "r1"}', b'{"state": "s", "decisions": "nope"}', b"[1, 2]"])
|
||||
def test_malformed_json_or_a_wrong_shape_is_422(content):
|
||||
client = make_client()
|
||||
for path in ("/decide", "/decide/shared"):
|
||||
response = client.post(path, content=content, headers={**AUTH, "content-type": "application/json"})
|
||||
assert response.status_code == 422
|
||||
assert response.json()["error"]["code"] == "invalid_request"
|
||||
|
||||
|
||||
def test_more_decisions_than_the_cap_is_422_and_the_engine_never_runs():
|
||||
engine = FakeEngine()
|
||||
client = make_client(engine, max_decisions=2)
|
||||
decisions = [{"id": str(i), "question": "Q?", "options": OPTIONS} for i in range(3)]
|
||||
response = client.post("/decide/shared", json={"state": "s", "decisions": decisions}, headers=AUTH)
|
||||
assert response.status_code == 422
|
||||
assert response.json()["error"]["code"] == "invalid_request"
|
||||
assert engine.shared_calls == []
|
||||
|
||||
|
||||
def test_an_empty_decision_list_is_422():
|
||||
response = make_client().post("/decide/shared", json={"state": "s", "decisions": []}, headers=AUTH)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
@pytest.mark.parametrize("chunked", [False, True])
|
||||
def test_a_body_over_the_limit_is_413_whether_or_not_it_declares_its_length(chunked):
|
||||
engine = FakeEngine()
|
||||
client = make_client(engine, max_body_bytes=200)
|
||||
body = ('{"id": "r1", "state": "' + "x" * 500 + '", "question": "Q?", "options": []}').encode()
|
||||
content = (chunk for chunk in [body[:100], body[100:]]) if chunked else body
|
||||
response = client.post("/decide", content=content, headers={**AUTH, "content-type": "application/json"})
|
||||
assert response.status_code == 413
|
||||
assert response.json()["error"]["code"] == "request_too_large"
|
||||
assert engine.direct_calls == []
|
||||
|
||||
|
||||
class SlowEngine(FakeEngine):
|
||||
"""Holds each scorer call until released, and records the peak number of calls inside at once."""
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
import threading
|
||||
self.inside = 0
|
||||
self.peak = 0
|
||||
self.guard = threading.Lock()
|
||||
self.release = threading.Event()
|
||||
self.entered = threading.Event()
|
||||
|
||||
def direct(self, row):
|
||||
with self.guard:
|
||||
self.inside += 1
|
||||
self.peak = max(self.peak, self.inside)
|
||||
self.entered.set()
|
||||
self.release.wait(5)
|
||||
with self.guard:
|
||||
self.inside -= 1
|
||||
return super().direct(row)
|
||||
|
||||
|
||||
def test_concurrent_requests_never_overlap_inside_the_scorer_and_health_still_answers():
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
engine = SlowEngine()
|
||||
with make_client(engine) as client, ThreadPoolExecutor(4) as pool:
|
||||
futures = [pool.submit(client.post, "/decide", json={**ROW, "id": f"r{i}"}, headers=AUTH) for i in range(4)]
|
||||
assert engine.entered.wait(5)
|
||||
import time
|
||||
started = time.monotonic()
|
||||
assert client.get("/health").status_code == 200
|
||||
assert time.monotonic() - started < 1.0 # answered while the scorer is still held
|
||||
assert not engine.release.is_set() and engine.inside == 1
|
||||
engine.release.set()
|
||||
assert [f.result().status_code for f in futures] == [200] * 4
|
||||
assert engine.peak == 1
|
||||
|
||||
|
||||
def test_health_reports_the_pins_limits_and_workloads():
|
||||
from semif_serve.config import SEMIF_COMMIT
|
||||
client = make_client(vram_cap_gib=12.0, max_decisions=8, calibration={"triage": 2.0, "alerts": 1.3})
|
||||
body = client.get("/health").json()
|
||||
assert body == {"status": "ok", "semif_commit": SEMIF_COMMIT, "model": FakeEngine().health(),
|
||||
"vram_cap_gib": 12.0, "max_tokens": 4096, "max_decisions": 8, "workloads": ["alerts", "triage"]}
|
||||
@@ -0,0 +1,36 @@
|
||||
"""Settings.from_env. Contract: semif-serve.contract.md § Configuration, INV-6."""
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from semif_serve.config import DEFAULT_REVISION, Settings
|
||||
|
||||
TOKEN = "t" * 40
|
||||
|
||||
|
||||
def test_defaults_from_a_minimal_env():
|
||||
s = Settings.from_env({"SEMIF_API_TOKEN": TOKEN})
|
||||
assert (s.api_token, s.revision, s.device, s.max_tokens, s.max_decisions) == (TOKEN, DEFAULT_REVISION, "cuda", 4096, 64)
|
||||
assert s.vram_cap_gib is None and s.calibration == {}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("env", [{}, {"SEMIF_API_TOKEN": "short"}, {"SEMIF_API_TOKEN": "x" * 31}])
|
||||
def test_a_missing_or_short_token_is_refused_at_startup(env):
|
||||
with pytest.raises(ValueError, match="SEMIF_API_TOKEN"):
|
||||
Settings.from_env(env)
|
||||
|
||||
|
||||
def test_numbers_and_calibration_file_are_parsed(tmp_path):
|
||||
cal = tmp_path / "cal.json"
|
||||
cal.write_text(json.dumps({"triage": 2.5}))
|
||||
s = Settings.from_env({"SEMIF_API_TOKEN": TOKEN, "SEMIF_VRAM_CAP_GIB": "12", "SEMIF_MAX_DECISIONS": "8",
|
||||
"SEMIF_CALIBRATION": str(cal)})
|
||||
assert (s.vram_cap_gib, s.max_decisions, s.calibration) == (12.0, 8, {"triage": 2.5})
|
||||
|
||||
|
||||
@pytest.mark.parametrize("table", [{"w": 0}, {"w": -1.0}, {"w": "2"}, {"w": float("inf")}, ["w", 2.0]])
|
||||
def test_a_calibration_table_needs_positive_finite_numbers(tmp_path, table):
|
||||
cal = tmp_path / "cal.json"
|
||||
cal.write_text(json.dumps(table))
|
||||
with pytest.raises(ValueError, match="SEMIF_CALIBRATION"):
|
||||
Settings.from_env({"SEMIF_API_TOKEN": TOKEN, "SEMIF_CALIBRATION": str(cal)})
|
||||
@@ -0,0 +1,83 @@
|
||||
"""TorchEngine's OOM path against a fake torch (INV-4). Found on the card 2026-09-27: after a
|
||||
503 the failed call's tensors stayed alive (11.92 GiB allocated) because the raised
|
||||
OutOfMemory chained back to the torch exception, whose traceback held the scorer's frames."""
|
||||
import weakref
|
||||
|
||||
import pytest
|
||||
|
||||
from semif_serve.config import Settings
|
||||
from semif_serve.engine import TorchEngine
|
||||
from semif_serve.errors import OutOfMemory
|
||||
|
||||
|
||||
class FakeTorch:
|
||||
class cuda:
|
||||
class OutOfMemoryError(RuntimeError):
|
||||
pass
|
||||
|
||||
empties = [] # for each empty_cache() call: was the failed call's tensor already freed?
|
||||
watched = []
|
||||
|
||||
@classmethod
|
||||
def empty_cache(cls):
|
||||
cls.empties.append(all(ref() is None for ref in cls.watched))
|
||||
|
||||
|
||||
class Tensor:
|
||||
pass
|
||||
|
||||
|
||||
def failing_scorer(*_args):
|
||||
kv_cache = Tensor() # stands in for the replicated prefix cache
|
||||
FakeTorch.cuda.watched.append(weakref.ref(kv_cache))
|
||||
raise FakeTorch.cuda.OutOfMemoryError("CUDA out of memory. Tried to allocate 490.00 MiB.\nGPU 0 has ...")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("call", ["direct", "shared"])
|
||||
def test_oom_frees_the_failed_call_before_emptying_the_cache_and_keeps_nothing_alive(call):
|
||||
FakeTorch.cuda.empties.clear(), FakeTorch.cuda.watched.clear()
|
||||
engine = TorchEngine(FakeTorch, model=None, tokenizer=None, metadata={}, settings=Settings(api_token="t" * 40),
|
||||
direct_fn=failing_scorer, shared_fn=failing_scorer)
|
||||
with pytest.raises(OutOfMemory) as info:
|
||||
getattr(engine, call)({})
|
||||
assert str(info.value) == "CUDA out of memory. Tried to allocate 490.00 MiB."
|
||||
assert info.value.__cause__ is None and info.value.__context__ is None # no chain back to the frames
|
||||
assert FakeTorch.cuda.watched[0]() is None # nothing keeps the tensor alive
|
||||
assert FakeTorch.cuda.empties == [True] # emptied once, after it was freed
|
||||
|
||||
|
||||
def test_other_scorer_errors_pass_through_unchanged():
|
||||
def bad(*_args):
|
||||
raise ValueError("Row r: 5000 input tokens exceed limit 4096")
|
||||
engine = TorchEngine(FakeTorch, None, None, {}, Settings(api_token="t" * 40), direct_fn=bad, shared_fn=bad)
|
||||
with pytest.raises(ValueError, match="exceed limit"):
|
||||
engine.direct({})
|
||||
|
||||
|
||||
class ReservingTorch(FakeTorch):
|
||||
class cuda(FakeTorch.cuda):
|
||||
reserved = 0
|
||||
emptied = 0
|
||||
|
||||
@classmethod
|
||||
def memory_reserved(cls, _device=0):
|
||||
return cls.reserved
|
||||
|
||||
@classmethod
|
||||
def empty_cache(cls):
|
||||
cls.emptied += 1
|
||||
|
||||
|
||||
@pytest.mark.parametrize("reserved_after, released", [(8 * 2**30, 0), (8 * 2**30 + 512 * 2**20, 0),
|
||||
(8 * 2**30 + 513 * 2**20, 1), (12 * 2**30, 1)])
|
||||
def test_a_burst_is_returned_to_the_driver_after_the_call(reserved_after, released):
|
||||
ReservingTorch.cuda.emptied = 0
|
||||
|
||||
def scorer(*_args):
|
||||
ReservingTorch.cuda.reserved = reserved_after
|
||||
return {"ok": True}
|
||||
|
||||
engine = TorchEngine(ReservingTorch, None, None, {}, Settings(api_token="t" * 40),
|
||||
direct_fn=scorer, shared_fn=scorer, release_above_bytes=8 * 2**30 + 512 * 2**20)
|
||||
assert engine.direct({}) == {"ok": True}
|
||||
assert ReservingTorch.cuda.emptied == released
|
||||
Generated
+1218
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user