Order averaging (Prime, after the 739aa03 spike):
- A decision may set orderings: rotations|all (all only for <= 4 options). Every
ordering goes to the engine in one shared batch.
- The reply keeps each native result and adds combined {probabilities (log-mean),
top, agreement, spread}.
- Through the service on SemIf's labelled sets (252 rows): 78.6% -> 88.1%
(group-bootstrap 95% CI +5.1..+14.3). Unanimous agreement is 94.5% accurate.
Fast kernels: flash-linear-attention 0.5.2 and causal-conv1d 1.7.0 are now the
default build. A/B on the empty GPU 3:
- parity with upstream went from 142/144 to 144/144;
- a ~2k-token /decide went from 169 to 92 ms server-side;
- short 3-rotation batches cost ~3-6 ms more.
triton builds a C shim at runtime, so the image carries gcc. Without it the
warm-up failed and startup failed closed.
Heid bug-hunt panel (4/4 arms, thread 01M3H3F4RR7XBP90KQ3A39H4SX), folded:
- Startup validation: VRAM cap 0 no longer means uncapped (C1); limits must be
>= 1 (S1); the token must be visible ASCII (S2); the calibration file must
exist and parse, with T in [0.05, 20] (S8, and C3's NaN leg).
- The body limit is checked before a chunk is kept, and a Unicode-digit
Content-Length no longer crashes (C2, S3).
- Failures while building the response now get the 500 envelope (C3).
- 429 busy past SEMIF_MAX_QUEUE requests in progress (C6).
- The engine releases memory on every non-validation failure, unchained after
gc; an empty OOM message is handled; 'out of memory' RuntimeErrors map to 503
(C4, C5, S9).
- The entry point forces HF_HUB_OFFLINE (S10). README wording fixed (S5, S6).
- New guard tests close the gaps the arms' mutation grids exposed: early stop of
the body read, a shared-route lock, calibration pass-through, the gc cycle,
the exact caps, TorchEngine.load's arch and device checks, and the offline
entry point.
86 tests.
Deployed on fv-ml1 GPU 1: parity 144/144, OOM and burst release verified, shared
capacity 63/51/26/16 rows at ~140/520/1960/3900 prefix tokens.
79 lines
3.5 KiB
Python
79 lines
3.5 KiB
Python
"""Settings.from_env. Contract: semif-serve.contract.md § Configuration, INV-6."""
|
|
import json
|
|
|
|
import pytest
|
|
|
|
from semif_serve.config import DEFAULT_REVISION, Settings
|
|
|
|
TOKEN = "t" * 40
|
|
|
|
|
|
def test_defaults_from_a_minimal_env():
|
|
s = Settings.from_env({"SEMIF_API_TOKEN": TOKEN})
|
|
assert (s.api_token, s.revision, s.device, s.max_tokens, s.max_decisions) == (TOKEN, DEFAULT_REVISION, "cuda", 4096, 64)
|
|
assert s.vram_cap_gib is None and s.calibration == {}
|
|
|
|
|
|
@pytest.mark.parametrize("env", [{}, {"SEMIF_API_TOKEN": "short"}, {"SEMIF_API_TOKEN": "x" * 31}])
|
|
def test_a_missing_or_short_token_is_refused_at_startup(env):
|
|
with pytest.raises(ValueError, match="SEMIF_API_TOKEN"):
|
|
Settings.from_env(env)
|
|
|
|
|
|
def test_numbers_and_calibration_file_are_parsed(tmp_path):
|
|
cal = tmp_path / "cal.json"
|
|
cal.write_text(json.dumps({"triage": 2.5}))
|
|
s = Settings.from_env({"SEMIF_API_TOKEN": TOKEN, "SEMIF_VRAM_CAP_GIB": "12", "SEMIF_MAX_DECISIONS": "8",
|
|
"SEMIF_CALIBRATION": str(cal)})
|
|
assert (s.vram_cap_gib, s.max_decisions, s.calibration) == (12.0, 8, {"triage": 2.5})
|
|
|
|
|
|
@pytest.mark.parametrize("table", [{"w": 0}, {"w": -1.0}, {"w": "2"}, {"w": float("inf")}, ["w", 2.0]])
|
|
def test_a_calibration_table_needs_positive_finite_numbers(tmp_path, table):
|
|
cal = tmp_path / "cal.json"
|
|
cal.write_text(json.dumps(table))
|
|
with pytest.raises(ValueError, match="SEMIF_CALIBRATION"):
|
|
Settings.from_env({"SEMIF_API_TOKEN": TOKEN, "SEMIF_CALIBRATION": str(cal)})
|
|
|
|
|
|
@pytest.mark.parametrize("var, value", [
|
|
("SEMIF_VRAM_CAP_GIB", "0"), ("SEMIF_VRAM_CAP_GIB", "-4"), ("SEMIF_VRAM_CAP_GIB", "nan"), ("SEMIF_VRAM_CAP_GIB", "inf"),
|
|
("SEMIF_MAX_TOKENS", "0"), ("SEMIF_MAX_DECISIONS", "-1"), ("SEMIF_MAX_BODY_BYTES", "0"), ("SEMIF_MAX_QUEUE", "0"),
|
|
("SEMIF_MAX_TOKENS", "lots"), ("SEMIF_VRAM_CAP_GIB", "twelve"),
|
|
])
|
|
def test_out_of_range_or_unparseable_values_are_refused_naming_the_variable(var, value):
|
|
with pytest.raises(ValueError, match=var):
|
|
Settings.from_env({"SEMIF_API_TOKEN": TOKEN, var: value})
|
|
|
|
|
|
@pytest.mark.parametrize("token", ["x" * 31 + "\n", "x" * 30 + "\r\n", "x" * 32 + "\x00", "x" * 16 + " " + "x" * 16,
|
|
"é" * 32])
|
|
def test_a_token_with_non_visible_ascii_is_refused(token):
|
|
with pytest.raises(ValueError, match="SEMIF_API_TOKEN"):
|
|
Settings.from_env({"SEMIF_API_TOKEN": token})
|
|
|
|
|
|
@pytest.mark.parametrize("table", [{"w": True}, {"w": 1e-300}, {"w": 0.04}, {"w": 21}])
|
|
def test_a_temperature_must_be_a_real_number_in_range(tmp_path, table):
|
|
cal = tmp_path / "cal.json"
|
|
cal.write_text(json.dumps(table))
|
|
with pytest.raises(ValueError, match="SEMIF_CALIBRATION"):
|
|
Settings.from_env({"SEMIF_API_TOKEN": TOKEN, "SEMIF_CALIBRATION": str(cal)})
|
|
|
|
|
|
@pytest.mark.parametrize("content", [None, "{not json"])
|
|
def test_a_missing_or_malformed_calibration_file_is_a_named_startup_error(tmp_path, content):
|
|
cal = tmp_path / "cal.json"
|
|
if content is not None:
|
|
cal.write_text(content)
|
|
with pytest.raises(ValueError, match="SEMIF_CALIBRATION"):
|
|
Settings.from_env({"SEMIF_API_TOKEN": TOKEN, "SEMIF_CALIBRATION": str(cal)})
|
|
|
|
|
|
def test_in_range_edges_are_accepted(tmp_path):
|
|
cal = tmp_path / "cal.json"
|
|
cal.write_text(json.dumps({"lo": 0.05, "hi": 20}))
|
|
s = Settings.from_env({"SEMIF_API_TOKEN": "!" + "~" * 31, "SEMIF_VRAM_CAP_GIB": "0.5", "SEMIF_MAX_QUEUE": "1",
|
|
"SEMIF_CALIBRATION": str(cal)})
|
|
assert (s.vram_cap_gib, s.max_queue, s.calibration) == (0.5, 1, {"lo": 0.05, "hi": 20.0})
|