feat(intern-decision-serve): Intern-Decision-4B behind semif-serve's HTTP surface
Contract, service and tests (fake engine, no GPU). Scores through the checkpoint's own inference.py (DecisionEngine.predict, sha256-pinned); maps semif decisions onto Jev choice questions, packs /decide/shared into calls of at most 16, runs orderings in waves, and keeps semif's error mapping, admission, body limit and hard VRAM cap. Deltas from semif-serve are listed in the contract.
This commit is contained in:
@@ -0,0 +1,48 @@
|
||||
"""Settings.from_env: every value validated at startup, a bad one refused naming the variable.
|
||||
Contract: intern-decision-serve.contract.md § Configuration."""
|
||||
import pytest
|
||||
|
||||
from intern_decision_serve.config import DEFAULT_CHECKPOINT, Settings
|
||||
|
||||
TOKEN = "t" * 40
|
||||
P = "INTERN_DECISION_"
|
||||
|
||||
|
||||
def env(**kw):
|
||||
return {f"{P}API_TOKEN": TOKEN, **{f"{P}{k}": v for k, v in kw.items()}}
|
||||
|
||||
|
||||
def test_defaults():
|
||||
s = Settings.from_env(env())
|
||||
assert (s.api_token, s.checkpoint, s.device, s.vram_cap_gib) == (TOKEN, DEFAULT_CHECKPOINT, "cuda", None)
|
||||
assert (s.max_tokens, s.max_decisions, s.max_body_bytes, s.max_queue) == (8192, 64, 1024 * 1024, 32)
|
||||
assert (s.release_slack_mib, s.keep_vision) == (512, False)
|
||||
|
||||
|
||||
def test_every_value_is_read():
|
||||
s = Settings.from_env(env(CHECKPOINT="/x", DEVICE="cpu", VRAM_CAP_GIB="10.5", MAX_TOKENS="6000",
|
||||
MAX_DECISIONS="32", MAX_BODY_BYTES="2048", MAX_QUEUE="4", RELEASE_SLACK_MIB="0",
|
||||
KEEP_VISION="1"))
|
||||
assert (s.checkpoint, s.device, s.vram_cap_gib, s.max_tokens, s.max_decisions) == ("/x", "cpu", 10.5, 6000, 32)
|
||||
assert (s.max_body_bytes, s.max_queue, s.release_slack_mib, s.keep_vision) == (2048, 4, 0, True)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("token", ["", "short", "x" * 31, "x" * 40 + " ", "x" * 40 + "\n", "x" * 39 + "é"])
|
||||
def test_a_short_or_non_visible_ascii_token_is_refused(token):
|
||||
with pytest.raises(ValueError, match=f"{P}API_TOKEN"):
|
||||
Settings.from_env({f"{P}API_TOKEN": token})
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name, value", [
|
||||
("VRAM_CAP_GIB", "0"), ("VRAM_CAP_GIB", "-1"), ("VRAM_CAP_GIB", "nan"), ("VRAM_CAP_GIB", "inf"),
|
||||
("VRAM_CAP_GIB", "lots"), ("MAX_TOKENS", "0"), ("MAX_TOKENS", "1.5"), ("MAX_DECISIONS", "0"),
|
||||
("MAX_BODY_BYTES", "x"), ("MAX_QUEUE", "-2"), ("RELEASE_SLACK_MIB", "-1"), ("KEEP_VISION", "yes"),
|
||||
("DEVICE", "mps"),
|
||||
])
|
||||
def test_a_bad_value_is_refused_naming_the_variable(name, value):
|
||||
with pytest.raises(ValueError, match=f"{P}{name}"):
|
||||
Settings.from_env(env(**{name: value}))
|
||||
|
||||
|
||||
def test_an_empty_cap_means_uncapped():
|
||||
assert Settings.from_env(env(VRAM_CAP_GIB="")).vram_cap_gib is None
|
||||
Reference in New Issue
Block a user