Files
esh-pfi-infrastructure/services/intern-decision-serve/pyproject.toml
T
vh ff552abf1f feat(intern-decision-serve): 0.1.1 adds POST /v1/systemone (Jev wire shape)
Straight passthrough to the checkpoint's own DecisionEngine.predict — never the
semif mapping, whose different prompt would change the answers. Reuses the one
inference thread, bearer auth, MAX_QUEUE, VRAM cap and error envelope; no new
concurrency. 1..16 questions in ONE call (never chunked: Jev questions share a
prompt); images 422; over MAX_TOKENS 422 before the forward. /health advertises
the surface. Response 'model' is a string name@revision (JevBench's runner
hashes it; a dict broke its manifest step).

Acceptance on the live service (see acceptance/systemone-2026-09-30/): JevBench
v1.2.16 typesafe adapter over the 231 public items scores all 202/231, hard
83/111, with 0 changed answers across all 924 rows of the bench's own r1..r4;
controls 401/422x3 (token boundary proven at 7168 pass / 7169 refuse); GPU 1
per-process peak 9,866 MiB under the largest accepted request (budget 9,876);
/decide/shared positive control unchanged. 120 tests green.
2026-09-30 12:58:19 -07:00

53 lines
1.7 KiB
TOML

[project]
name = "intern-decision-serve"
version = "0.1.1"
description = "Intern-Decision-4B (its own inference.py) behind semif-serve's HTTP surface"
requires-python = ">=3.12"
dependencies = [
"fastapi==0.118.0",
"uvicorn==0.37.0",
]
[project.optional-dependencies]
# The real engine: the 2026-09-30 bench stack (semif-serve 0.1.4's torch/transformers plus what
# the checkpoint's own inference.py imports: PIL and the Qwen3.5 processor, which needs torchvision).
model = [
"torch==2.10.0",
"torchvision==0.25.0",
"transformers==5.17.0",
"pillow==12.3.0",
]
# Qwen3.5's fast kernels, as in the bench image (without them transformers runs its slower
# reference PyTorch paths, and the bench numbers were measured with them).
fast = [
"flash-linear-attention==0.5.2",
"causal-conv1d @ https://github.com/Dao-AILab/causal-conv1d/releases/download/v1.7.0/causal_conv1d-1.7.0+cu12torch2.10cxx11abiTRUE-cp312-cp312-linux_x86_64.whl ; sys_platform == 'linux' and platform_machine == 'x86_64'",
]
[dependency-groups]
dev = ["pytest==8.4.2", "httpx==0.28.1"]
[build-system]
requires = ["setuptools>=68"]
build-backend = "setuptools.build_meta"
[tool.setuptools.packages.find]
where = ["src"]
[tool.pytest.ini_options]
testpaths = ["tests"]
[[tool.uv.index]]
name = "pytorch-cu128"
url = "https://download.pytorch.org/whl/cu128"
explicit = true
[tool.uv.sources]
torch = { index = "pytorch-cu128" }
torchvision = { index = "pytorch-cu128" }
[tool.uv]
# Hold the transitive pins to the 2026-09-30 bench image (semif-serve:0.1.4's lock), so the
# service runs the stack its acceptance numbers are compared against.
constraint-dependencies = ["numpy==2.2.6", "huggingface-hub==1.31.0", "regex==2026.9.10", "tokenizers==0.23.2", "safetensors==0.8.0"]