The ~6.5-9 s first-call-per-bucket autotune lived in the container's writable layer and died on every recreate. 0.1.3 creates /tmp/triton-cache in the image owned by 10001 so the named volume intern-decision_triton-cache inherits a writable mount point, and compose mounts it. Bucket model PROVEN, not inferred: 2,048-token buckets, 16 up to 32,768. After one warmed call per bucket, 12 random sizes across 8k-32k were all warm (worst 2.09 s); cold entries cost 6.5-9 s. Full cold warm-up 109 s; warm re-run 17 s. scripts/intern-decision-warmup: one noul call per bucket, MAX_TOKENS from /health, two-point live calibration of the tokenizer's linear token model (a single probe overcorrects and the aim oscillates around the bucket edge), per-bucket wall times, non-zero exit on a missed bucket. Run it after an IMAGE CHANGE only; the volume carries ordinary recreates (measured: force-recreate, then a warmed 32k call answered in 2.11 s). Acceptance on 0.1.3: JevBench 202/231, hard 83/111, 0 diffs / 924; warm 32k GPU 1 peak 15,218 MiB (budget 15,220; a COLD autotune touched 15,224 once, README caveat); /decide answers. Artifacts in the acceptance dir.
53 lines
1.7 KiB
TOML
53 lines
1.7 KiB
TOML
[project]
|
|
name = "intern-decision-serve"
|
|
version = "0.1.3"
|
|
description = "Intern-Decision-4B (its own inference.py) behind semif-serve's HTTP surface"
|
|
requires-python = ">=3.12"
|
|
dependencies = [
|
|
"fastapi==0.118.0",
|
|
"uvicorn==0.37.0",
|
|
]
|
|
|
|
[project.optional-dependencies]
|
|
# The real engine: the 2026-09-30 bench stack (semif-serve 0.1.4's torch/transformers plus what
|
|
# the checkpoint's own inference.py imports: PIL and the Qwen3.5 processor, which needs torchvision).
|
|
model = [
|
|
"torch==2.10.0",
|
|
"torchvision==0.25.0",
|
|
"transformers==5.17.0",
|
|
"pillow==12.3.0",
|
|
]
|
|
# Qwen3.5's fast kernels, as in the bench image (without them transformers runs its slower
|
|
# reference PyTorch paths, and the bench numbers were measured with them).
|
|
fast = [
|
|
"flash-linear-attention==0.5.2",
|
|
"causal-conv1d @ https://github.com/Dao-AILab/causal-conv1d/releases/download/v1.7.0/causal_conv1d-1.7.0+cu12torch2.10cxx11abiTRUE-cp312-cp312-linux_x86_64.whl ; sys_platform == 'linux' and platform_machine == 'x86_64'",
|
|
]
|
|
|
|
[dependency-groups]
|
|
dev = ["pytest==8.4.2", "httpx==0.28.1"]
|
|
|
|
[build-system]
|
|
requires = ["setuptools>=68"]
|
|
build-backend = "setuptools.build_meta"
|
|
|
|
[tool.setuptools.packages.find]
|
|
where = ["src"]
|
|
|
|
[tool.pytest.ini_options]
|
|
testpaths = ["tests"]
|
|
|
|
[[tool.uv.index]]
|
|
name = "pytorch-cu128"
|
|
url = "https://download.pytorch.org/whl/cu128"
|
|
explicit = true
|
|
|
|
[tool.uv.sources]
|
|
torch = { index = "pytorch-cu128" }
|
|
torchvision = { index = "pytorch-cu128" }
|
|
|
|
[tool.uv]
|
|
# Hold the transitive pins to the 2026-09-30 bench image (semif-serve:0.1.4's lock), so the
|
|
# service runs the stack its acceptance numbers are compared against.
|
|
constraint-dependencies = ["numpy==2.2.6", "huggingface-hub==1.31.0", "regex==2026.9.10", "tokenizers==0.23.2", "safetensors==0.8.0"]
|