[project] name = "intern-decision-serve" version = "0.1.0" description = "Intern-Decision-4B (its own inference.py) behind semif-serve's HTTP surface" requires-python = ">=3.12" dependencies = [ "fastapi==0.118.0", "uvicorn==0.37.0", ] [project.optional-dependencies] # The real engine: the 2026-09-30 bench stack (semif-serve 0.1.4's torch/transformers plus what # the checkpoint's own inference.py imports: PIL and the Qwen3.5 processor, which needs torchvision). model = [ "torch==2.10.0", "torchvision==0.25.0", "transformers==5.17.0", "pillow==12.3.0", ] # Qwen3.5's fast kernels, as in the bench image (without them transformers runs its slower # reference PyTorch paths, and the bench numbers were measured with them). fast = [ "flash-linear-attention==0.5.2", "causal-conv1d @ https://github.com/Dao-AILab/causal-conv1d/releases/download/v1.7.0/causal_conv1d-1.7.0+cu12torch2.10cxx11abiTRUE-cp312-cp312-linux_x86_64.whl ; sys_platform == 'linux' and platform_machine == 'x86_64'", ] [dependency-groups] dev = ["pytest==8.4.2", "httpx==0.28.1"] [build-system] requires = ["setuptools>=68"] build-backend = "setuptools.build_meta" [tool.setuptools.packages.find] where = ["src"] [tool.pytest.ini_options] testpaths = ["tests"] [[tool.uv.index]] name = "pytorch-cu128" url = "https://download.pytorch.org/whl/cu128" explicit = true [tool.uv.sources] torch = { index = "pytorch-cu128" } torchvision = { index = "pytorch-cu128" } [tool.uv] # Hold the transitive pins to the 2026-09-30 bench image (semif-serve:0.1.4's lock), so the # service runs the stack its acceptance numbers are compared against. constraint-dependencies = ["numpy==2.2.6", "huggingface-hub==1.31.0", "regex==2026.9.10", "tokenizers==0.23.2", "safetensors==0.8.0"]