# syntax=docker/dockerfile:1 # semif-serve: SemIf (pinned commit) behind a small FastAPI service. Contract: semif-serve.contract.md. # docker build -t semif-serve: . # Weights are NOT in the image: the pinned Qwen3.5-4B revision is read from the mounted # HF cache, offline (INV-5). # The dependency manifest with semif-serve's own version blanked to 0.0.0. A version bump # then leaves these two files byte-identical, and COPY --from compares CONTENT, so the ~4 GB # torch/CUDA install below stays cached across releases (the same problem augaman hit). # `uv sync` keeps uv's per-package index routing (torch from the cu128 index, everything # else from PyPI). An exported requirements.txt loses that, and then fetches triton from the # wrong index and fails its hash check. FROM python:3.12-slim-bookworm AS deps WORKDIR /deps COPY pyproject.toml uv.lock ./ RUN python - <<'EOF' import re, pathlib p = pathlib.Path("pyproject.toml") p.write_text(re.sub(r'(?m)^version = "[^"]+"', 'version = "0.0.0"', p.read_text(), count=1)) l = pathlib.Path("uv.lock") l.write_text(re.sub(r'(name = "semif-serve"\nversion = )"[^"]+"', r'\1"0.0.0"', l.read_text(), count=1)) EOF FROM python:3.12-slim-bookworm COPY --from=ghcr.io/astral-sh/uv:0.6.9 /uv /bin/uv ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy UV_PYTHON_DOWNLOADS=never # git: semif-phase1 installs from a pinned GitHub commit. RUN apt-get update && apt-get install -y --no-install-recommends git ca-certificates \ && rm -rf /var/lib/apt/lists/* WORKDIR /app COPY --from=deps /deps/pyproject.toml /deps/uv.lock ./ RUN --mount=type=cache,target=/root/.cache/uv \ uv sync --frozen --no-dev --extra model --no-install-project COPY pyproject.toml uv.lock ./ COPY src ./src RUN uv sync --frozen --no-dev --extra model --no-editable --no-cache RUN groupadd --system --gid 10001 semif \ && useradd --system --uid 10001 --gid 10001 --no-create-home --shell /usr/sbin/nologin semif USER semif ENV PATH=/app/.venv/bin:$PATH \ HF_HOME=/hf \ HF_HUB_OFFLINE=1 \ HF_HUB_DISABLE_TELEMETRY=1 \ NVIDIA_DRIVER_CAPABILITIES=compute,utility EXPOSE 8000 # One worker (INV-2): the model and the inference lock live in this one process. CMD ["uvicorn", "semif_serve.main:app_from_env", "--factory", "--host", "0.0.0.0", "--port", "8000", "--workers", "1"]