# intern-decision: Intern-Decision-4B (internlm, Apache-2.0) behind intern-decision-serve, on # fv-ml1 GPU 1 (the utility card, beside vllm-coder, the erp/meromero seats and scriberr). # Replaces semif (Prime, 2026-09-30: "replace semif with intern-decision now"). # # One forward pass per call, scored by the checkpoint's OWN inference.py (sha256-pinned); the # service keeps semif-serve's HTTP surface (/decide, /decide/shared, /health). Service code + # contract: services/intern-decision-serve/ (intern-decision-serve.contract.md). Image built on # fv-ml1 from that dir. # # ⚠ VRAM_CAP_GIB is a HARD cap on torch's allocator (per-process memory fraction), set so the # container's WHOLE nvidia-smi footprint, CUDA context included, fits beside scriberr's peak # (infra-ops budget, 2026-09-30); MAX_TOKENS keeps every accepted call under the cap. A request that needs more # gets 503 out_of_memory and the service stays up. See the README before changing it. # # .env (tunables): IMAGE, PORT, GPU_ID, VRAM_CAP_GIB, MAX_TOKENS, HOST_IP, INTERN_DECISION_API_TOKEN # (vault intern-decision/api-token, >= 32 chars). name: intern-decision services: intern-decision: image: ${IMAGE:?set IMAGE} container_name: intern-decision restart: unless-stopped ports: - "${PORT:-8033}:8000" environment: INTERN_DECISION_API_TOKEN: ${INTERN_DECISION_API_TOKEN:?set INTERN_DECISION_API_TOKEN} INTERN_DECISION_DEVICE: cuda INTERN_DECISION_VRAM_CAP_GIB: ${VRAM_CAP_GIB:?set VRAM_CAP_GIB} # Coupled to VRAM_CAP_GIB: the largest call measured to fit under the cap (README "VRAM"). INTERN_DECISION_MAX_TOKENS: ${MAX_TOKENS:?set MAX_TOKENS} INTERN_DECISION_MAX_DECISIONS: ${MAX_DECISIONS:-64} # POSTs in progress (queued + scoring) before new ones get 429 busy. INTERN_DECISION_MAX_QUEUE: ${MAX_QUEUE:-32} volumes: # Pinned weights AND the checkpoint's inference.py, read offline. Never downloads. - /tank/aimodels/huggingface:/hf:ro deploy: resources: reservations: devices: - driver: nvidia device_ids: ["${GPU_ID:-1}"] capabilities: [gpu] healthcheck: test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"] interval: 30s timeout: 10s retries: 3 # Startup loads ~9 GB of weights and scores the warm-up three times before it serves. start_period: 300s networks: - tnet labels: - homepage.group=AI - Eval & Retrieval - homepage.name=Intern-Decision — typed decisions - homepage.icon=mdi-scale-balance - homepage.description=Typed decisions from one forward pass (Intern-Decision-4B, fv-ml1 GPU1) - homepage.href=http://${HOST_IP:-10.251.50.54}:${PORT:-8033}/health networks: tnet: name: traefik-net external: true