# intern-decision — copy to /opt/docker/compose/intern-decision/.env on fv-ml1 (mode 0600). # Built on fv-ml1 from services/intern-decision-serve (see README "Building"). IMAGE=intern-decision-serve:0.1.0 PORT=8033 HOST_IP=10.251.50.54 # fv-ml1 GPU 1 = the utility card (vllm-coder, erp, meromero, scriberr). GPU_ID=1 # HARD torch-allocator cap: the single knob that holds the container's WHOLE nvidia-smi footprint # (CUDA context included) <= 10,300 MiB, the GPU 1 budget next to scriberr (2026-09-30). # footprint <= cap + non-allocator overhead = 9,472 MiB + 662 MiB = 10,134 MiB # (overhead measured flat at 660-662 MiB with the single inference thread; README "VRAM"). # 9.25 still answers the largest request the API accepts (4 calls x 8,191 tokens) with a 200. VRAM_CAP_GIB=9.25 # >= 32 characters; source of truth: secret get intern-decision/api-token INTERN_DECISION_API_TOKEN=