# semif: SemIf option-logit decisions (github TheoLeeCJ/SemIf-OpenJev, MIT) behind semif-serve, # on fv-ml1 GPU 1 (the utility card, beside vllm-coder and scriberr). Prime, 2026-09-27. # # One forward pass of a pinned Qwen3.5-4B (BF16) per decision; the answer is read from the # option-letter logits, so there is no decoding. Service code + contract: # services/semif-serve/ (semif-serve.contract.md). Image built on fv-ml1 from that dir. # # ⚠ Scores are "conditional option score; uncalibrated as decision confidence". A caller # that needs thresholds brings labelled rows; we fit a per-workload temperature into # conf/calibration.json and the caller passes `workload`. See the README. # ⚠ SEMIF_VRAM_CAP_GIB is a HARD cap (torch per-process memory fraction), sized from a # measured peak, so SemIf cannot squeeze scriberr or the vLLM seats on this card. A # request that needs more gets 503 out_of_memory and the service stays up. # # .env (tunables): IMAGE, PORT, GPU_ID, VRAM_CAP_GIB, HOST_IP, SEMIF_API_TOKEN (vault # semif/api-token, >= 32 chars). name: semif services: semif: image: ${IMAGE:?set IMAGE} container_name: semif restart: unless-stopped ports: - "${PORT:-8032}:8000" environment: SEMIF_API_TOKEN: ${SEMIF_API_TOKEN:?set SEMIF_API_TOKEN} SEMIF_DEVICE: cuda SEMIF_VRAM_CAP_GIB: ${VRAM_CAP_GIB:?set VRAM_CAP_GIB} SEMIF_MAX_TOKENS: ${MAX_TOKENS:-4096} SEMIF_MAX_DECISIONS: ${MAX_DECISIONS:-64} # POSTs in progress (queued + scoring) before new ones get 429 busy. SEMIF_MAX_QUEUE: ${MAX_QUEUE:-32} SEMIF_CALIBRATION: /conf/calibration.json volumes: # Pinned weights, read offline (HF_HUB_OFFLINE=1 in the image). Never downloads. - /tank/aimodels/huggingface:/hf:ro - /opt/docker/conf/semif:/conf:ro deploy: resources: reservations: devices: - driver: nvidia device_ids: ["${GPU_ID:-1}"] capabilities: [gpu] healthcheck: test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health', timeout=5)"] interval: 30s timeout: 10s retries: 3 # Startup loads ~9 GB of weights and scores one warm-up decision before it serves. start_period: 300s networks: - tnet labels: - homepage.group=AI - Eval & Retrieval - homepage.name=SemIf — option-logit decisions - homepage.icon=mdi-scale-balance - homepage.description=Typed decisions from one forward pass (Qwen3.5-4B, fv-ml1 GPU1) - homepage.href=http://${HOST_IP:-10.251.50.54}:${PORT:-8032}/health networks: tnet: name: traefik-net external: true