# Infinity stack tunables. Copy this to `.env` on the server before deploying. # # cp .env.example .env # # edit .env with real values # docker compose up -d # Image version — pin for reproducibility (`latest` for edge) INFINITY_VERSION=latest # Port exposed on host INFINITY_PORT=7997 # GPU assignment (ana-ml2 has 0 and 1; default 1 keeps 0 free for heavy LLM work) GPU_ID=1 # Models — both served simultaneously; reference by the full repo name in requests EMBED_MODEL=Qwen/Qwen3-Embedding-0.6B RERANK_MODEL=Qwen/Qwen3-Reranker-0.6B # Inference engine: torch (widest support) or optimum (ONNX, sometimes faster) ENGINE=torch # Batch size — 32 is a safe default; bump for throughput if VRAM allows BATCH_SIZE=32 # Optional API key — leave blank for no auth (fine on the internal network) API_KEY= # HuggingFace token — only needed for gated models HF_TOKEN=