"""uvicorn entry point: `uvicorn intern_decision_serve.main:app_from_env --factory --workers 1`.""" from __future__ import annotations import os from fastapi import FastAPI from .app import create_app, inference_thread from .config import Settings def app_from_env() -> FastAPI: settings = Settings.from_env(os.environ) # INV-5: never download at runtime, inside the image or out of it. Set before torch / # transformers / huggingface_hub are imported, since they read it at import time. os.environ["HF_HUB_OFFLINE"] = "1" os.environ["TRANSFORMERS_OFFLINE"] = "1" from .engine import TorchEngine # torch loads only inside load(), never in the unit tests # INV-2: load, warm-up and every later call run on this one thread (per-thread CUDA state). executor = inference_thread() engine = executor.submit(TorchEngine.load, settings).result() return create_app(settings, engine, executor)