torch keeps CUDA state per host thread (cuBLAS handles and workspaces), partly outside the per-process VRAM cap. Scoring on anyio's threadpool let 40 threads each create it: measured on fv-ml1 GPU 3, +252 MiB outside the cap and +326 MiB inside, which pushed the process past the 10,300 MiB GPU 1 budget. Load, warm-up and every call now run on the same single thread.
24 lines
932 B
Python
24 lines
932 B
Python
"""uvicorn entry point: `uvicorn intern_decision_serve.main:app_from_env --factory --workers 1`."""
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
|
|
from fastapi import FastAPI
|
|
|
|
from .app import create_app, inference_thread
|
|
from .config import Settings
|
|
|
|
|
|
def app_from_env() -> FastAPI:
|
|
settings = Settings.from_env(os.environ)
|
|
# INV-5: never download at runtime, inside the image or out of it. Set before torch /
|
|
# transformers / huggingface_hub are imported, since they read it at import time.
|
|
os.environ["HF_HUB_OFFLINE"] = "1"
|
|
os.environ["TRANSFORMERS_OFFLINE"] = "1"
|
|
from .engine import TorchEngine # torch loads only inside load(), never in the unit tests
|
|
|
|
# INV-2: load, warm-up and every later call run on this one thread (per-thread CUDA state).
|
|
executor = inference_thread()
|
|
engine = executor.submit(TorchEngine.load, settings).result()
|
|
return create_app(settings, engine, executor)
|