diff --git a/stacks/litellm/.env.example b/stacks/litellm/.env.example index 0c2fde8..b26f72e 100644 --- a/stacks/litellm/.env.example +++ b/stacks/litellm/.env.example @@ -3,7 +3,7 @@ # Image tag. main-stable is the rolling stable; pin to a dated/SHA tag # (e.g. main-v1.74.0-stable) once a known-good build is confirmed. -LITELLM_TAG=main-stable +LITELLM_TAG=v1.97.0 # Publish. Bind to all interfaces on the LAN; 4000 is the LiteLLM default # (proxy API + admin/Logs UI at /ui). diff --git a/stacks/litellm/conf/config.yaml b/stacks/litellm/conf/config.yaml index 3ef7fe5..0b5f8b7 100644 --- a/stacks/litellm/conf/config.yaml +++ b/stacks/litellm/conf/config.yaml @@ -537,7 +537,20 @@ general_settings: # THE log switch: persists full request messages + response bodies into # SpendLogs so they render in the Logs UI. Without this you get metadata # (tokens, latency, model) but not the prompt/completion text. - store_prompts_in_spend_logs: true + # + # ⚠️ TURNED OFF 2026-08-16 (operator: "I don't need any of that information"). + # With this TRUE the SpendLogs table stored every prompt+completion body and + # grew to 6.0 GB (of a 6.08 GB DB). Off = lightweight cost/usage rows only + # (tokens, latency, model, cost) — the cross-project spend tracking survives, + # the bulky bodies do not. Re-enable ONLY for a bounded debugging window, not + # standing. + store_prompts_in_spend_logs: false + # HARD CAP on SpendLogs growth (operator: "if there's a way to cap it, CAP + # it"). The retention job deletes rows older than the period on the interval + # cadence, so the table is bounded by ~7 days of lightweight rows rather than + # unbounded. Names verified against LiteLLM docs (proxy/spend_logs_deletion). + maximum_spend_logs_retention_period: "7d" + maximum_spend_logs_retention_interval: "1d" # scalar-judge → Skywork-Reward-V2 (scalar reward model; vLLM pooling on # ana-ml2:8003). LiteLLM has no reward/pooling MODE, so this is a passthrough, # not a model_list alias. Gateway-key-gated. Consumers POST the reward body to