# LiteLLM proxy config: free-tier models behind stable aliases, with a
# fallback chain. Consumers ask for the aliases (analyst, hermes,
# analyst-lite), never for a provider model, so a provider can be swapped
# without touching any agent. Keys come from the container environment
# (.env via env_file in docker-compose.yaml).

model_list:
  # --- main analytical model (the scripts) ---
  - model_name: analyst
    litellm_params:
      model: gemini/gemini-3.5-flash
      api_key: os.environ/GEMINI_API_KEY
      # Free tier for this model showed 10 RPM / 1,500 RPD in AI Studio at
      # the time of writing; stay under it so LiteLLM stops before Google does.
      rpm: 8
      max_retries: 2

  # --- Hermes Agent: same model and key, its own lane. Free-tier RPM is per
  # project, so 8 + 6 is over the 10; the two rarely run at once and Flash-Lite
  # catches the overflow. No Groq in this chain: the agent loop is streaming +
  # tool calls, on which Llama answers finish_reason=tool_use_failed and
  # LiteLLM 1.83 turns that into a 500 (int("tool_use_failed")).
  - model_name: hermes
    litellm_params:
      model: gemini/gemini-3.5-flash
      api_key: os.environ/GEMINI_API_KEY
      rpm: 6
      max_retries: 2

  # --- fallback 1: Groq, fast, generous RPM, small token allowance ---
  - model_name: analyst-groq
    litellm_params:
      model: groq/llama-3.3-70b-versatile
      api_key: os.environ/GROQ_API_KEY
      rpm: 25

  # --- fallback 2 / cheap routine work: Flash-Lite, its own quota ---
  - model_name: analyst-lite
    litellm_params:
      model: gemini/gemini-3.1-flash-lite
      api_key: os.environ/GEMINI_API_KEY
      rpm: 12

litellm_settings:
  # When Flash hits its quota (429) go to Groq, then Flash-Lite.
  fallbacks:
    - analyst: ["analyst-groq", "analyst-lite"]
    - analyst-groq: ["analyst-lite"]
    - hermes: ["analyst-lite"]
  num_retries: 2
  request_timeout: 120
  # Response cache: the same prompt again (a re-run while debugging) costs no
  # quota. In-memory is enough here; Redis would be for a second replica.
  cache: true
  cache_params:
    type: local
    ttl: 3600
  # Keep provider keys out of logs and error messages.
  redact_user_api_key_info: true
  drop_params: true

general_settings:
  master_key: os.environ/LITELLM_MASTER_KEY
  # Manage model_list from the admin UI as well (models then live in the DB,
  # not only in this file).
  store_model_in_db: true
  # Without this the UI's spend log entries have no request/response body.
  store_prompts_in_spend_logs: true
  # Provider outages / budget alerts can go to a webhook (Telegram, Slack):
  # alerting: ["webhook"]
  # alerting_args:
  #   webhook_url: os.environ/ALERT_WEBHOOK_URL

router_settings:
  # On free quotas a pause and a retry beat speed.
  retry_after: 5
  allowed_fails: 3
  cooldown_time: 60
