feat(proxy): make DB config-reload interval configurable via config.yaml and UI (#34130)

The add_deployment and get_credentials background jobs that keep a multi-pod
deployment in sync with config-in-DB objects (models, credentials, guardrails,
general settings, etc.) polled the database on a hardcoded 30s interval, with
no way to trade convergence latency against DB load.

Expose it as the general_setting proxy_config_reload_interval_seconds (env
PROXY_CONFIG_RELOAD_INTERVAL_SECONDS parsed via get_env_int, default 30),
threaded like the existing proxy_batch_polling_interval knob, and surface it on
the admin general-settings page so it is reachable from the dashboard and
persists to the DB for all pods. Non-positive values are rejected at the UI
(gt=0) and fall back to 30s with a warning on the env/config/DB paths.
This commit is contained in:
Yassin Kortam 2026-07-21 15:03:22 -07:00 • committed by GitHub
parent 090c4b8dd8
commit c2ae52709c
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 225 additions and 2 deletions

View file

@ -1469,6 +1469,7 @@ _batch_polling_env = os.getenv("PROXY_BATCH_POLLING_ENABLED", "true").lower()
PROXY_BATCH_POLLING_ENABLED = _batch_polling_env == "true"
PROXY_BUDGET_RESCHEDULER_MAX_TIME = int(os.getenv("PROXY_BUDGET_RESCHEDULER_MAX_TIME", 605))
PROXY_BATCH_WRITE_AT = int(os.getenv("PROXY_BATCH_WRITE_AT", 10)) # in seconds, increased from 10
PROXY_CONFIG_RELOAD_INTERVAL_SECONDS = get_env_int("PROXY_CONFIG_RELOAD_INTERVAL_SECONDS", 30)
# APScheduler Configuration - MEMORY LEAK FIX
# These settings prevent memory leaks in APScheduler's normalize() and _apply_jitter() functions

View file

@ -2297,6 +2297,11 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
None,
description="max response size in MB, if a response is larger than this size it will be rejected",
)
proxy_config_reload_interval_seconds: int = Field(
30,
gt=0,
description="how often (in seconds) each pod reloads config-in-DB objects (models, credentials, guardrails, etc.) when store_model_in_db is enabled; lower values speed up multi-pod convergence at the cost of more DB load. Applied on proxy startup",
)
cancel_on_disconnect: Optional[bool] = Field(
None,
description="cancel the in-flight upstream LLM request (non-streaming) when the client disconnects, freeing backend capacity (e.g. a vLLM GPU slot); the request is logged as a 499 failure",

View file

@ -236,6 +236,7 @@ from litellm.constants import (
PROXY_BATCH_WRITE_AT,
PROXY_BUDGET_RESCHEDULER_MAX_TIME,
PROXY_BUDGET_RESCHEDULER_MIN_TIME,
PROXY_CONFIG_RELOAD_INTERVAL_SECONDS,
)
from litellm.exceptions import RejectedRequestError
from litellm.integrations.custom_guardrail import ModifyResponseException
@ -2001,6 +2002,7 @@ proxy_budget_rescheduler_min_time = PROXY_BUDGET_RESCHEDULER_MIN_TIME
proxy_budget_rescheduler_max_time = PROXY_BUDGET_RESCHEDULER_MAX_TIME
proxy_batch_polling_interval = PROXY_BATCH_POLLING_INTERVAL
proxy_batch_write_at = PROXY_BATCH_WRITE_AT
proxy_config_reload_interval_seconds = PROXY_CONFIG_RELOAD_INTERVAL_SECONDS
litellm_master_key_hash = None
disable_spend_logs = False
jwt_handler = JWTHandler()
@ -4337,6 +4339,7 @@ class ProxyConfig:
open_telemetry_logger, \
health_check_details, \
proxy_batch_polling_interval, \
proxy_config_reload_interval_seconds, \
config_passthrough_endpoints
config: dict = await self.get_config(config_file_path=config_file_path)
@ -4826,6 +4829,10 @@ class ProxyConfig:
)
## BATCH WRITER ##
proxy_batch_write_at = general_settings.get("proxy_batch_write_at", proxy_batch_write_at)
## DB CONFIG RELOAD INTERVAL ##
proxy_config_reload_interval_seconds = general_settings.get(
"proxy_config_reload_interval_seconds", proxy_config_reload_interval_seconds
)
## DISABLE SPEND LOGS ## - gives a perf improvement
disable_spend_logs = general_settings.get("disable_spend_logs", disable_spend_logs)
### BACKGROUND HEALTH CHECKS ###
@ -7998,12 +8005,20 @@ class ProxyStartupEvent:
verbose_proxy_logger.debug("Failed to check DB for store_model_in_db: %s", str(e))
if store_model_in_db is True:
config_reload_interval_seconds = proxy_config_reload_interval_seconds
if not isinstance(config_reload_interval_seconds, int) or config_reload_interval_seconds <= 0:
verbose_proxy_logger.warning(
"proxy_config_reload_interval_seconds=%s must be a positive integer; falling back to 30s",
config_reload_interval_seconds,
)
config_reload_interval_seconds = 30
# MEMORY LEAK FIX: Increase interval from 10s to 30s minimum
# Frequent polling was causing excessive memory allocations
scheduler.add_job(
proxy_config.add_deployment,
"interval",
seconds=30, # increased from 10s to reduce memory pressure
seconds=config_reload_interval_seconds,
# REMOVED jitter parameter - major cause of memory leak
args=[prisma_client, proxy_logging_obj],
id="add_deployment_job",
@ -8018,7 +8033,7 @@ class ProxyStartupEvent:
scheduler.add_job(
proxy_config.get_credentials,
"interval",
seconds=30, # increased from 10s to reduce memory pressure
seconds=config_reload_interval_seconds,
# REMOVED jitter parameter - major cause of memory leak
args=[prisma_client],
id="get_credentials_job",
@ -15052,6 +15067,7 @@ async def get_config_list(
"global_max_parallel_requests": {"type": "Integer"},
"max_request_size_mb": {"type": "Integer"},
"max_response_size_mb": {"type": "Integer"},
"proxy_config_reload_interval_seconds": {"type": "Integer"},
"pass_through_endpoints": {"type": "PydanticModel"},
"store_model_in_db": {"type": "Boolean"},
"store_prompts_in_spend_logs": {"type": "Boolean"},

View file

@ -1069,6 +1069,32 @@ async def test_ProxyConfig_load_config_wires_general_settings_url_validation(tmp
litellm.provider_url_destination_allowed_hosts = original_provider_hosts
@pytest.mark.asyncio
async def test_ProxyConfig_load_config_wires_config_reload_interval(tmp_path, monkeypatch):
"""general_settings.proxy_config_reload_interval_seconds must reach the proxy_server
module global that schedules the DB config-reload jobs, so operators can tune multi-pod
convergence from config.yaml."""
import litellm.proxy.proxy_server as proxy_server
f = tmp_path / "c.yaml"
f.write_text(
"model_list: []\n"
"general_settings:\n"
" proxy_config_reload_interval_seconds: 47\n"
"litellm_settings: {}\n"
)
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)
monkeypatch.setattr("litellm.proxy.proxy_server.store_model_in_db", False)
monkeypatch.delenv("LITELLM_CONFIG_BUCKET_NAME", raising=False)
original = proxy_server.proxy_config_reload_interval_seconds
try:
await ProxyConfig().load_config(router=None, config_file_path=str(f))
assert proxy_server.proxy_config_reload_interval_seconds == 47
finally:
proxy_server.proxy_config_reload_interval_seconds = original
@pytest.mark.asyncio
async def test_ProxyConfig_load_config_missing_file_raises(monkeypatch):
monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None)

View file

@ -13,6 +13,7 @@ Routes covered:
from __future__ import annotations
import json
from unittest.mock import AsyncMock, MagicMock
from .conftest import VOLATILE_KEYS, normalize
@ -473,6 +474,83 @@ def test_config_list_happy_admin(client, auth_as, mock_prisma, monkeypatch):
}
def test_config_list_exposes_config_reload_interval(client, auth_as, mock_prisma, monkeypatch):
"""proxy_config_reload_interval_seconds must surface in the admin UI general-settings
list as an Integer field defaulting to 30, so operators can tune multi-pod convergence
from the dashboard."""
from litellm.proxy import proxy_server as ps
from litellm.proxy._types import LitellmUserRoles
table = _install_litellm_config(mock_prisma)
row = MagicMock()
row.param_value = {}
table.find_first = AsyncMock(return_value=row)
monkeypatch.setattr(ps, "prisma_client", mock_prisma)
with auth_as(LitellmUserRoles.PROXY_ADMIN):
response = client.get("/config/list", params={"config_type": "general_settings"})
assert response.status_code == 200
by_name = {entry["field_name"]: entry for entry in response.json()}
assert "proxy_config_reload_interval_seconds" in by_name
entry = by_name["proxy_config_reload_interval_seconds"]
assert entry["field_type"] == "Integer"
assert entry["field_default_value"] == 30
def test_config_field_update_accepts_config_reload_interval(client, auth_as, mock_prisma, monkeypatch):
"""POST /config/field/update accepts proxy_config_reload_interval_seconds and persists
it to the DB general_settings row for all pods to pick up."""
from litellm.proxy import proxy_server as ps
from litellm.proxy._types import LitellmUserRoles
table = _install_litellm_config(mock_prisma)
table.find_first = AsyncMock(return_value=None)
upsert_row = {
"param_name": "general_settings",
"param_value": {"proxy_config_reload_interval_seconds": 45},
"id": "row-1",
}
table.upsert = AsyncMock(return_value=upsert_row)
monkeypatch.setattr(ps, "prisma_client", mock_prisma)
with auth_as(LitellmUserRoles.PROXY_ADMIN):
response = client.post(
"/config/field/update",
json={
"field_name": "proxy_config_reload_interval_seconds",
"field_value": 45,
"config_type": "general_settings",
},
)
assert response.status_code == 200
upserted = table.upsert.call_args.kwargs["data"]["create"]["param_value"]
assert json.loads(upserted)["proxy_config_reload_interval_seconds"] == 45
def test_config_field_update_rejects_non_positive_config_reload_interval(client, auth_as, mock_prisma, monkeypatch):
"""A non-positive proxy_config_reload_interval_seconds from the UI is rejected with a 400
and never persisted, since APScheduler requires a positive interval."""
from litellm.proxy import proxy_server as ps
from litellm.proxy._types import LitellmUserRoles
table = _install_litellm_config(mock_prisma)
table.find_first = AsyncMock(return_value=None)
table.upsert = AsyncMock()
monkeypatch.setattr(ps, "prisma_client", mock_prisma)
with auth_as(LitellmUserRoles.PROXY_ADMIN):
response = client.post(
"/config/field/update",
json={
"field_name": "proxy_config_reload_interval_seconds",
"field_value": 0,
"config_type": "general_settings",
},
)
assert response.status_code == 400
table.upsert.assert_not_called()
def test_config_list_non_admin_rejected(client, auth_as, mock_prisma, monkeypatch):
"""Non-admin gets a 400 with the role embedded in the error message."""
from litellm.proxy import proxy_server as ps

View file

@ -751,6 +751,97 @@ async def test_initialize_scheduled_jobs_credentials(monkeypatch):
assert len(mock_scheduler_calls) > 0
@pytest.mark.asyncio
async def test_initialize_scheduled_jobs_uses_configured_config_reload_interval(monkeypatch):
"""
The DB config-reload jobs (add_deployment, get_credentials) that keep multi-pod
deployments in sync must be scheduled at the configured
proxy_config_reload_interval_seconds, not a hardcoded value.
"""
monkeypatch.delenv("DISABLE_PRISMA_SCHEMA_UPDATE", raising=False)
monkeypatch.delenv("STORE_MODEL_IN_DB", raising=False)
from litellm.proxy.proxy_server import ProxyStartupEvent
from litellm.proxy.utils import ProxyLogging
mock_prisma_client = MagicMock()
mock_proxy_logging = MagicMock(spec=ProxyLogging)
mock_proxy_logging.slack_alerting_instance = MagicMock()
mock_proxy_config = AsyncMock()
mock_scheduler = MagicMock()
configured_interval = 47
with (
patch("litellm.proxy.proxy_server.proxy_config", mock_proxy_config),
patch("litellm.proxy.proxy_server.store_model_in_db", True),
patch("litellm.proxy.proxy_server.get_secret_bool", return_value=True),
patch(
"litellm.proxy.proxy_server.proxy_config_reload_interval_seconds",
configured_interval,
),
patch("litellm.proxy.proxy_server.AsyncIOScheduler", return_value=mock_scheduler),
):
await ProxyStartupEvent.initialize_scheduled_background_jobs(
general_settings={},
prisma_client=mock_prisma_client,
proxy_budget_rescheduler_min_time=1,
proxy_budget_rescheduler_max_time=2,
proxy_batch_write_at=5,
proxy_logging_obj=mock_proxy_logging,
)
scheduled_seconds = {
job_call.kwargs["id"]: job_call.kwargs.get("seconds")
for job_call in mock_scheduler.add_job.call_args_list
if "id" in job_call.kwargs
}
assert scheduled_seconds["add_deployment_job"] == configured_interval
assert scheduled_seconds["get_credentials_job"] == configured_interval
@pytest.mark.asyncio
async def test_initialize_scheduled_jobs_rejects_non_positive_config_reload_interval(monkeypatch):
"""
A non-positive proxy_config_reload_interval_seconds (misconfig via env/config/DB) would
make APScheduler reject the job and crash startup, so the scheduler must fall back to the
30s default instead of forwarding the bad value.
"""
monkeypatch.delenv("DISABLE_PRISMA_SCHEMA_UPDATE", raising=False)
monkeypatch.delenv("STORE_MODEL_IN_DB", raising=False)
from litellm.proxy.proxy_server import ProxyStartupEvent
from litellm.proxy.utils import ProxyLogging
mock_prisma_client = MagicMock()
mock_proxy_logging = MagicMock(spec=ProxyLogging)
mock_proxy_logging.slack_alerting_instance = MagicMock()
mock_proxy_config = AsyncMock()
mock_scheduler = MagicMock()
with (
patch("litellm.proxy.proxy_server.proxy_config", mock_proxy_config),
patch("litellm.proxy.proxy_server.store_model_in_db", True),
patch("litellm.proxy.proxy_server.get_secret_bool", return_value=True),
patch("litellm.proxy.proxy_server.proxy_config_reload_interval_seconds", 0),
patch("litellm.proxy.proxy_server.AsyncIOScheduler", return_value=mock_scheduler),
):
await ProxyStartupEvent.initialize_scheduled_background_jobs(
general_settings={},
prisma_client=mock_prisma_client,
proxy_budget_rescheduler_min_time=1,
proxy_budget_rescheduler_max_time=2,
proxy_batch_write_at=5,
proxy_logging_obj=mock_proxy_logging,
)
scheduled_seconds = {
job_call.kwargs["id"]: job_call.kwargs.get("seconds")
for job_call in mock_scheduler.add_job.call_args_list
if "id" in job_call.kwargs
}
assert scheduled_seconds["add_deployment_job"] == 30
assert scheduled_seconds["get_credentials_job"] == 30
@pytest.mark.asyncio
async def test_initialize_scheduled_jobs_hydrates_mcp_when_store_model_in_db_false(monkeypatch):
"""

View file

@ -22629,6 +22629,12 @@ export interface components {
* @description Allowlist of hosts a request may redirect a provider call's destination URL to.
*/
provider_url_destination_allowed_hosts?: string[] | null;
/**
* Proxy Config Reload Interval Seconds
* @description how often (in seconds) each pod reloads config-in-DB objects (models, credentials, guardrails, etc.) when store_model_in_db is enabled; lower values speed up multi-pod convergence at the cost of more DB load. Applied on proxy startup
* @default 30
*/
proxy_config_reload_interval_seconds: number;
/**
* Reject Clientside Metadata Tags
* @description When set to True, rejects requests that contain client-side 'metadata.tags' to prevent users from influencing budgets by sending different tags. Tags can only be inherited from the API key metadata.