From fa73c52f9fba99215a76863960533ec08679d772 Mon Sep 17 00:00:00 2001 From: Yash Raj Pandey Date: Tue, 1 Sep 2026 22:57:21 -0400 Subject: [PATCH] fix(proxy): hydrate non-LLM objects when the model reconcile fails _add_deployment_locked ran the model reconcile and the non-LLM object hydration inside one try/except. An exception in prefetch_config_params, _get_models_from_db, _update_llm_router or _update_general_settings therefore skipped _init_non_llm_objects_in_db, so guardrails, policies, vector stores, MCP servers, agents and pass-through endpoints were not hydrated for that pass. The handler only logs, so nothing surfaced. add_deployment also runs on a timer, so a transient DB error dropped every non-LLM object until a later pass happened to succeed. Give the non-LLM hydration its own try/except. It reads its own tables and does not depend on the model reconcile. --- litellm/proxy/proxy_server.py | 13 +++- tests/test_litellm/proxy/test_proxy_server.py | 74 +++++++++++++++++++ 2 files changed, 84 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 0915b8dd1b9..30658f3529b 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -7123,12 +7123,19 @@ class ProxyConfig: db_general_settings=db_general_settings.param_value, ) - # initialize vector stores, guardrails, etc. table in db - await self._init_non_llm_objects_in_db(prisma_client=prisma_client) - except Exception as e: verbose_proxy_logger.exception("litellm.proxy.proxy_server.py::ProxyConfig:add_deployment - %s", e) + # Own try: the model reconcile above must not decide whether guardrails, vector + # stores, MCP servers and agents get hydrated. They read their own tables. + try: + # initialize vector stores, guardrails, etc. table in db + await self._init_non_llm_objects_in_db(prisma_client=prisma_client) + except Exception as e: + verbose_proxy_logger.exception( + "litellm.proxy.proxy_server.py::ProxyConfig:add_deployment non-LLM objects - %s", e + ) + # Read while the lock is still held: once it is released the next reconcile can # begin, and clear_cache's leading wipe would make this look like a mass drop. return ReconcileOutcome( diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py index b7bb58378d4..06f47a7715b 100644 --- a/tests/test_litellm/proxy/test_proxy_server.py +++ b/tests/test_litellm/proxy/test_proxy_server.py @@ -1052,6 +1052,80 @@ async def test_initialize_scheduled_jobs_hydrates_mcp_when_store_model_in_db_fal mock_proxy_config.init_mcp_servers_from_db.assert_awaited_once() +@pytest.mark.asyncio +async def test_add_deployment_hydrates_non_llm_objects_when_model_reconcile_fails(): + """ + Regression: _add_deployment_locked ran the model reconcile and the non-LLM object + hydration inside one try/except. A failure in the reconcile skipped guardrails, + vector stores, MCP servers, agents and pass-through endpoints for that pass, and + the swallowed exception left only a log line behind. add_deployment also runs on a + timer, so a transient DB error dropped every non-LLM object until a later pass. + """ + from litellm.proxy.proxy_server import ProxyConfig + from litellm.proxy.utils import ProxyLogging + + class FailingReconcileProxyConfig(ProxyConfig): + """Reconcile fails the way a DB outage makes it fail; hydration records itself.""" + + def __init__(self): + super().__init__() + self.non_llm_hydrations = 0 + + async def _get_models_from_db(self, prisma_client): + raise RuntimeError("db unavailable") + + async def _init_non_llm_objects_in_db(self, prisma_client): + self.non_llm_hydrations += 1 + + proxy_config = FailingReconcileProxyConfig() + + outcome = await proxy_config._add_deployment_locked( + prisma_client=MagicMock(), + proxy_logging_obj=MagicMock(spec=ProxyLogging), + ) + + assert proxy_config.non_llm_hydrations == 1 + assert outcome.still_desired is None + + +@pytest.mark.asyncio +async def test_add_deployment_survives_a_non_llm_hydration_failure(): + """ + The new hydration boundary must behave like the reconcile one: log the error and + carry on, so a failing guardrail or vector-store table cannot take down the whole + reconcile pass or escape into the scheduler job that calls it. + """ + from litellm.proxy.proxy_server import ProxyConfig + from litellm.proxy.utils import ProxyLogging + + class FailingHydrationProxyConfig(ProxyConfig): + """Reconcile succeeds, hydration raises the way an unavailable table does.""" + + def __init__(self): + super().__init__() + self.reconcile_ran = False + + async def _get_models_from_db(self, prisma_client): + return [] + + async def _update_llm_router(self, new_models, proxy_logging_obj): + self.reconcile_ran = True + return frozenset() + + async def _init_non_llm_objects_in_db(self, prisma_client): + raise RuntimeError("guardrail table unavailable") + + proxy_config = FailingHydrationProxyConfig() + + outcome = await proxy_config._add_deployment_locked( + prisma_client=MagicMock(), + proxy_logging_obj=MagicMock(spec=ProxyLogging), + ) + + assert proxy_config.reconcile_ran is True + assert outcome.still_desired == frozenset() + + @pytest.mark.asyncio async def test_init_mcp_servers_from_db_respects_supported_db_objects(monkeypatch): """