diff --git a/litellm/router_strategy/lowest_tpm_rpm_v2.py b/litellm/router_strategy/lowest_tpm_rpm_v2.py index 911d6bece65..44d04507540 100644 --- a/litellm/router_strategy/lowest_tpm_rpm_v2.py +++ b/litellm/router_strategy/lowest_tpm_rpm_v2.py @@ -27,6 +27,7 @@ else: class RoutingArgs(LiteLLMPydanticObjectBase): ttl: int = 1 * 60 # 1min (RPM/TPM expire key) + allow_routing_on_cache_read_failure: bool = False class LowestTPMLoggingHandler_v2(BaseRoutingStrategy, CustomLogger): @@ -383,6 +384,8 @@ class LowestTPMLoggingHandler_v2(BaseRoutingStrategy, CustomLogger): """ if tpm_values is None or rpm_values is None: + if not self.routing_args.allow_routing_on_cache_read_failure: + return None verbose_router_logger.warning( "usage-based-routing-v2: tpm/rpm cache read failed for model_group=%s - " "falling back to routing without usage data for this request", diff --git a/tests/test_litellm/router_strategy/test_lowest_tpm_rpm_v2.py b/tests/test_litellm/router_strategy/test_lowest_tpm_rpm_v2.py index ee5683091c9..5cf6c6ce3a6 100644 --- a/tests/test_litellm/router_strategy/test_lowest_tpm_rpm_v2.py +++ b/tests/test_litellm/router_strategy/test_lowest_tpm_rpm_v2.py @@ -2,8 +2,8 @@ Regression tests for https://github.com/BerriAI/litellm/issues/16060 When the tpm/rpm usage cache read fails (DualCache.[async_]batch_get_cache -returns None, e.g. on a transient Redis error), usage-based-routing-v2 must -fail open and still return a healthy deployment, instead of raising +returns None, e.g. on a transient Redis error), usage-based-routing-v2 can +explicitly opt into returning a healthy deployment instead of raising "No deployments available" (RateLimitError / 429). """ @@ -34,14 +34,35 @@ HEALTHY_DEPLOYMENTS = [ ] -def _handler() -> LowestTPMLoggingHandler_v2: - return LowestTPMLoggingHandler_v2(router_cache=DualCache()) +def _handler(*, allow_routing_on_cache_read_failure: bool = False) -> LowestTPMLoggingHandler_v2: + return LowestTPMLoggingHandler_v2( + router_cache=DualCache(), + routing_args={"allow_routing_on_cache_read_failure": allow_routing_on_cache_read_failure}, + ) + + +@pytest.mark.asyncio +async def test_async_cache_read_failure_fails_closed_by_default(): + import litellm + + handler = _handler() + with patch.object( + handler.router_cache, + "async_batch_get_cache", + new=AsyncMock(return_value=None), + ): + with pytest.raises(litellm.RateLimitError): + await handler.async_get_available_deployments( + model_group=MODEL_GROUP, + healthy_deployments=HEALTHY_DEPLOYMENTS, + messages=[{"role": "user", "content": "hey"}], + ) @pytest.mark.asyncio async def test_async_cache_read_failure_fails_open(): """Batch cache read returning None must not fail the request.""" - handler = _handler() + handler = _handler(allow_routing_on_cache_read_failure=True) with patch.object( handler.router_cache, "async_batch_get_cache", @@ -58,7 +79,7 @@ async def test_async_cache_read_failure_fails_open(): def test_sync_cache_read_failure_fails_open(): """Sync path: batch cache read returning None must not fail the request.""" - handler = _handler() + handler = _handler(allow_routing_on_cache_read_failure=True) with patch.object(handler.router_cache, "batch_get_cache", return_value=None): deployment = handler.get_available_deployments( model_group=MODEL_GROUP,