From 455e68bf4aea00d1bc051200ede85f6c7ebf3a35 Mon Sep 17 00:00:00 2001 From: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Date: Fri, 11 Sep 2026 11:16:54 +0000 Subject: [PATCH] fix(router): carry 429 on no-deployment errors, skip them in cooldown When every deployment of a model group is unavailable the router raises RouterRateLimitError or RouterRateLimitErrorBasic, the provider budget limiter raises a bare ValueError, and the sync usage-based-routing-v2 strategy raises a bare ValueError. None of them carry a status_code. A developer calling litellm.Router directly gets a plain ValueError with nothing to branch on, while the async strategy path already raises litellm.RateLimitError for the same condition. The proxy is not affected on the wire: ProxyException rewrites the code to 429 whenever the message says "No deployments available", and the async strategy path raises a typed 429 All four raises now share RouterNoDeploymentsAvailableError, a ValueError subclass that carries status_code 429 and the cooldown time, so Router users get one typed error for "nothing can serve this call" and code that maps exceptions by status_code no longer needs to sniff the message Router.deployment_callback_on_failure returns early for these errors. The callback counts a failure and runs the cooldown logic against whatever model_info sits in the kwargs it receives, and a routing error is never that deployment's own failure. Today the proxy does not reach this branch with a deployment id attached: the retry resets the logging kwargs before re-selection, and the non-streaming failure log is deduped. The early return keeps a future caller, or a 429 now present on the exception, from turning an exhausted pool into an extra failure or a refreshed cooldown on the last deployment that was tried Co-authored-by: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Co-authored-by: songkuan-zheng --- litellm/router.py | 4 ++ litellm/router_strategy/budget_limiter.py | 9 +++- litellm/router_strategy/lowest_tpm_rpm_v2.py | 4 +- litellm/types/router.py | 9 +++- .../test_budget_limiter_hotpath.py | 20 ++++++++ .../router_strategy/test_lowest_tpm_rpm_v2.py | 27 +++++++++++ tests/unit/test_router/test_router.py | 48 +++++++++++++++++++ 7 files changed, 115 insertions(+), 6 deletions(-) create mode 100644 tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py diff --git a/litellm/router.py b/litellm/router.py index 115faad000c..3a2dc2eeb97 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -296,6 +296,7 @@ from litellm.types.router import ( RouterErrors, RouterGeneralSettings, RouterModelGroupAliasItem, + RouterNoDeploymentsAvailableError, RouterRateLimitError, RouterRateLimitErrorBasic, RoutingContext, @@ -8253,6 +8254,9 @@ class Router: ) return False + if isinstance(exception, RouterNoDeploymentsAvailableError): + return False + # Cache litellm_params to avoid repeated dict lookups litellm_params: Final = kwargs.get("litellm_params", {}) _model_info: Final = litellm_params.get("model_info", {}) diff --git a/litellm/router_strategy/budget_limiter.py b/litellm/router_strategy/budget_limiter.py index 64252cbbfb3..cd5ae66357b 100644 --- a/litellm/router_strategy/budget_limiter.py +++ b/litellm/router_strategy/budget_limiter.py @@ -41,7 +41,12 @@ from litellm.router_utils.cooldown_callbacks import ( _get_prometheus_logger_from_callbacks, ) from litellm.types.llms.openai import AllMessageValues -from litellm.types.router import DeploymentTypedDict, LiteLLM_Params, RouterErrors +from litellm.types.router import ( + DeploymentTypedDict, + LiteLLM_Params, + RouterErrors, + RouterNoDeploymentsAvailableError, +) from litellm.types.utils import BudgetConfig, GenericBudgetConfigType, StandardLoggingPayload from litellm.types.utils import BudgetConfig as GenericBudgetInfo @@ -195,7 +200,7 @@ class RouterBudgetLimiting(CustomLogger): ) if len(potential_deployments) == 0: - raise ValueError( + raise RouterNoDeploymentsAvailableError( f"{RouterErrors.no_deployments_with_provider_budget_routing.value}: {deployment_above_budget_info}" ) diff --git a/litellm/router_strategy/lowest_tpm_rpm_v2.py b/litellm/router_strategy/lowest_tpm_rpm_v2.py index 6c9acbdfecb..1aba8fd1c56 100644 --- a/litellm/router_strategy/lowest_tpm_rpm_v2.py +++ b/litellm/router_strategy/lowest_tpm_rpm_v2.py @@ -16,7 +16,7 @@ from litellm.caching.caching import DualCache from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.core_helpers import _get_parent_otel_span_from_kwargs from litellm.router_utils.batch_utils import is_batch_retrieve_call_type -from litellm.types.router import RouterErrors +from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError from litellm.types.utils import LiteLLMPydanticObjectBase, StandardLoggingPayload from litellm.utils import get_utc_datetime, print_verbose @@ -670,6 +670,6 @@ class LowestTPMLoggingHandler_v2(BaseRoutingStrategy, CustomLogger): "current_rpm": current_rpm, "rpm_limit": _deployment_rpm, } - raise ValueError( + raise RouterNoDeploymentsAvailableError( f"{RouterErrors.no_deployments_available.value}. Passed model={model_group}. Deployments={deployment_dict}" ) diff --git a/litellm/types/router.py b/litellm/types/router.py index 2ab1a1185ed..89c9b075d25 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -942,7 +942,12 @@ class RouterGeneralSettings(BaseModel): ) # if passed a model not llm_router model list, pass through the request to litellm.acompletion/embedding -class RouterRateLimitErrorBasic(ValueError): +class RouterNoDeploymentsAvailableError(ValueError): + status_code: int = 429 + cooldown_time: float | None = None + + +class RouterRateLimitErrorBasic(RouterNoDeploymentsAvailableError): """ Raise a basic error inside helper functions. """ @@ -961,7 +966,7 @@ class RouterErrorTypes(str, enum.Enum): all_deployments_in_cooldown = "all_deployments_in_cooldown" -class RouterRateLimitError(ValueError): +class RouterRateLimitError(RouterNoDeploymentsAvailableError): def __init__( self, model: str, diff --git a/tests/unit/router_strategy/test_budget_limiter_hotpath.py b/tests/unit/router_strategy/test_budget_limiter_hotpath.py index a2c38a898e9..c504d46fe83 100644 --- a/tests/unit/router_strategy/test_budget_limiter_hotpath.py +++ b/tests/unit/router_strategy/test_budget_limiter_hotpath.py @@ -784,3 +784,23 @@ async def test_cancelled_flush_does_not_requeue_an_applied_batch(cancellations: await limiter._push_in_memory_increments_to_redis() assert redis_cache.values[_SPEND_KEY] == 10.0 assert limiter.redis_increment_operation_queue == [] + + +@pytest.mark.asyncio +async def test_every_provider_over_budget_raises_a_429(disable_budget_sync): + from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError + + cache = DualCache() + limiter = RouterBudgetLimiting( + dual_cache=cache, + provider_budget_config={"openai": BudgetConfig(budget_duration="1d", max_budget=1.0)}, + ) + await cache.async_set_cache(key="provider_spend:openai:1d", value=5.0) + deployment = {"litellm_params": {"model": "openai/gpt-4o-mini"}, "model_info": {"id": "d1"}} + + with pytest.raises(RouterNoDeploymentsAvailableError) as raised: + await limiter.async_filter_deployments( + model="gpt-4o-mini", healthy_deployments=[deployment], messages=None, request_kwargs={} + ) + assert raised.value.status_code == 429 + assert RouterErrors.no_deployments_with_provider_budget_routing.value in str(raised.value) diff --git a/tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py b/tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py new file mode 100644 index 00000000000..2619278e6ad --- /dev/null +++ b/tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py @@ -0,0 +1,27 @@ +import pytest + +from litellm.caching.caching import DualCache +from litellm.router_strategy.lowest_tpm_rpm_v2 import LowestTPMLoggingHandler_v2 +from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError +from litellm.utils import get_utc_datetime + + +def test_every_deployment_over_its_tpm_limit_raises_a_429(): + cache = DualCache() + handler = LowestTPMLoggingHandler_v2(router_cache=cache) + deployment = { + "model_name": "gpt-4o-mini", + "litellm_params": {"model": "openai/gpt-4o-mini", "tpm": 10}, + "model_info": {"id": "d1"}, + } + minute = get_utc_datetime().strftime("%H-%M") + cache.set_cache(key=f"d1:openai/gpt-4o-mini:tpm:{minute}", value=100) + + with pytest.raises(RouterNoDeploymentsAvailableError) as raised: + handler.get_available_deployments( + model_group="gpt-4o-mini", + healthy_deployments=[deployment], + messages=[{"role": "user", "content": "hi"}], + ) + assert raised.value.status_code == 429 + assert RouterErrors.no_deployments_available.value in str(raised.value) diff --git a/tests/unit/test_router/test_router.py b/tests/unit/test_router/test_router.py index 96dddf15869..809c6a86ae8 100644 --- a/tests/unit/test_router/test_router.py +++ b/tests/unit/test_router/test_router.py @@ -9821,6 +9821,54 @@ class TestCallerTimeoutCooldown: assert self._cooled_down_ids(router) == [] +class TestPoolExhaustionStatus: + def _router(self): + return litellm.Router( + model_list=[ + { + "model_name": "claude-sonnet-5", + "litellm_params": {"model": "anthropic/claude-sonnet-5", "api_key": "sk-fake"}, + "model_info": {"id": "dep-a"}, + }, + { + "model_name": "claude-sonnet-5", + "litellm_params": {"model": "anthropic/claude-sonnet-5", "api_key": "sk-fake"}, + "model_info": {"id": "dep-b"}, + }, + ], + cooldown_time=60, + ) + + def test_exhausted_pool_raises_a_429(self): + from litellm.router_utils.router_callbacks.track_deployment_metrics import ( + get_deployment_failures_for_current_minute, + ) + from litellm.types.router import RouterRateLimitError + + router = self._router() + for dep_id in ("dep-a", "dep-b"): + router.cooldown_cache.add_deployment_to_cooldown( + model_id=dep_id, + original_exception=litellm.RateLimitError(message="slow down", llm_provider="anthropic", model="x"), + exception_status=429, + cooldown_time=60, + ) + + with pytest.raises(RouterRateLimitError) as raised: + router.get_available_deployment(model="claude-sonnet-5", messages=[{"role": "user", "content": "hi"}]) + assert raised.value.status_code == 429 + assert raised.value.cooldown_time > 0 + + cooled = router.deployment_callback_on_failure( + {"exception": raised.value, "litellm_params": {"model_info": {"id": "dep-a"}, "metadata": {}}}, + None, + datetime.now(), + datetime.now(), + ) + assert cooled is False + assert get_deployment_failures_for_current_minute(litellm_router_instance=router, deployment_id="dep-a") == 0 + + def test_stream_chunks_have_generated_content_detects_text_and_non_text(): from litellm.router import _stream_chunks_have_generated_content from litellm.types.utils import (