From 455e68bf4aea00d1bc051200ede85f6c7ebf3a35 Mon Sep 17 00:00:00 2001 From: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Date: Fri, 11 Sep 2026 11:16:54 +0000 Subject: [PATCH 1/5] fix(router): carry 429 on no-deployment errors, skip them in cooldown When every deployment of a model group is unavailable the router raises RouterRateLimitError or RouterRateLimitErrorBasic, the provider budget limiter raises a bare ValueError, and the sync usage-based-routing-v2 strategy raises a bare ValueError. None of them carry a status_code. A developer calling litellm.Router directly gets a plain ValueError with nothing to branch on, while the async strategy path already raises litellm.RateLimitError for the same condition. The proxy is not affected on the wire: ProxyException rewrites the code to 429 whenever the message says "No deployments available", and the async strategy path raises a typed 429 All four raises now share RouterNoDeploymentsAvailableError, a ValueError subclass that carries status_code 429 and the cooldown time, so Router users get one typed error for "nothing can serve this call" and code that maps exceptions by status_code no longer needs to sniff the message Router.deployment_callback_on_failure returns early for these errors. The callback counts a failure and runs the cooldown logic against whatever model_info sits in the kwargs it receives, and a routing error is never that deployment's own failure. Today the proxy does not reach this branch with a deployment id attached: the retry resets the logging kwargs before re-selection, and the non-streaming failure log is deduped. The early return keeps a future caller, or a 429 now present on the exception, from turning an exhausted pool into an extra failure or a refreshed cooldown on the last deployment that was tried Co-authored-by: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Co-authored-by: songkuan-zheng --- litellm/router.py | 4 ++ litellm/router_strategy/budget_limiter.py | 9 +++- litellm/router_strategy/lowest_tpm_rpm_v2.py | 4 +- litellm/types/router.py | 9 +++- .../test_budget_limiter_hotpath.py | 20 ++++++++ .../router_strategy/test_lowest_tpm_rpm_v2.py | 27 +++++++++++ tests/unit/test_router/test_router.py | 48 +++++++++++++++++++ 7 files changed, 115 insertions(+), 6 deletions(-) create mode 100644 tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py diff --git a/litellm/router.py b/litellm/router.py index 115faad000c..3a2dc2eeb97 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -296,6 +296,7 @@ from litellm.types.router import ( RouterErrors, RouterGeneralSettings, RouterModelGroupAliasItem, + RouterNoDeploymentsAvailableError, RouterRateLimitError, RouterRateLimitErrorBasic, RoutingContext, @@ -8253,6 +8254,9 @@ class Router: ) return False + if isinstance(exception, RouterNoDeploymentsAvailableError): + return False + # Cache litellm_params to avoid repeated dict lookups litellm_params: Final = kwargs.get("litellm_params", {}) _model_info: Final = litellm_params.get("model_info", {}) diff --git a/litellm/router_strategy/budget_limiter.py b/litellm/router_strategy/budget_limiter.py index 64252cbbfb3..cd5ae66357b 100644 --- a/litellm/router_strategy/budget_limiter.py +++ b/litellm/router_strategy/budget_limiter.py @@ -41,7 +41,12 @@ from litellm.router_utils.cooldown_callbacks import ( _get_prometheus_logger_from_callbacks, ) from litellm.types.llms.openai import AllMessageValues -from litellm.types.router import DeploymentTypedDict, LiteLLM_Params, RouterErrors +from litellm.types.router import ( + DeploymentTypedDict, + LiteLLM_Params, + RouterErrors, + RouterNoDeploymentsAvailableError, +) from litellm.types.utils import BudgetConfig, GenericBudgetConfigType, StandardLoggingPayload from litellm.types.utils import BudgetConfig as GenericBudgetInfo @@ -195,7 +200,7 @@ class RouterBudgetLimiting(CustomLogger): ) if len(potential_deployments) == 0: - raise ValueError( + raise RouterNoDeploymentsAvailableError( f"{RouterErrors.no_deployments_with_provider_budget_routing.value}: {deployment_above_budget_info}" ) diff --git a/litellm/router_strategy/lowest_tpm_rpm_v2.py b/litellm/router_strategy/lowest_tpm_rpm_v2.py index 6c9acbdfecb..1aba8fd1c56 100644 --- a/litellm/router_strategy/lowest_tpm_rpm_v2.py +++ b/litellm/router_strategy/lowest_tpm_rpm_v2.py @@ -16,7 +16,7 @@ from litellm.caching.caching import DualCache from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.core_helpers import _get_parent_otel_span_from_kwargs from litellm.router_utils.batch_utils import is_batch_retrieve_call_type -from litellm.types.router import RouterErrors +from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError from litellm.types.utils import LiteLLMPydanticObjectBase, StandardLoggingPayload from litellm.utils import get_utc_datetime, print_verbose @@ -670,6 +670,6 @@ class LowestTPMLoggingHandler_v2(BaseRoutingStrategy, CustomLogger): "current_rpm": current_rpm, "rpm_limit": _deployment_rpm, } - raise ValueError( + raise RouterNoDeploymentsAvailableError( f"{RouterErrors.no_deployments_available.value}. Passed model={model_group}. Deployments={deployment_dict}" ) diff --git a/litellm/types/router.py b/litellm/types/router.py index 2ab1a1185ed..89c9b075d25 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -942,7 +942,12 @@ class RouterGeneralSettings(BaseModel): ) # if passed a model not llm_router model list, pass through the request to litellm.acompletion/embedding -class RouterRateLimitErrorBasic(ValueError): +class RouterNoDeploymentsAvailableError(ValueError): + status_code: int = 429 + cooldown_time: float | None = None + + +class RouterRateLimitErrorBasic(RouterNoDeploymentsAvailableError): """ Raise a basic error inside helper functions. """ @@ -961,7 +966,7 @@ class RouterErrorTypes(str, enum.Enum): all_deployments_in_cooldown = "all_deployments_in_cooldown" -class RouterRateLimitError(ValueError): +class RouterRateLimitError(RouterNoDeploymentsAvailableError): def __init__( self, model: str, diff --git a/tests/unit/router_strategy/test_budget_limiter_hotpath.py b/tests/unit/router_strategy/test_budget_limiter_hotpath.py index a2c38a898e9..c504d46fe83 100644 --- a/tests/unit/router_strategy/test_budget_limiter_hotpath.py +++ b/tests/unit/router_strategy/test_budget_limiter_hotpath.py @@ -784,3 +784,23 @@ async def test_cancelled_flush_does_not_requeue_an_applied_batch(cancellations: await limiter._push_in_memory_increments_to_redis() assert redis_cache.values[_SPEND_KEY] == 10.0 assert limiter.redis_increment_operation_queue == [] + + +@pytest.mark.asyncio +async def test_every_provider_over_budget_raises_a_429(disable_budget_sync): + from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError + + cache = DualCache() + limiter = RouterBudgetLimiting( + dual_cache=cache, + provider_budget_config={"openai": BudgetConfig(budget_duration="1d", max_budget=1.0)}, + ) + await cache.async_set_cache(key="provider_spend:openai:1d", value=5.0) + deployment = {"litellm_params": {"model": "openai/gpt-4o-mini"}, "model_info": {"id": "d1"}} + + with pytest.raises(RouterNoDeploymentsAvailableError) as raised: + await limiter.async_filter_deployments( + model="gpt-4o-mini", healthy_deployments=[deployment], messages=None, request_kwargs={} + ) + assert raised.value.status_code == 429 + assert RouterErrors.no_deployments_with_provider_budget_routing.value in str(raised.value) diff --git a/tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py b/tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py new file mode 100644 index 00000000000..2619278e6ad --- /dev/null +++ b/tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py @@ -0,0 +1,27 @@ +import pytest + +from litellm.caching.caching import DualCache +from litellm.router_strategy.lowest_tpm_rpm_v2 import LowestTPMLoggingHandler_v2 +from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError +from litellm.utils import get_utc_datetime + + +def test_every_deployment_over_its_tpm_limit_raises_a_429(): + cache = DualCache() + handler = LowestTPMLoggingHandler_v2(router_cache=cache) + deployment = { + "model_name": "gpt-4o-mini", + "litellm_params": {"model": "openai/gpt-4o-mini", "tpm": 10}, + "model_info": {"id": "d1"}, + } + minute = get_utc_datetime().strftime("%H-%M") + cache.set_cache(key=f"d1:openai/gpt-4o-mini:tpm:{minute}", value=100) + + with pytest.raises(RouterNoDeploymentsAvailableError) as raised: + handler.get_available_deployments( + model_group="gpt-4o-mini", + healthy_deployments=[deployment], + messages=[{"role": "user", "content": "hi"}], + ) + assert raised.value.status_code == 429 + assert RouterErrors.no_deployments_available.value in str(raised.value) diff --git a/tests/unit/test_router/test_router.py b/tests/unit/test_router/test_router.py index 96dddf15869..809c6a86ae8 100644 --- a/tests/unit/test_router/test_router.py +++ b/tests/unit/test_router/test_router.py @@ -9821,6 +9821,54 @@ class TestCallerTimeoutCooldown: assert self._cooled_down_ids(router) == [] +class TestPoolExhaustionStatus: + def _router(self): + return litellm.Router( + model_list=[ + { + "model_name": "claude-sonnet-5", + "litellm_params": {"model": "anthropic/claude-sonnet-5", "api_key": "sk-fake"}, + "model_info": {"id": "dep-a"}, + }, + { + "model_name": "claude-sonnet-5", + "litellm_params": {"model": "anthropic/claude-sonnet-5", "api_key": "sk-fake"}, + "model_info": {"id": "dep-b"}, + }, + ], + cooldown_time=60, + ) + + def test_exhausted_pool_raises_a_429(self): + from litellm.router_utils.router_callbacks.track_deployment_metrics import ( + get_deployment_failures_for_current_minute, + ) + from litellm.types.router import RouterRateLimitError + + router = self._router() + for dep_id in ("dep-a", "dep-b"): + router.cooldown_cache.add_deployment_to_cooldown( + model_id=dep_id, + original_exception=litellm.RateLimitError(message="slow down", llm_provider="anthropic", model="x"), + exception_status=429, + cooldown_time=60, + ) + + with pytest.raises(RouterRateLimitError) as raised: + router.get_available_deployment(model="claude-sonnet-5", messages=[{"role": "user", "content": "hi"}]) + assert raised.value.status_code == 429 + assert raised.value.cooldown_time > 0 + + cooled = router.deployment_callback_on_failure( + {"exception": raised.value, "litellm_params": {"model_info": {"id": "dep-a"}, "metadata": {}}}, + None, + datetime.now(), + datetime.now(), + ) + assert cooled is False + assert get_deployment_failures_for_current_minute(litellm_router_instance=router, deployment_id="dep-a") == 0 + + def test_stream_chunks_have_generated_content_detects_text_and_non_text(): from litellm.router import _stream_chunks_have_generated_content from litellm.types.utils import ( From 20e7ac6d8a1f8c56d5b79f71aacd22099c6b8142 Mon Sep 17 00:00:00 2001 From: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Date: Fri, 11 Sep 2026 12:06:35 +0000 Subject: [PATCH 2/5] test(router): move the sync tpm regression into the existing tpm routing test file tests/local_testing/test_tpm_rpm_routing_v2.py is the test file that already covers LowestTPMLoggingHandler_v2, so the regression for the sync get_available_deployments raise lives there instead of in a new file Co-authored-by: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Co-authored-by: songkuan-zheng --- .../local_testing/test_tpm_rpm_routing_v2.py | 23 ++++++++++++++++ .../router_strategy/test_lowest_tpm_rpm_v2.py | 27 ------------------- 2 files changed, 23 insertions(+), 27 deletions(-) delete mode 100644 tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py diff --git a/tests/local_testing/test_tpm_rpm_routing_v2.py b/tests/local_testing/test_tpm_rpm_routing_v2.py index 7478bd253b6..228046e8cb8 100644 --- a/tests/local_testing/test_tpm_rpm_routing_v2.py +++ b/tests/local_testing/test_tpm_rpm_routing_v2.py @@ -770,3 +770,26 @@ async def test_tpm_rpm_routing_model_name_checks(): standard_logging_payload["hidden_params"]["litellm_model_name"] == "azure/gpt-4.1-mini" ) + + +def test_every_deployment_over_its_tpm_limit_raises_a_429(): + from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError + + test_cache = DualCache() + lowest_tpm_logger = LowestTPMLoggingHandler(router_cache=test_cache) + deployment = { + "model_name": "gpt-4o-mini", + "litellm_params": {"model": "openai/gpt-4o-mini", "tpm": 10}, + "model_info": {"id": "d1"}, + } + minute = get_utc_datetime().strftime("%H-%M") + test_cache.set_cache(key=f"d1:openai/gpt-4o-mini:tpm:{minute}", value=100) + + with pytest.raises(RouterNoDeploymentsAvailableError) as raised: + lowest_tpm_logger.get_available_deployments( + model_group="gpt-4o-mini", + healthy_deployments=[deployment], + messages=[{"role": "user", "content": "hi"}], + ) + assert raised.value.status_code == 429 + assert RouterErrors.no_deployments_available.value in str(raised.value) diff --git a/tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py b/tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py deleted file mode 100644 index 2619278e6ad..00000000000 --- a/tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py +++ /dev/null @@ -1,27 +0,0 @@ -import pytest - -from litellm.caching.caching import DualCache -from litellm.router_strategy.lowest_tpm_rpm_v2 import LowestTPMLoggingHandler_v2 -from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError -from litellm.utils import get_utc_datetime - - -def test_every_deployment_over_its_tpm_limit_raises_a_429(): - cache = DualCache() - handler = LowestTPMLoggingHandler_v2(router_cache=cache) - deployment = { - "model_name": "gpt-4o-mini", - "litellm_params": {"model": "openai/gpt-4o-mini", "tpm": 10}, - "model_info": {"id": "d1"}, - } - minute = get_utc_datetime().strftime("%H-%M") - cache.set_cache(key=f"d1:openai/gpt-4o-mini:tpm:{minute}", value=100) - - with pytest.raises(RouterNoDeploymentsAvailableError) as raised: - handler.get_available_deployments( - model_group="gpt-4o-mini", - healthy_deployments=[deployment], - messages=[{"role": "user", "content": "hi"}], - ) - assert raised.value.status_code == 429 - assert RouterErrors.no_deployments_available.value in str(raised.value) From 05abb93c5d4ddfb4b20d6440c2fbacc6d88a06a9 Mon Sep 17 00:00:00 2001 From: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Date: Fri, 11 Sep 2026 12:19:36 +0000 Subject: [PATCH 3/5] fix(router): declare cooldown_time as float on RouterRateLimitError The shared base declares cooldown_time as float | None because the budget limiter and the sync tpm strategy have no window to report. RouterRateLimitError always receives one, so it redeclares the attribute as float and callers that compare it keep a narrowed type Co-authored-by: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Co-authored-by: songkuan-zheng --- litellm/types/router.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/litellm/types/router.py b/litellm/types/router.py index 89c9b075d25..5ebf37d4022 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -967,6 +967,8 @@ class RouterErrorTypes(str, enum.Enum): class RouterRateLimitError(RouterNoDeploymentsAvailableError): + cooldown_time: float + def __init__( self, model: str, From 8f3a7b54c7b7b63729d502b25d5de331dd88b36c Mon Sep 17 00:00:00 2001 From: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Date: Fri, 11 Sep 2026 12:29:43 +0000 Subject: [PATCH 4/5] fix(router): keep cooldown_time off the shared no-deployment base error Declaring cooldown_time as float | None on the base widened the attribute on RouterRateLimitError, whose callers compare it as a float, and redeclaring it as float on the subclass is an incompatible override. The base now carries only status_code; RouterRateLimitError keeps its own float cooldown_time exactly as before Co-authored-by: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Co-authored-by: songkuan-zheng --- litellm/types/router.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/litellm/types/router.py b/litellm/types/router.py index 5ebf37d4022..5ae5c3a91b0 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -944,7 +944,6 @@ class RouterGeneralSettings(BaseModel): class RouterNoDeploymentsAvailableError(ValueError): status_code: int = 429 - cooldown_time: float | None = None class RouterRateLimitErrorBasic(RouterNoDeploymentsAvailableError): @@ -967,8 +966,6 @@ class RouterErrorTypes(str, enum.Enum): class RouterRateLimitError(RouterNoDeploymentsAvailableError): - cooldown_time: float - def __init__( self, model: str, From be50ed37731647cc361b3991561fe500f76fc381 Mon Sep 17 00:00:00 2001 From: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Date: Tue, 15 Sep 2026 07:35:30 +0000 Subject: [PATCH 5/5] fix(router): import the no-deployments error where the cooldown callback uses it Keeps the new symbol out of the module-level litellm.router and litellm.types.router cycle that CodeQL flags. Co-authored-by: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Co-authored-by: songkuan-zheng --- litellm/router.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/litellm/router.py b/litellm/router.py index 3a2dc2eeb97..074a160c760 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -296,7 +296,6 @@ from litellm.types.router import ( RouterErrors, RouterGeneralSettings, RouterModelGroupAliasItem, - RouterNoDeploymentsAvailableError, RouterRateLimitError, RouterRateLimitErrorBasic, RoutingContext, @@ -8254,6 +8253,8 @@ class Router: ) return False + from litellm.types.router import RouterNoDeploymentsAvailableError + if isinstance(exception, RouterNoDeploymentsAvailableError): return False