mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(router): carry 429 on no-deployment errors, skip them in cooldown
When every deployment of a model group is unavailable the router raises RouterRateLimitError or RouterRateLimitErrorBasic, the provider budget limiter raises a bare ValueError, and the sync usage-based-routing-v2 strategy raises a bare ValueError. None of them carry a status_code. A developer calling litellm.Router directly gets a plain ValueError with nothing to branch on, while the async strategy path already raises litellm.RateLimitError for the same condition. The proxy is not affected on the wire: ProxyException rewrites the code to 429 whenever the message says "No deployments available", and the async strategy path raises a typed 429 All four raises now share RouterNoDeploymentsAvailableError, a ValueError subclass that carries status_code 429 and the cooldown time, so Router users get one typed error for "nothing can serve this call" and code that maps exceptions by status_code no longer needs to sniff the message Router.deployment_callback_on_failure returns early for these errors. The callback counts a failure and runs the cooldown logic against whatever model_info sits in the kwargs it receives, and a routing error is never that deployment's own failure. Today the proxy does not reach this branch with a deployment id attached: the retry resets the logging kwargs before re-selection, and the non-streaming failure log is deduped. The early return keeps a future caller, or a 429 now present on the exception, from turning an exhausted pool into an extra failure or a refreshed cooldown on the last deployment that was tried Co-authored-by: songkuan-zheng <252822057+songkuan-zheng@users.noreply.github.com> Co-authored-by: songkuan-zheng <songkuan-zheng@users.noreply.github.com>
This commit is contained in:
parent
04fa760bf2
commit
455e68bf4a
7 changed files with 115 additions and 6 deletions
|
|
@ -296,6 +296,7 @@ from litellm.types.router import (
|
|||
RouterErrors,
|
||||
RouterGeneralSettings,
|
||||
RouterModelGroupAliasItem,
|
||||
RouterNoDeploymentsAvailableError,
|
||||
RouterRateLimitError,
|
||||
RouterRateLimitErrorBasic,
|
||||
RoutingContext,
|
||||
|
|
@ -8253,6 +8254,9 @@ class Router:
|
|||
)
|
||||
return False
|
||||
|
||||
if isinstance(exception, RouterNoDeploymentsAvailableError):
|
||||
return False
|
||||
|
||||
# Cache litellm_params to avoid repeated dict lookups
|
||||
litellm_params: Final = kwargs.get("litellm_params", {})
|
||||
_model_info: Final = litellm_params.get("model_info", {})
|
||||
|
|
|
|||
|
|
@ -41,7 +41,12 @@ from litellm.router_utils.cooldown_callbacks import (
|
|||
_get_prometheus_logger_from_callbacks,
|
||||
)
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
from litellm.types.router import DeploymentTypedDict, LiteLLM_Params, RouterErrors
|
||||
from litellm.types.router import (
|
||||
DeploymentTypedDict,
|
||||
LiteLLM_Params,
|
||||
RouterErrors,
|
||||
RouterNoDeploymentsAvailableError,
|
||||
)
|
||||
from litellm.types.utils import BudgetConfig, GenericBudgetConfigType, StandardLoggingPayload
|
||||
from litellm.types.utils import BudgetConfig as GenericBudgetInfo
|
||||
|
||||
|
|
@ -195,7 +200,7 @@ class RouterBudgetLimiting(CustomLogger):
|
|||
)
|
||||
|
||||
if len(potential_deployments) == 0:
|
||||
raise ValueError(
|
||||
raise RouterNoDeploymentsAvailableError(
|
||||
f"{RouterErrors.no_deployments_with_provider_budget_routing.value}: {deployment_above_budget_info}"
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ from litellm.caching.caching import DualCache
|
|||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.litellm_core_utils.core_helpers import _get_parent_otel_span_from_kwargs
|
||||
from litellm.router_utils.batch_utils import is_batch_retrieve_call_type
|
||||
from litellm.types.router import RouterErrors
|
||||
from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError
|
||||
from litellm.types.utils import LiteLLMPydanticObjectBase, StandardLoggingPayload
|
||||
from litellm.utils import get_utc_datetime, print_verbose
|
||||
|
||||
|
|
@ -670,6 +670,6 @@ class LowestTPMLoggingHandler_v2(BaseRoutingStrategy, CustomLogger):
|
|||
"current_rpm": current_rpm,
|
||||
"rpm_limit": _deployment_rpm,
|
||||
}
|
||||
raise ValueError(
|
||||
raise RouterNoDeploymentsAvailableError(
|
||||
f"{RouterErrors.no_deployments_available.value}. Passed model={model_group}. Deployments={deployment_dict}"
|
||||
)
|
||||
|
|
|
|||
|
|
@ -942,7 +942,12 @@ class RouterGeneralSettings(BaseModel):
|
|||
) # if passed a model not llm_router model list, pass through the request to litellm.acompletion/embedding
|
||||
|
||||
|
||||
class RouterRateLimitErrorBasic(ValueError):
|
||||
class RouterNoDeploymentsAvailableError(ValueError):
|
||||
status_code: int = 429
|
||||
cooldown_time: float | None = None
|
||||
|
||||
|
||||
class RouterRateLimitErrorBasic(RouterNoDeploymentsAvailableError):
|
||||
"""
|
||||
Raise a basic error inside helper functions.
|
||||
"""
|
||||
|
|
@ -961,7 +966,7 @@ class RouterErrorTypes(str, enum.Enum):
|
|||
all_deployments_in_cooldown = "all_deployments_in_cooldown"
|
||||
|
||||
|
||||
class RouterRateLimitError(ValueError):
|
||||
class RouterRateLimitError(RouterNoDeploymentsAvailableError):
|
||||
def __init__(
|
||||
self,
|
||||
model: str,
|
||||
|
|
|
|||
|
|
@ -784,3 +784,23 @@ async def test_cancelled_flush_does_not_requeue_an_applied_batch(cancellations:
|
|||
await limiter._push_in_memory_increments_to_redis()
|
||||
assert redis_cache.values[_SPEND_KEY] == 10.0
|
||||
assert limiter.redis_increment_operation_queue == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_every_provider_over_budget_raises_a_429(disable_budget_sync):
|
||||
from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError
|
||||
|
||||
cache = DualCache()
|
||||
limiter = RouterBudgetLimiting(
|
||||
dual_cache=cache,
|
||||
provider_budget_config={"openai": BudgetConfig(budget_duration="1d", max_budget=1.0)},
|
||||
)
|
||||
await cache.async_set_cache(key="provider_spend:openai:1d", value=5.0)
|
||||
deployment = {"litellm_params": {"model": "openai/gpt-4o-mini"}, "model_info": {"id": "d1"}}
|
||||
|
||||
with pytest.raises(RouterNoDeploymentsAvailableError) as raised:
|
||||
await limiter.async_filter_deployments(
|
||||
model="gpt-4o-mini", healthy_deployments=[deployment], messages=None, request_kwargs={}
|
||||
)
|
||||
assert raised.value.status_code == 429
|
||||
assert RouterErrors.no_deployments_with_provider_budget_routing.value in str(raised.value)
|
||||
|
|
|
|||
27
tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py
Normal file
27
tests/unit/router_strategy/test_lowest_tpm_rpm_v2.py
Normal file
|
|
@ -0,0 +1,27 @@
|
|||
import pytest
|
||||
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.router_strategy.lowest_tpm_rpm_v2 import LowestTPMLoggingHandler_v2
|
||||
from litellm.types.router import RouterErrors, RouterNoDeploymentsAvailableError
|
||||
from litellm.utils import get_utc_datetime
|
||||
|
||||
|
||||
def test_every_deployment_over_its_tpm_limit_raises_a_429():
|
||||
cache = DualCache()
|
||||
handler = LowestTPMLoggingHandler_v2(router_cache=cache)
|
||||
deployment = {
|
||||
"model_name": "gpt-4o-mini",
|
||||
"litellm_params": {"model": "openai/gpt-4o-mini", "tpm": 10},
|
||||
"model_info": {"id": "d1"},
|
||||
}
|
||||
minute = get_utc_datetime().strftime("%H-%M")
|
||||
cache.set_cache(key=f"d1:openai/gpt-4o-mini:tpm:{minute}", value=100)
|
||||
|
||||
with pytest.raises(RouterNoDeploymentsAvailableError) as raised:
|
||||
handler.get_available_deployments(
|
||||
model_group="gpt-4o-mini",
|
||||
healthy_deployments=[deployment],
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
)
|
||||
assert raised.value.status_code == 429
|
||||
assert RouterErrors.no_deployments_available.value in str(raised.value)
|
||||
|
|
@ -9821,6 +9821,54 @@ class TestCallerTimeoutCooldown:
|
|||
assert self._cooled_down_ids(router) == []
|
||||
|
||||
|
||||
class TestPoolExhaustionStatus:
|
||||
def _router(self):
|
||||
return litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "claude-sonnet-5",
|
||||
"litellm_params": {"model": "anthropic/claude-sonnet-5", "api_key": "sk-fake"},
|
||||
"model_info": {"id": "dep-a"},
|
||||
},
|
||||
{
|
||||
"model_name": "claude-sonnet-5",
|
||||
"litellm_params": {"model": "anthropic/claude-sonnet-5", "api_key": "sk-fake"},
|
||||
"model_info": {"id": "dep-b"},
|
||||
},
|
||||
],
|
||||
cooldown_time=60,
|
||||
)
|
||||
|
||||
def test_exhausted_pool_raises_a_429(self):
|
||||
from litellm.router_utils.router_callbacks.track_deployment_metrics import (
|
||||
get_deployment_failures_for_current_minute,
|
||||
)
|
||||
from litellm.types.router import RouterRateLimitError
|
||||
|
||||
router = self._router()
|
||||
for dep_id in ("dep-a", "dep-b"):
|
||||
router.cooldown_cache.add_deployment_to_cooldown(
|
||||
model_id=dep_id,
|
||||
original_exception=litellm.RateLimitError(message="slow down", llm_provider="anthropic", model="x"),
|
||||
exception_status=429,
|
||||
cooldown_time=60,
|
||||
)
|
||||
|
||||
with pytest.raises(RouterRateLimitError) as raised:
|
||||
router.get_available_deployment(model="claude-sonnet-5", messages=[{"role": "user", "content": "hi"}])
|
||||
assert raised.value.status_code == 429
|
||||
assert raised.value.cooldown_time > 0
|
||||
|
||||
cooled = router.deployment_callback_on_failure(
|
||||
{"exception": raised.value, "litellm_params": {"model_info": {"id": "dep-a"}, "metadata": {}}},
|
||||
None,
|
||||
datetime.now(),
|
||||
datetime.now(),
|
||||
)
|
||||
assert cooled is False
|
||||
assert get_deployment_failures_for_current_minute(litellm_router_instance=router, deployment_id="dep-a") == 0
|
||||
|
||||
|
||||
def test_stream_chunks_have_generated_content_detects_text_and_non_text():
|
||||
from litellm.router import _stream_chunks_have_generated_content
|
||||
from litellm.types.utils import (
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue