diff --git a/litellm/__init__.py b/litellm/__init__.py index 3ffb2124956..ea30c5e123f 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1256,6 +1256,7 @@ from .exceptions import ( PermissionDeniedError, RateLimitError, RateLimitErrorCategory, + RateLimitType, ServiceUnavailableError, BadGatewayError, OpenAIError, diff --git a/litellm/exceptions.py b/litellm/exceptions.py index 4f319842119..586ccc77985 100644 --- a/litellm/exceptions.py +++ b/litellm/exceptions.py @@ -48,6 +48,39 @@ class RateLimitErrorCategory(str, enum.Enum): """LiteLLM's own batch rate limiter (token/request budget across a batch input file) blocked the request.""" +class RateLimitType(str, enum.Enum): + """ + The dimension that was exceeded when a rate-limit error fired. + + This is orthogonal to :class:`RateLimitErrorCategory` — *category* tells + callers **who** rate-limited the request (the upstream vendor vs. one of + litellm's own limiters), while *type* tells them **which limit dimension** + was exceeded (an RPM ceiling, a TPM ceiling, a max-parallel-requests + ceiling, a budget cap, or a max-iterations cap). + + Surfaced both on every :class:`RateLimitError` instance via the + ``rate_limit_type`` attribute and on the structured + ``StandardLoggingPayload.error_information.error_rate_limit_type`` field + so custom callbacks / metrics consumers can split rate-limit failures by + cause without parsing free-text error messages. + """ + + REQUESTS = "requests" + """Requests-per-minute (RPM) or requests-per-window ceiling exceeded.""" + + TOKENS = "tokens" + """Tokens-per-minute (TPM) or tokens-per-window ceiling exceeded.""" + + CONCURRENT_REQUESTS = "concurrent_requests" + """``max_parallel_requests`` — too many in-flight requests at once.""" + + BUDGET = "budget" + """Spend budget cap reached (key, team, user, or per-session).""" + + MAX_ITERATIONS = "max_iterations" + """Per-session max-iterations cap reached (agent-style flows).""" + + _MINIMAL_ERROR_RESPONSE: Optional[httpx.Response] = None @@ -377,6 +410,7 @@ class RateLimitError(openai.RateLimitError): # type: ignore category: Union[str, RateLimitErrorCategory] = ( RateLimitErrorCategory.VENDOR_RATE_LIMIT ), + rate_limit_type: Optional[Union[str, RateLimitType]] = None, headers: Optional[Dict[str, str]] = None, detail: Any = None, ): @@ -390,6 +424,14 @@ class RateLimitError(openai.RateLimitError): # type: ignore self.category = ( category.value if isinstance(category, RateLimitErrorCategory) else category ) + # Which dimension was exceeded — request count, token count, parallel + # requests, budget, max iterations. None when the source didn't + # classify the failure (e.g. legacy vendor 429 with no header hints). + self.rate_limit_type: Optional[str] = ( + rate_limit_type.value + if isinstance(rate_limit_type, RateLimitType) + else rate_limit_type + ) # Headers explicitly attached to the error (e.g. retry-after, # rate_limit_type, reset_at). Preserved across the proxy boundary so # clients can react appropriately. diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index c09f191c340..3bcb5aa5fc8 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -5160,12 +5160,20 @@ class StandardLoggingPayloadSetup: error_message = str(original_exception) # For rate-limit errors (litellm.RateLimitError + the proxy-side - # ProxyRateLimitError subclass), surface the unified `category` field - # so callbacks can distinguish vendor vs. litellm rate limits without - # reaching for the raw exception object. + # ProxyRateLimitError subclass), surface the unified `category` and + # `rate_limit_type` fields so callbacks can distinguish vendor vs. + # litellm rate limits AND split by the dimension that was exceeded + # (requests / tokens / concurrent_requests / budget / max_iterations) + # without reaching for the raw exception object. + is_rate_limit_error = isinstance(original_exception, litellm.RateLimitError) rate_limit_category: Optional[str] = ( getattr(original_exception, "category", None) - if isinstance(original_exception, litellm.RateLimitError) + if is_rate_limit_error + else None + ) + rate_limit_type: Optional[str] = ( + getattr(original_exception, "rate_limit_type", None) + if is_rate_limit_error else None ) @@ -5176,6 +5184,7 @@ class StandardLoggingPayloadSetup: traceback=traceback_info, error_message=error_message if original_exception else "", error_rate_limit_category=rate_limit_category, + error_rate_limit_type=rate_limit_type, ) @staticmethod diff --git a/litellm/proxy/common_utils/proxy_rate_limit_error.py b/litellm/proxy/common_utils/proxy_rate_limit_error.py index 084f2150b26..c3a00add875 100644 --- a/litellm/proxy/common_utils/proxy_rate_limit_error.py +++ b/litellm/proxy/common_utils/proxy_rate_limit_error.py @@ -40,7 +40,29 @@ from typing import Any, Dict, Mapping, Optional, Union from fastapi import HTTPException -from litellm.exceptions import RateLimitError, RateLimitErrorCategory +from litellm.exceptions import RateLimitError, RateLimitErrorCategory, RateLimitType + + +def map_v3_rate_limit_type( + v3_value: Optional[str], +) -> Optional[RateLimitType]: + """ + Map the v3 rate limiter's internal `status["rate_limit_type"]` strings + onto the public :class:`RateLimitType` enum. + + The v3 limiter uses the literal values ``"requests"``, ``"tokens"``, and + ``"max_parallel_requests"``. We collapse the last one onto + :attr:`RateLimitType.CONCURRENT_REQUESTS` because that's the public name + documented for users and dashboards. Unrecognized values return ``None`` + so the field stays absent rather than carrying garbage downstream. + """ + if v3_value == "tokens": + return RateLimitType.TOKENS + if v3_value == "max_parallel_requests": + return RateLimitType.CONCURRENT_REQUESTS + if v3_value == "requests": + return RateLimitType.REQUESTS + return None def _coerce_message(detail: Any) -> str: @@ -116,6 +138,7 @@ class ProxyRateLimitError(HTTPException, RateLimitError): # type: ignore[misc] category: Union[ str, RateLimitErrorCategory ] = RateLimitErrorCategory.LITELLM_RATE_LIMIT, + rate_limit_type: Optional[Union[str, RateLimitType]] = None, model: Optional[str] = None, llm_provider: str = "litellm_proxy", ): @@ -144,6 +167,7 @@ class ProxyRateLimitError(HTTPException, RateLimitError): # type: ignore[misc] llm_provider=llm_provider, model=model or "", category=category, + rate_limit_type=rate_limit_type, headers=stringified_headers, detail=detail, ) diff --git a/litellm/types/utils.py b/litellm/types/utils.py index a00f5f982e3..0deb07e8acf 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2705,6 +2705,15 @@ class StandardLoggingPayloadErrorInformation(TypedDict, total=False): # Surfaced here so custom callbacks / metrics consumers can switch on # the rate-limit source without reaching for the raw exception. error_rate_limit_category: Optional[str] + # error_rate_limit_type: + # For 429 / rate-limit errors, the dimension that was exceeded. One of + # the string values defined by `litellm.exceptions.RateLimitType` + # (requests, tokens, concurrent_requests, budget, max_iterations). + # None for non-rate-limit exceptions and for rate-limit exceptions that + # did not classify the failure (e.g. legacy vendor 429 with no header + # hints). Lets dashboards split rate-limit failures by cause without + # parsing free-text error messages. + error_rate_limit_type: Optional[str] class GuardrailMode(TypedDict, total=False):