feat(rate-limit): add orthogonal RateLimitType (requests/tokens/concurrent_requests/budget/max_iterations)

trho's last ask in the LIT-2968 thread: distinguish rate-limit failures by
the dimension that was exceeded, not just by who rate-limited (vendor vs.
litellm). Adds:

- RateLimitType str-enum exposed at `litellm.RateLimitType` with values
  requests / tokens / concurrent_requests / budget / max_iterations.
- `rate_limit_type` kwarg on litellm.RateLimitError + ProxyRateLimitError;
  None default so existing callers (vendor-429 path in exception_mapping_utils)
  remain a no-op.
- StandardLoggingPayloadErrorInformation.error_rate_limit_type so custom
  callbacks can split rate-limit failures by cause without parsing free-text
  error messages. Mirror to error_rate_limit_category extraction in
  get_error_information(); single isinstance(RateLimitError) check covers both.
- map_v3_rate_limit_type() helper to collapse the v3 limiter's internal labels
  ("requests", "tokens", "max_parallel_requests") onto the public enum so
  the v3 limiter and dynamic_rate_limiter_v3 share one mapping. Defensive
  None on unknown values rather than silently picking a wrong dimension.

Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-05-12 02:08:06 +00:00
parent bcf1989aef
commit 9778f94ec2
No known key found for this signature in database
5 changed files with 90 additions and 5 deletions

View file

@ -1256,6 +1256,7 @@ from .exceptions import (
PermissionDeniedError,
RateLimitError,
RateLimitErrorCategory,
RateLimitType,
ServiceUnavailableError,
BadGatewayError,
OpenAIError,

View file

@ -48,6 +48,39 @@ class RateLimitErrorCategory(str, enum.Enum):
"""LiteLLM's own batch rate limiter (token/request budget across a batch input file) blocked the request."""
class RateLimitType(str, enum.Enum):
"""
The dimension that was exceeded when a rate-limit error fired.
This is orthogonal to :class:`RateLimitErrorCategory` — *category* tells
callers **who** rate-limited the request (the upstream vendor vs. one of
litellm's own limiters), while *type* tells them **which limit dimension**
was exceeded (an RPM ceiling, a TPM ceiling, a max-parallel-requests
ceiling, a budget cap, or a max-iterations cap).
Surfaced both on every :class:`RateLimitError` instance via the
``rate_limit_type`` attribute and on the structured
``StandardLoggingPayload.error_information.error_rate_limit_type`` field
so custom callbacks / metrics consumers can split rate-limit failures by
cause without parsing free-text error messages.
"""
REQUESTS = "requests"
"""Requests-per-minute (RPM) or requests-per-window ceiling exceeded."""
TOKENS = "tokens"
"""Tokens-per-minute (TPM) or tokens-per-window ceiling exceeded."""
CONCURRENT_REQUESTS = "concurrent_requests"
"""``max_parallel_requests`` — too many in-flight requests at once."""
BUDGET = "budget"
"""Spend budget cap reached (key, team, user, or per-session)."""
MAX_ITERATIONS = "max_iterations"
"""Per-session max-iterations cap reached (agent-style flows)."""
_MINIMAL_ERROR_RESPONSE: Optional[httpx.Response] = None
@ -377,6 +410,7 @@ class RateLimitError(openai.RateLimitError): # type: ignore
category: Union[str, RateLimitErrorCategory] = (
RateLimitErrorCategory.VENDOR_RATE_LIMIT
),
rate_limit_type: Optional[Union[str, RateLimitType]] = None,
headers: Optional[Dict[str, str]] = None,
detail: Any = None,
):
@ -390,6 +424,14 @@ class RateLimitError(openai.RateLimitError): # type: ignore
self.category = (
category.value if isinstance(category, RateLimitErrorCategory) else category
)
# Which dimension was exceeded — request count, token count, parallel
# requests, budget, max iterations. None when the source didn't
# classify the failure (e.g. legacy vendor 429 with no header hints).
self.rate_limit_type: Optional[str] = (
rate_limit_type.value
if isinstance(rate_limit_type, RateLimitType)
else rate_limit_type
)
# Headers explicitly attached to the error (e.g. retry-after,
# rate_limit_type, reset_at). Preserved across the proxy boundary so
# clients can react appropriately.

View file

@ -5160,12 +5160,20 @@ class StandardLoggingPayloadSetup:
error_message = str(original_exception)
# For rate-limit errors (litellm.RateLimitError + the proxy-side
# ProxyRateLimitError subclass), surface the unified `category` field
# so callbacks can distinguish vendor vs. litellm rate limits without
# reaching for the raw exception object.
# ProxyRateLimitError subclass), surface the unified `category` and
# `rate_limit_type` fields so callbacks can distinguish vendor vs.
# litellm rate limits AND split by the dimension that was exceeded
# (requests / tokens / concurrent_requests / budget / max_iterations)
# without reaching for the raw exception object.
is_rate_limit_error = isinstance(original_exception, litellm.RateLimitError)
rate_limit_category: Optional[str] = (
getattr(original_exception, "category", None)
if isinstance(original_exception, litellm.RateLimitError)
if is_rate_limit_error
else None
)
rate_limit_type: Optional[str] = (
getattr(original_exception, "rate_limit_type", None)
if is_rate_limit_error
else None
)
@ -5176,6 +5184,7 @@ class StandardLoggingPayloadSetup:
traceback=traceback_info,
error_message=error_message if original_exception else "",
error_rate_limit_category=rate_limit_category,
error_rate_limit_type=rate_limit_type,
)
@staticmethod

View file

@ -40,7 +40,29 @@ from typing import Any, Dict, Mapping, Optional, Union
from fastapi import HTTPException
from litellm.exceptions import RateLimitError, RateLimitErrorCategory
from litellm.exceptions import RateLimitError, RateLimitErrorCategory, RateLimitType
def map_v3_rate_limit_type(
v3_value: Optional[str],
) -> Optional[RateLimitType]:
"""
Map the v3 rate limiter's internal `status["rate_limit_type"]` strings
onto the public :class:`RateLimitType` enum.
The v3 limiter uses the literal values ``"requests"``, ``"tokens"``, and
``"max_parallel_requests"``. We collapse the last one onto
:attr:`RateLimitType.CONCURRENT_REQUESTS` because that's the public name
documented for users and dashboards. Unrecognized values return ``None``
so the field stays absent rather than carrying garbage downstream.
"""
if v3_value == "tokens":
return RateLimitType.TOKENS
if v3_value == "max_parallel_requests":
return RateLimitType.CONCURRENT_REQUESTS
if v3_value == "requests":
return RateLimitType.REQUESTS
return None
def _coerce_message(detail: Any) -> str:
@ -116,6 +138,7 @@ class ProxyRateLimitError(HTTPException, RateLimitError): # type: ignore[misc]
category: Union[
str, RateLimitErrorCategory
] = RateLimitErrorCategory.LITELLM_RATE_LIMIT,
rate_limit_type: Optional[Union[str, RateLimitType]] = None,
model: Optional[str] = None,
llm_provider: str = "litellm_proxy",
):
@ -144,6 +167,7 @@ class ProxyRateLimitError(HTTPException, RateLimitError): # type: ignore[misc]
llm_provider=llm_provider,
model=model or "",
category=category,
rate_limit_type=rate_limit_type,
headers=stringified_headers,
detail=detail,
)

View file

@ -2705,6 +2705,15 @@ class StandardLoggingPayloadErrorInformation(TypedDict, total=False):
# Surfaced here so custom callbacks / metrics consumers can switch on
# the rate-limit source without reaching for the raw exception.
error_rate_limit_category: Optional[str]
# error_rate_limit_type:
# For 429 / rate-limit errors, the dimension that was exceeded. One of
# the string values defined by `litellm.exceptions.RateLimitType`
# (requests, tokens, concurrent_requests, budget, max_iterations).
# None for non-rate-limit exceptions and for rate-limit exceptions that
# did not classify the failure (e.g. legacy vendor 429 with no header
# hints). Lets dashboards split rate-limit failures by cause without
# parsing free-text error messages.
error_rate_limit_type: Optional[str]
class GuardrailMode(TypedDict, total=False):