mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
feat(rate-limit): add orthogonal RateLimitType (requests/tokens/concurrent_requests/budget/max_iterations)
trho's last ask in the LIT-2968 thread: distinguish rate-limit failures by
the dimension that was exceeded, not just by who rate-limited (vendor vs.
litellm). Adds:
- RateLimitType str-enum exposed at `litellm.RateLimitType` with values
requests / tokens / concurrent_requests / budget / max_iterations.
- `rate_limit_type` kwarg on litellm.RateLimitError + ProxyRateLimitError;
None default so existing callers (vendor-429 path in exception_mapping_utils)
remain a no-op.
- StandardLoggingPayloadErrorInformation.error_rate_limit_type so custom
callbacks can split rate-limit failures by cause without parsing free-text
error messages. Mirror to error_rate_limit_category extraction in
get_error_information(); single isinstance(RateLimitError) check covers both.
- map_v3_rate_limit_type() helper to collapse the v3 limiter's internal labels
("requests", "tokens", "max_parallel_requests") onto the public enum so
the v3 limiter and dynamic_rate_limiter_v3 share one mapping. Defensive
None on unknown values rather than silently picking a wrong dimension.
Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
bcf1989aef
commit
9778f94ec2
5 changed files with 90 additions and 5 deletions
|
|
@ -1256,6 +1256,7 @@ from .exceptions import (
|
|||
PermissionDeniedError,
|
||||
RateLimitError,
|
||||
RateLimitErrorCategory,
|
||||
RateLimitType,
|
||||
ServiceUnavailableError,
|
||||
BadGatewayError,
|
||||
OpenAIError,
|
||||
|
|
|
|||
|
|
@ -48,6 +48,39 @@ class RateLimitErrorCategory(str, enum.Enum):
|
|||
"""LiteLLM's own batch rate limiter (token/request budget across a batch input file) blocked the request."""
|
||||
|
||||
|
||||
class RateLimitType(str, enum.Enum):
|
||||
"""
|
||||
The dimension that was exceeded when a rate-limit error fired.
|
||||
|
||||
This is orthogonal to :class:`RateLimitErrorCategory` — *category* tells
|
||||
callers **who** rate-limited the request (the upstream vendor vs. one of
|
||||
litellm's own limiters), while *type* tells them **which limit dimension**
|
||||
was exceeded (an RPM ceiling, a TPM ceiling, a max-parallel-requests
|
||||
ceiling, a budget cap, or a max-iterations cap).
|
||||
|
||||
Surfaced both on every :class:`RateLimitError` instance via the
|
||||
``rate_limit_type`` attribute and on the structured
|
||||
``StandardLoggingPayload.error_information.error_rate_limit_type`` field
|
||||
so custom callbacks / metrics consumers can split rate-limit failures by
|
||||
cause without parsing free-text error messages.
|
||||
"""
|
||||
|
||||
REQUESTS = "requests"
|
||||
"""Requests-per-minute (RPM) or requests-per-window ceiling exceeded."""
|
||||
|
||||
TOKENS = "tokens"
|
||||
"""Tokens-per-minute (TPM) or tokens-per-window ceiling exceeded."""
|
||||
|
||||
CONCURRENT_REQUESTS = "concurrent_requests"
|
||||
"""``max_parallel_requests`` — too many in-flight requests at once."""
|
||||
|
||||
BUDGET = "budget"
|
||||
"""Spend budget cap reached (key, team, user, or per-session)."""
|
||||
|
||||
MAX_ITERATIONS = "max_iterations"
|
||||
"""Per-session max-iterations cap reached (agent-style flows)."""
|
||||
|
||||
|
||||
_MINIMAL_ERROR_RESPONSE: Optional[httpx.Response] = None
|
||||
|
||||
|
||||
|
|
@ -377,6 +410,7 @@ class RateLimitError(openai.RateLimitError): # type: ignore
|
|||
category: Union[str, RateLimitErrorCategory] = (
|
||||
RateLimitErrorCategory.VENDOR_RATE_LIMIT
|
||||
),
|
||||
rate_limit_type: Optional[Union[str, RateLimitType]] = None,
|
||||
headers: Optional[Dict[str, str]] = None,
|
||||
detail: Any = None,
|
||||
):
|
||||
|
|
@ -390,6 +424,14 @@ class RateLimitError(openai.RateLimitError): # type: ignore
|
|||
self.category = (
|
||||
category.value if isinstance(category, RateLimitErrorCategory) else category
|
||||
)
|
||||
# Which dimension was exceeded — request count, token count, parallel
|
||||
# requests, budget, max iterations. None when the source didn't
|
||||
# classify the failure (e.g. legacy vendor 429 with no header hints).
|
||||
self.rate_limit_type: Optional[str] = (
|
||||
rate_limit_type.value
|
||||
if isinstance(rate_limit_type, RateLimitType)
|
||||
else rate_limit_type
|
||||
)
|
||||
# Headers explicitly attached to the error (e.g. retry-after,
|
||||
# rate_limit_type, reset_at). Preserved across the proxy boundary so
|
||||
# clients can react appropriately.
|
||||
|
|
|
|||
|
|
@ -5160,12 +5160,20 @@ class StandardLoggingPayloadSetup:
|
|||
error_message = str(original_exception)
|
||||
|
||||
# For rate-limit errors (litellm.RateLimitError + the proxy-side
|
||||
# ProxyRateLimitError subclass), surface the unified `category` field
|
||||
# so callbacks can distinguish vendor vs. litellm rate limits without
|
||||
# reaching for the raw exception object.
|
||||
# ProxyRateLimitError subclass), surface the unified `category` and
|
||||
# `rate_limit_type` fields so callbacks can distinguish vendor vs.
|
||||
# litellm rate limits AND split by the dimension that was exceeded
|
||||
# (requests / tokens / concurrent_requests / budget / max_iterations)
|
||||
# without reaching for the raw exception object.
|
||||
is_rate_limit_error = isinstance(original_exception, litellm.RateLimitError)
|
||||
rate_limit_category: Optional[str] = (
|
||||
getattr(original_exception, "category", None)
|
||||
if isinstance(original_exception, litellm.RateLimitError)
|
||||
if is_rate_limit_error
|
||||
else None
|
||||
)
|
||||
rate_limit_type: Optional[str] = (
|
||||
getattr(original_exception, "rate_limit_type", None)
|
||||
if is_rate_limit_error
|
||||
else None
|
||||
)
|
||||
|
||||
|
|
@ -5176,6 +5184,7 @@ class StandardLoggingPayloadSetup:
|
|||
traceback=traceback_info,
|
||||
error_message=error_message if original_exception else "",
|
||||
error_rate_limit_category=rate_limit_category,
|
||||
error_rate_limit_type=rate_limit_type,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -40,7 +40,29 @@ from typing import Any, Dict, Mapping, Optional, Union
|
|||
|
||||
from fastapi import HTTPException
|
||||
|
||||
from litellm.exceptions import RateLimitError, RateLimitErrorCategory
|
||||
from litellm.exceptions import RateLimitError, RateLimitErrorCategory, RateLimitType
|
||||
|
||||
|
||||
def map_v3_rate_limit_type(
|
||||
v3_value: Optional[str],
|
||||
) -> Optional[RateLimitType]:
|
||||
"""
|
||||
Map the v3 rate limiter's internal `status["rate_limit_type"]` strings
|
||||
onto the public :class:`RateLimitType` enum.
|
||||
|
||||
The v3 limiter uses the literal values ``"requests"``, ``"tokens"``, and
|
||||
``"max_parallel_requests"``. We collapse the last one onto
|
||||
:attr:`RateLimitType.CONCURRENT_REQUESTS` because that's the public name
|
||||
documented for users and dashboards. Unrecognized values return ``None``
|
||||
so the field stays absent rather than carrying garbage downstream.
|
||||
"""
|
||||
if v3_value == "tokens":
|
||||
return RateLimitType.TOKENS
|
||||
if v3_value == "max_parallel_requests":
|
||||
return RateLimitType.CONCURRENT_REQUESTS
|
||||
if v3_value == "requests":
|
||||
return RateLimitType.REQUESTS
|
||||
return None
|
||||
|
||||
|
||||
def _coerce_message(detail: Any) -> str:
|
||||
|
|
@ -116,6 +138,7 @@ class ProxyRateLimitError(HTTPException, RateLimitError): # type: ignore[misc]
|
|||
category: Union[
|
||||
str, RateLimitErrorCategory
|
||||
] = RateLimitErrorCategory.LITELLM_RATE_LIMIT,
|
||||
rate_limit_type: Optional[Union[str, RateLimitType]] = None,
|
||||
model: Optional[str] = None,
|
||||
llm_provider: str = "litellm_proxy",
|
||||
):
|
||||
|
|
@ -144,6 +167,7 @@ class ProxyRateLimitError(HTTPException, RateLimitError): # type: ignore[misc]
|
|||
llm_provider=llm_provider,
|
||||
model=model or "",
|
||||
category=category,
|
||||
rate_limit_type=rate_limit_type,
|
||||
headers=stringified_headers,
|
||||
detail=detail,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -2705,6 +2705,15 @@ class StandardLoggingPayloadErrorInformation(TypedDict, total=False):
|
|||
# Surfaced here so custom callbacks / metrics consumers can switch on
|
||||
# the rate-limit source without reaching for the raw exception.
|
||||
error_rate_limit_category: Optional[str]
|
||||
# error_rate_limit_type:
|
||||
# For 429 / rate-limit errors, the dimension that was exceeded. One of
|
||||
# the string values defined by `litellm.exceptions.RateLimitType`
|
||||
# (requests, tokens, concurrent_requests, budget, max_iterations).
|
||||
# None for non-rate-limit exceptions and for rate-limit exceptions that
|
||||
# did not classify the failure (e.g. legacy vendor 429 with no header
|
||||
# hints). Lets dashboards split rate-limit failures by cause without
|
||||
# parsing free-text error messages.
|
||||
error_rate_limit_type: Optional[str]
|
||||
|
||||
|
||||
class GuardrailMode(TypedDict, total=False):
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue