mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
feat(proxy): add ProxyRateLimitError unifying RateLimitError + HTTPException
Adds a single proxy-side error class that subclasses BOTH litellm.exceptions.RateLimitError AND fastapi.HTTPException via cooperative multiple inheritance. Why both bases: * Subclassing RateLimitError lets user code catch every rate-limit source with one 'except RateLimitError' and switch on the new .category field. * Subclassing HTTPException keeps every existing FastAPI plumbing path (the isinstance(e, HTTPException) branches in proxy_server.py route handlers, FastAPI's own dispatcher, and tests asserting pytest.raises(HTTPException)) working without modification, and preserves retry-after / rate_limit_type / reset_at headers on the wire. The class declaration order is (HTTPException, RateLimitError) so the MRO puts HTTPException's no-super-call __init__ ahead of openai's cooperative __init__ chain — preventing openai.APIError.super().__init__(message) from landing in HTTPException.__init__(status_code=message). LIT-2968 Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
150616c1a1
commit
8360519f2b
1 changed files with 144 additions and 0 deletions
144
litellm/proxy/common_utils/proxy_rate_limit_error.py
Normal file
144
litellm/proxy/common_utils/proxy_rate_limit_error.py
Normal file
|
|
@ -0,0 +1,144 @@
|
|||
"""
|
||||
ProxyRateLimitError — a unified rate-limit exception used by litellm's
|
||||
proxy-side hooks.
|
||||
|
||||
Background
|
||||
----------
|
||||
LiteLLM previously surfaced rate-limit conditions through *several* unrelated
|
||||
exception types:
|
||||
|
||||
* :class:`litellm.exceptions.RateLimitError` — raised by exception mapping when
|
||||
an upstream LLM provider returns 429.
|
||||
* :class:`fastapi.HTTPException` (status 429) — raised directly by proxy hooks
|
||||
such as ``parallel_request_limiter``, ``dynamic_rate_limiter``,
|
||||
``batch_rate_limiter``, ``max_budget_limiter``, ``max_iterations_limiter``,
|
||||
etc.
|
||||
* :class:`litellm.llms.base_llm.chat.transformation.BaseLLMException` (status
|
||||
429) — raised by some provider transports.
|
||||
|
||||
This made it impossible for downstream code (and end users) to express
|
||||
"is this a rate limit?" with a single ``except`` clause, and impossible to
|
||||
distinguish *where* the rate limit originated (vendor vs. litellm, batch vs.
|
||||
chat) without ad-hoc string-matching on the message.
|
||||
|
||||
This module provides a single proxy-side error class that:
|
||||
|
||||
1. Is a subclass of :class:`litellm.exceptions.RateLimitError`, so user code
|
||||
that catches ``RateLimitError`` works for *every* rate-limit source.
|
||||
2. Is also a subclass of :class:`fastapi.HTTPException`, so existing proxy
|
||||
plumbing (``isinstance(e, HTTPException)`` branches in route handlers and
|
||||
FastAPI's own dispatcher) continues to behave the same way and the
|
||||
``retry-after`` / ``rate_limit_type`` / ``reset_at`` headers are preserved
|
||||
on the wire.
|
||||
3. Carries a :attr:`category` field (one of
|
||||
:class:`litellm.exceptions.RateLimitErrorCategory`) so callers can switch on
|
||||
the rate limit source.
|
||||
"""
|
||||
|
||||
import json
|
||||
from typing import Any, Dict, Mapping, Optional, Union
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
from litellm.exceptions import RateLimitError, RateLimitErrorCategory
|
||||
|
||||
|
||||
def _coerce_message(detail: Any) -> str:
|
||||
"""Best-effort, JSON-friendly stringification of an HTTPException-style detail."""
|
||||
if isinstance(detail, str):
|
||||
return detail
|
||||
if isinstance(detail, Mapping):
|
||||
for key in ("error", "message"):
|
||||
if isinstance(detail.get(key), str):
|
||||
return detail[key]
|
||||
inner = detail.get(key)
|
||||
if isinstance(inner, Mapping) and isinstance(inner.get("message"), str):
|
||||
return inner["message"]
|
||||
try:
|
||||
return json.dumps(detail)
|
||||
except (TypeError, ValueError):
|
||||
return str(detail)
|
||||
return str(detail)
|
||||
|
||||
|
||||
class ProxyRateLimitError(HTTPException, RateLimitError):
|
||||
"""
|
||||
A 429 raised by litellm's proxy-side rate limiting hooks.
|
||||
|
||||
This class deliberately inherits from BOTH
|
||||
:class:`litellm.exceptions.RateLimitError` and :class:`fastapi.HTTPException`
|
||||
so the same instance can flow through:
|
||||
|
||||
* ``except RateLimitError`` (user / SDK code that wants a category-aware
|
||||
handler), and
|
||||
* ``isinstance(e, HTTPException)`` (FastAPI / proxy_server.py route
|
||||
handlers that need to forward ``status_code``, ``detail`` and
|
||||
``headers`` back to the client).
|
||||
|
||||
Downstream code should prefer this class over
|
||||
``raise HTTPException(status_code=429, ...)`` for litellm-internal rate
|
||||
limits.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
detail:
|
||||
The structured error payload. Forwarded as ``HTTPException.detail`` so
|
||||
FastAPI's default exception handler will serialize it verbatim.
|
||||
headers:
|
||||
Optional response headers (e.g. ``retry-after``). Values are stringified
|
||||
to satisfy FastAPI's typing.
|
||||
category:
|
||||
One of :class:`RateLimitErrorCategory`. Defaults to
|
||||
``LITELLM_RATE_LIMIT`` since this class is only used by litellm's own
|
||||
proxy-side limiters; pass ``LITELLM_BATCH_RATE_LIMIT`` for the batch
|
||||
limiter, etc.
|
||||
model / llm_provider:
|
||||
Optional context, propagated to the inherited ``RateLimitError`` for
|
||||
compatibility with logging / standard payload extraction.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
detail: Any,
|
||||
headers: Optional[Mapping[str, Any]] = None,
|
||||
category: Union[
|
||||
str, RateLimitErrorCategory
|
||||
] = RateLimitErrorCategory.LITELLM_RATE_LIMIT,
|
||||
model: Optional[str] = None,
|
||||
llm_provider: str = "litellm_proxy",
|
||||
):
|
||||
message = _coerce_message(detail)
|
||||
stringified_headers: Optional[Dict[str, str]] = (
|
||||
{k: str(v) for k, v in headers.items()} if headers else None
|
||||
)
|
||||
|
||||
# Initialize the FastAPI HTTPException portion first so its attributes
|
||||
# (status_code, detail, headers) are already on the instance before
|
||||
# RateLimitError.__init__ runs and possibly overrides them.
|
||||
HTTPException.__init__(
|
||||
self,
|
||||
status_code=429,
|
||||
detail=detail,
|
||||
headers=stringified_headers,
|
||||
)
|
||||
|
||||
# Now initialize the litellm RateLimitError portion. We deliberately
|
||||
# pass the structured detail through so RateLimitError preserves it as
|
||||
# its `.detail` attribute too — keeping both sides of the MRO
|
||||
# consistent.
|
||||
RateLimitError.__init__(
|
||||
self,
|
||||
message=message,
|
||||
llm_provider=llm_provider,
|
||||
model=model or "",
|
||||
category=category,
|
||||
headers=stringified_headers,
|
||||
detail=detail,
|
||||
)
|
||||
# RateLimitError.__init__ overwrites self.headers with its own copy and
|
||||
# leaves self.status_code at 429 — restore the HTTPException-style
|
||||
# headers value so downstream code that pulls headers off the
|
||||
# instance gets back exactly what the limiter passed in.
|
||||
self.headers = stringified_headers
|
||||
self.detail = detail
|
||||
self.status_code = 429
|
||||
Loading…
Add table
Reference in a new issue