feat(proxy/hooks): add ProxyHTTPRateLimitError + provider resolver

Introduces a small helper layer used by every proxy-side rate-limit
hook so that the 429 they raise carries a populated llm_provider /
model — instead of an empty exception.llm_provider that downstream
loggers (Prometheus failure metric, observability callbacks) read as
'no provider attribution'.

ProxyHTTPRateLimitError inherits from both fastapi.HTTPException
(so the proxy server still renders it as a 429) and
litellm.exceptions.RateLimitError (so isinstance checks and
PrometheusLogger._get_exception_class_name pick up llm_provider).
We deliberately don't call RateLimitError.__init__ — it constructs
an httpx.Response we don't need and would just add failure surface;
attribute parity is what downstream consumers care about.

resolve_llm_provider_for_rate_limit() wraps litellm.get_llm_provider
defensively. Internal limiter hooks fire from async_pre_call_hook —
well before get_llm_provider runs anywhere else in the request
lifecycle — so we have to call it ourselves at raise time. If the
model is missing or unparseable (alias, router-only model) we fall
back to llm_provider='litellm_proxy' rather than letting a second
exception leak out and break the request path.

Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-05-12 02:58:27 +00:00
parent 6de00e24b4
commit 921cd06858
No known key found for this signature in database

View file

@ -2,11 +2,99 @@
Shared utility functions for rate limiter hooks.
"""
from typing import Optional, Union
from typing import Any, Optional, Tuple, Union
from fastapi import HTTPException
import litellm
from litellm._logging import verbose_proxy_logger
from litellm.exceptions import RateLimitError
from litellm.types.router import ModelGroupInfo
from litellm.types.utils import PriorityReservationDict
PROXY_LLM_PROVIDER_FALLBACK = "litellm_proxy"
def resolve_llm_provider_for_rate_limit(
model: Optional[str],
) -> Tuple[str, str]:
"""
Resolve ``(model, llm_provider)`` for a request being rejected by an
internal proxy-side rate-limit hook.
These hooks fire from ``async_pre_call_hook`` — well before
:func:`litellm.get_llm_provider` is invoked anywhere else in the request
lifecycle — so the raised 429 would otherwise have an empty
``llm_provider`` field, making the resulting Prometheus
``litellm_proxy_failed_requests_metric`` show up with
``exception_class="RateLimitError"`` and no provider attribution.
Wrapped defensively: if ``model`` is missing, malformed, or
``get_llm_provider`` raises (unknown alias, router-only model, etc.) we
fall back to ``("", "litellm_proxy")`` so we never break the request path
by piling a second exception on top of the rate-limit one we're trying to
raise.
"""
if not model:
return "", PROXY_LLM_PROVIDER_FALLBACK
try:
resolved_model, custom_llm_provider, _, _ = litellm.get_llm_provider(
model=model,
)
return (
resolved_model or model,
custom_llm_provider or PROXY_LLM_PROVIDER_FALLBACK,
)
except Exception as e:
verbose_proxy_logger.debug(
"rate_limiter_utils.resolve_llm_provider_for_rate_limit: "
"could not resolve provider for model=%s, falling back to %s. err=%s",
model,
PROXY_LLM_PROVIDER_FALLBACK,
str(e),
)
return model, PROXY_LLM_PROVIDER_FALLBACK
class ProxyHTTPRateLimitError(HTTPException, RateLimitError):
"""
HTTPException raised by proxy-side rate-limit hooks that *also* exposes
``model`` and ``llm_provider`` attributes.
Why both base classes:
- The proxy server's exception handler keys off ``HTTPException`` to render
a 429 response, so we must remain an ``HTTPException``.
- Downstream loggers (Prometheus ``async_post_call_failure_hook``,
structured logging, observability callbacks) read ``exception.llm_provider``
via :meth:`litellm.integrations.prometheus.PrometheusLogger._get_exception_class_name`
and ``isinstance(exc, RateLimitError)`` for category routing. Inheriting
from :class:`litellm.exceptions.RateLimitError` keeps that wiring intact.
We intentionally do not call ``RateLimitError.__init__`` (which constructs
an httpx.Response) — it isn't needed here and just adds failure surface.
Attribute parity is what downstream consumers rely on.
"""
def __init__(
self,
status_code: int,
detail: Any = None,
headers: Optional[dict] = None,
*,
model: str = "",
llm_provider: str = PROXY_LLM_PROVIDER_FALLBACK,
) -> None:
HTTPException.__init__(
self, status_code=status_code, detail=detail, headers=headers
)
self.status_code = status_code
self.model = model or ""
self.llm_provider = llm_provider or PROXY_LLM_PROVIDER_FALLBACK
# `message` is what RateLimitError.__str__ would print and what some
# observability callbacks log. Keep it human-readable.
self.message = detail if isinstance(detail, str) else str(detail)
def convert_priority_to_percent(
value: Union[float, PriorityReservationDict], model_info: Optional[ModelGroupInfo]