mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
feat(proxy/hooks): add ProxyHTTPRateLimitError + provider resolver
Introduces a small helper layer used by every proxy-side rate-limit hook so that the 429 they raise carries a populated llm_provider / model — instead of an empty exception.llm_provider that downstream loggers (Prometheus failure metric, observability callbacks) read as 'no provider attribution'. ProxyHTTPRateLimitError inherits from both fastapi.HTTPException (so the proxy server still renders it as a 429) and litellm.exceptions.RateLimitError (so isinstance checks and PrometheusLogger._get_exception_class_name pick up llm_provider). We deliberately don't call RateLimitError.__init__ — it constructs an httpx.Response we don't need and would just add failure surface; attribute parity is what downstream consumers care about. resolve_llm_provider_for_rate_limit() wraps litellm.get_llm_provider defensively. Internal limiter hooks fire from async_pre_call_hook — well before get_llm_provider runs anywhere else in the request lifecycle — so we have to call it ourselves at raise time. If the model is missing or unparseable (alias, router-only model) we fall back to llm_provider='litellm_proxy' rather than letting a second exception leak out and break the request path. Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
6de00e24b4
commit
921cd06858
1 changed files with 89 additions and 1 deletions
|
|
@ -2,11 +2,99 @@
|
|||
Shared utility functions for rate limiter hooks.
|
||||
"""
|
||||
|
||||
from typing import Optional, Union
|
||||
from typing import Any, Optional, Tuple, Union
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.exceptions import RateLimitError
|
||||
from litellm.types.router import ModelGroupInfo
|
||||
from litellm.types.utils import PriorityReservationDict
|
||||
|
||||
PROXY_LLM_PROVIDER_FALLBACK = "litellm_proxy"
|
||||
|
||||
|
||||
def resolve_llm_provider_for_rate_limit(
|
||||
model: Optional[str],
|
||||
) -> Tuple[str, str]:
|
||||
"""
|
||||
Resolve ``(model, llm_provider)`` for a request being rejected by an
|
||||
internal proxy-side rate-limit hook.
|
||||
|
||||
These hooks fire from ``async_pre_call_hook`` — well before
|
||||
:func:`litellm.get_llm_provider` is invoked anywhere else in the request
|
||||
lifecycle — so the raised 429 would otherwise have an empty
|
||||
``llm_provider`` field, making the resulting Prometheus
|
||||
``litellm_proxy_failed_requests_metric`` show up with
|
||||
``exception_class="RateLimitError"`` and no provider attribution.
|
||||
|
||||
Wrapped defensively: if ``model`` is missing, malformed, or
|
||||
``get_llm_provider`` raises (unknown alias, router-only model, etc.) we
|
||||
fall back to ``("", "litellm_proxy")`` so we never break the request path
|
||||
by piling a second exception on top of the rate-limit one we're trying to
|
||||
raise.
|
||||
"""
|
||||
if not model:
|
||||
return "", PROXY_LLM_PROVIDER_FALLBACK
|
||||
try:
|
||||
resolved_model, custom_llm_provider, _, _ = litellm.get_llm_provider(
|
||||
model=model,
|
||||
)
|
||||
return (
|
||||
resolved_model or model,
|
||||
custom_llm_provider or PROXY_LLM_PROVIDER_FALLBACK,
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.debug(
|
||||
"rate_limiter_utils.resolve_llm_provider_for_rate_limit: "
|
||||
"could not resolve provider for model=%s, falling back to %s. err=%s",
|
||||
model,
|
||||
PROXY_LLM_PROVIDER_FALLBACK,
|
||||
str(e),
|
||||
)
|
||||
return model, PROXY_LLM_PROVIDER_FALLBACK
|
||||
|
||||
|
||||
class ProxyHTTPRateLimitError(HTTPException, RateLimitError):
|
||||
"""
|
||||
HTTPException raised by proxy-side rate-limit hooks that *also* exposes
|
||||
``model`` and ``llm_provider`` attributes.
|
||||
|
||||
Why both base classes:
|
||||
|
||||
- The proxy server's exception handler keys off ``HTTPException`` to render
|
||||
a 429 response, so we must remain an ``HTTPException``.
|
||||
- Downstream loggers (Prometheus ``async_post_call_failure_hook``,
|
||||
structured logging, observability callbacks) read ``exception.llm_provider``
|
||||
via :meth:`litellm.integrations.prometheus.PrometheusLogger._get_exception_class_name`
|
||||
and ``isinstance(exc, RateLimitError)`` for category routing. Inheriting
|
||||
from :class:`litellm.exceptions.RateLimitError` keeps that wiring intact.
|
||||
|
||||
We intentionally do not call ``RateLimitError.__init__`` (which constructs
|
||||
an httpx.Response) — it isn't needed here and just adds failure surface.
|
||||
Attribute parity is what downstream consumers rely on.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
status_code: int,
|
||||
detail: Any = None,
|
||||
headers: Optional[dict] = None,
|
||||
*,
|
||||
model: str = "",
|
||||
llm_provider: str = PROXY_LLM_PROVIDER_FALLBACK,
|
||||
) -> None:
|
||||
HTTPException.__init__(
|
||||
self, status_code=status_code, detail=detail, headers=headers
|
||||
)
|
||||
self.status_code = status_code
|
||||
self.model = model or ""
|
||||
self.llm_provider = llm_provider or PROXY_LLM_PROVIDER_FALLBACK
|
||||
# `message` is what RateLimitError.__str__ would print and what some
|
||||
# observability callbacks log. Keep it human-readable.
|
||||
self.message = detail if isinstance(detail, str) else str(detail)
|
||||
|
||||
|
||||
def convert_priority_to_percent(
|
||||
value: Union[float, PriorityReservationDict], model_info: Optional[ModelGroupInfo]
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue