fix(proxy/hooks): populate llm_provider on parallel-request 429s

Both v1 and v3 parallel-request limiters fired bare HTTPException(429)
from inside async_pre_call_hook. The downstream Prometheus failure
metric reads exception.llm_provider via _get_exception_class_name —
the empty value showed up as exception_class='HTTPException' and
left model_id='None' on the time series.

Threads requested_model through every raise site in:

* parallel_request_limiter.py:
  - check_key_in_limits (the per-key/per-model/per-user/per-team/
    per-customer over-limit path)
  - raise_rate_limit_error (zero-limit + global_max_parallel_requests
    paths) — now takes an optional requested_model kwarg
* parallel_request_limiter_v3.py:
  - _handle_rate_limit_error (the OVER_LIMIT translator), called
    from both the should_rate_limit pre-check and the TPM
    reservation path

Resolved via resolve_llm_provider_for_rate_limit so unknown / missing
models silently fall back to llm_provider='litellm_proxy' instead of
breaking the request path with a second exception.

Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-05-12 02:58:41 +00:00
parent 921cd06858
commit 66976f7311
No known key found for this signature in database
2 changed files with 44 additions and 9 deletions

View file

@ -17,6 +17,10 @@ from litellm.proxy.auth.auth_utils import (
get_key_model_rpm_limit,
get_key_model_tpm_limit,
)
from litellm.proxy.hooks.rate_limiter_utils import (
ProxyHTTPRateLimitError,
resolve_llm_provider_for_rate_limit,
)
if TYPE_CHECKING:
from opentelemetry.trace import Span as _Span
@ -73,7 +77,8 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
if max_parallel_requests == 0 or tpm_limit == 0 or rpm_limit == 0:
# base case
raise self.raise_rate_limit_error(
additional_details=f"{CommonProxyErrors.max_parallel_request_limit_reached.value}. Hit limit for {rate_limit_type}. Current limits: max_parallel_requests: {max_parallel_requests}, tpm_limit: {tpm_limit}, rpm_limit: {rpm_limit}"
additional_details=f"{CommonProxyErrors.max_parallel_request_limit_reached.value}. Hit limit for {rate_limit_type}. Current limits: max_parallel_requests: {max_parallel_requests}, tpm_limit: {tpm_limit}, rpm_limit: {rpm_limit}",
requested_model=data.get("model") if data else None,
)
new_val = {
"current_requests": 1,
@ -95,10 +100,16 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
values_to_update_in_cache.append((request_count_api_key, new_val))
else:
raise HTTPException(
requested_model = data.get("model") if data else None
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
requested_model
)
raise ProxyHTTPRateLimitError(
status_code=429,
detail=f"LiteLLM Rate Limit Handler for rate limit type = {rate_limit_type}. {CommonProxyErrors.max_parallel_request_limit_reached.value}. current rpm: {current['current_rpm']}, rpm limit: {rpm_limit}, current tpm: {current['current_tpm']}, tpm limit: {tpm_limit}, current max_parallel_requests: {current['current_requests']}, max_parallel_requests: {max_parallel_requests}",
headers={"retry-after": str(self.time_to_next_minute())},
model=resolved_model,
llm_provider=llm_provider,
)
await self.internal_usage_cache.async_batch_set_cache(
@ -122,18 +133,31 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
return seconds_to_next_minute
def raise_rate_limit_error(
self, additional_details: Optional[str] = None
self,
additional_details: Optional[str] = None,
requested_model: Optional[str] = None,
) -> HTTPException:
"""
Raise an HTTPException with a 429 status code and a retry-after header
Raise an HTTPException with a 429 status code and a retry-after header.
``requested_model`` is resolved via :func:`get_llm_provider` so the
raised exception carries ``llm_provider`` for downstream loggers
(Prometheus failure metric, observability callbacks). Falls back to
``llm_provider="litellm_proxy"`` when the model is missing or
unparseable — see ``resolve_llm_provider_for_rate_limit``.
"""
error_message = "Max parallel request limit reached"
if additional_details is not None:
error_message = error_message + " " + additional_details
raise HTTPException(
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
requested_model
)
raise ProxyHTTPRateLimitError(
status_code=429,
detail=f"Max parallel request limit reached {additional_details}",
headers={"retry-after": str(self.time_to_next_minute())},
model=resolved_model,
llm_provider=llm_provider,
)
async def get_all_cache_objects(
@ -225,7 +249,8 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
# if above -> raise error
if current_global_requests >= global_max_parallel_requests:
return self.raise_rate_limit_error(
additional_details=f"Hit Global Limit: Limit={global_max_parallel_requests}, current: {current_global_requests}"
additional_details=f"Hit Global Limit: Limit={global_max_parallel_requests}, current: {current_global_requests}",
requested_model=data.get("model") if data else None,
)
# if below -> increment
else:

View file

@ -23,8 +23,6 @@ from typing import (
cast,
)
from fastapi import HTTPException
from litellm import DualCache
from litellm._logging import verbose_proxy_logger
from litellm.constants import DYNAMIC_RATE_LIMIT_ERROR_THRESHOLD_PER_MINUTE
@ -34,6 +32,10 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import (
)
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.auth.auth_utils import get_model_rate_limit_from_metadata
from litellm.proxy.hooks.rate_limiter_utils import (
ProxyHTTPRateLimitError,
resolve_llm_provider_for_rate_limit,
)
from litellm.types.caching import RedisPipelineIncrementOperation
from litellm.types.llms.openai import BaseLiteLLMOpenAIResponseObject
from litellm.types.utils import ModelResponse, Usage
@ -1837,6 +1839,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
self,
response: RateLimitResponse,
descriptors: List[RateLimitDescriptor],
requested_model: Optional[str] = None,
) -> None:
"""Handle rate limit exceeded error by raising HTTPException."""
for status in response["statuses"]:
@ -1869,7 +1872,10 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
f"Limit resets at: {reset_time_formatted}"
)
raise HTTPException(
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
requested_model
)
raise ProxyHTTPRateLimitError(
status_code=429,
detail=detail,
headers={
@ -1877,6 +1883,8 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
"rate_limit_type": str(status["rate_limit_type"]),
"reset_at": reset_time_formatted,
},
model=resolved_model,
llm_provider=llm_provider,
)
async def async_pre_call_hook(
@ -1977,6 +1985,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
self._handle_rate_limit_error(
response=response,
descriptors=descriptors,
requested_model=requested_model,
)
else:
# add descriptors to request headers
@ -2022,6 +2031,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
self._handle_rate_limit_error(
response=tpm_response,
descriptors=descriptors,
requested_model=requested_model,
)
else:
data["_litellm_rate_limit_descriptors"] = descriptors