fix(proxy/hooks): populate llm_provider on dynamic-rate-limit 429s

Same plumbing change as the parallel limiters, applied to both
dynamic_rate_limiter (v1) and dynamic_rate_limiter_v3:

* v1: TPM-zero and RPM-zero paths in async_pre_call_hook now resolve
  data['model'] -> (model, llm_provider) once and pass it into both
  raises.
* v3: All three raise sites in _check_rate_limits — the
  model_saturation_check enforced raise, the priority_model
  enforced raise, and the fail-closed unknown-descriptor branch —
  now attribute the 429 to the actual provider.

Falls back to llm_provider='litellm_proxy' when the model can't be
resolved.

Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-05-12 02:58:51 +00:00
parent 66976f7311
commit e5d5515e57
No known key found for this signature in database
2 changed files with 28 additions and 8 deletions

View file

@ -6,14 +6,16 @@ import asyncio
import os
from typing import List, Optional, Tuple, Union
from fastapi import HTTPException
import litellm
from litellm import ModelResponse, Router
from litellm._logging import verbose_proxy_logger
from litellm.caching.caching import DualCache
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.hooks.rate_limiter_utils import (
ProxyHTTPRateLimitError,
resolve_llm_provider_for_rate_limit,
)
from litellm.types.router import ModelGroupInfo
from litellm.types.utils import CallTypesLiteral
from litellm.utils import get_utc_datetime
@ -216,9 +218,12 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
) = await self.check_available_usage(
model=data["model"], priority=key_priority
)
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
data.get("model")
)
### CHECK TPM ###
if available_tpm is not None and available_tpm == 0:
raise HTTPException(
raise ProxyHTTPRateLimitError(
status_code=429,
detail={
"error": "Key={} over available TPM={}. Model TPM={}, Active keys={}".format(
@ -228,10 +233,12 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
active_projects,
)
},
model=resolved_model,
llm_provider=llm_provider,
)
### CHECK RPM ###
elif available_rpm is not None and available_rpm == 0:
raise HTTPException(
raise ProxyHTTPRateLimitError(
status_code=429,
detail={
"error": "Key={} over available RPM={}. Model RPM={}, Active keys={}".format(
@ -241,6 +248,8 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
active_projects,
)
},
model=resolved_model,
llm_provider=llm_provider,
)
elif available_rpm is not None or available_tpm is not None:
## UPDATE CACHE WITH ACTIVE PROJECT

View file

@ -19,7 +19,11 @@ from litellm.proxy.hooks.parallel_request_limiter_v3 import (
RateLimitDescriptorRateLimitObject,
_PROXY_MaxParallelRequestsHandler_v3,
)
from litellm.proxy.hooks.rate_limiter_utils import convert_priority_to_percent
from litellm.proxy.hooks.rate_limiter_utils import (
ProxyHTTPRateLimitError,
convert_priority_to_percent,
resolve_llm_provider_for_rate_limit,
)
from litellm.proxy.utils import InternalUsageCache
from litellm.types.router import ModelGroupInfo
from litellm.types.utils import CallTypesLiteral
@ -487,12 +491,13 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
)
if atomic_response["overall_code"] == "OVER_LIMIT":
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(model)
for status in atomic_response["statuses"]:
if status["code"] != "OVER_LIMIT":
continue
descriptor_key = status["descriptor_key"]
if descriptor_key == "model_saturation_check":
raise HTTPException(
raise ProxyHTTPRateLimitError(
status_code=429,
detail={
"error": f"Model capacity reached for {model}. "
@ -507,13 +512,15 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
"rate_limit_type": str(status["rate_limit_type"]),
"x-litellm-priority": priority or "default",
},
model=resolved_model,
llm_provider=llm_provider,
)
if descriptor_key == "priority_model":
verbose_proxy_logger.debug(
f"Enforcing priority limits for {model}, saturation: {saturation:.1%}, "
f"priority: {priority}"
)
raise HTTPException(
raise ProxyHTTPRateLimitError(
status_code=429,
detail={
"error": f"Priority-based rate limit exceeded. "
@ -531,6 +538,8 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
"x-litellm-priority": priority or "default",
"x-litellm-saturation": f"{saturation:.2%}",
},
model=resolved_model,
llm_provider=llm_provider,
)
# Fail-closed guard: overall_code says OVER_LIMIT but no status
@ -547,7 +556,7 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
f"Dynamic rate limiter: OVER_LIMIT response with unknown "
f"descriptor_key(s) — refusing request. response={atomic_response}"
)
raise HTTPException(
raise ProxyHTTPRateLimitError(
status_code=429,
detail={
"error": "Rate limit exceeded",
@ -562,6 +571,8 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
"retry-after": str(self.v3_limiter.window_size),
"x-litellm-priority": priority or "default",
},
model=resolved_model,
llm_provider=llm_provider,
)
# If priority is NOT enforced (saturation below threshold) but