mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
fix(proxy/hooks): populate llm_provider on dynamic-rate-limit 429s
Same plumbing change as the parallel limiters, applied to both dynamic_rate_limiter (v1) and dynamic_rate_limiter_v3: * v1: TPM-zero and RPM-zero paths in async_pre_call_hook now resolve data['model'] -> (model, llm_provider) once and pass it into both raises. * v3: All three raise sites in _check_rate_limits — the model_saturation_check enforced raise, the priority_model enforced raise, and the fail-closed unknown-descriptor branch — now attribute the 429 to the actual provider. Falls back to llm_provider='litellm_proxy' when the model can't be resolved. Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
66976f7311
commit
e5d5515e57
2 changed files with 28 additions and 8 deletions
|
|
@ -6,14 +6,16 @@ import asyncio
|
|||
import os
|
||||
from typing import List, Optional, Tuple, Union
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
import litellm
|
||||
from litellm import ModelResponse, Router
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.hooks.rate_limiter_utils import (
|
||||
ProxyHTTPRateLimitError,
|
||||
resolve_llm_provider_for_rate_limit,
|
||||
)
|
||||
from litellm.types.router import ModelGroupInfo
|
||||
from litellm.types.utils import CallTypesLiteral
|
||||
from litellm.utils import get_utc_datetime
|
||||
|
|
@ -216,9 +218,12 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
|
|||
) = await self.check_available_usage(
|
||||
model=data["model"], priority=key_priority
|
||||
)
|
||||
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
|
||||
data.get("model")
|
||||
)
|
||||
### CHECK TPM ###
|
||||
if available_tpm is not None and available_tpm == 0:
|
||||
raise HTTPException(
|
||||
raise ProxyHTTPRateLimitError(
|
||||
status_code=429,
|
||||
detail={
|
||||
"error": "Key={} over available TPM={}. Model TPM={}, Active keys={}".format(
|
||||
|
|
@ -228,10 +233,12 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
|
|||
active_projects,
|
||||
)
|
||||
},
|
||||
model=resolved_model,
|
||||
llm_provider=llm_provider,
|
||||
)
|
||||
### CHECK RPM ###
|
||||
elif available_rpm is not None and available_rpm == 0:
|
||||
raise HTTPException(
|
||||
raise ProxyHTTPRateLimitError(
|
||||
status_code=429,
|
||||
detail={
|
||||
"error": "Key={} over available RPM={}. Model RPM={}, Active keys={}".format(
|
||||
|
|
@ -241,6 +248,8 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
|
|||
active_projects,
|
||||
)
|
||||
},
|
||||
model=resolved_model,
|
||||
llm_provider=llm_provider,
|
||||
)
|
||||
elif available_rpm is not None or available_tpm is not None:
|
||||
## UPDATE CACHE WITH ACTIVE PROJECT
|
||||
|
|
|
|||
|
|
@ -19,7 +19,11 @@ from litellm.proxy.hooks.parallel_request_limiter_v3 import (
|
|||
RateLimitDescriptorRateLimitObject,
|
||||
_PROXY_MaxParallelRequestsHandler_v3,
|
||||
)
|
||||
from litellm.proxy.hooks.rate_limiter_utils import convert_priority_to_percent
|
||||
from litellm.proxy.hooks.rate_limiter_utils import (
|
||||
ProxyHTTPRateLimitError,
|
||||
convert_priority_to_percent,
|
||||
resolve_llm_provider_for_rate_limit,
|
||||
)
|
||||
from litellm.proxy.utils import InternalUsageCache
|
||||
from litellm.types.router import ModelGroupInfo
|
||||
from litellm.types.utils import CallTypesLiteral
|
||||
|
|
@ -487,12 +491,13 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
)
|
||||
|
||||
if atomic_response["overall_code"] == "OVER_LIMIT":
|
||||
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(model)
|
||||
for status in atomic_response["statuses"]:
|
||||
if status["code"] != "OVER_LIMIT":
|
||||
continue
|
||||
descriptor_key = status["descriptor_key"]
|
||||
if descriptor_key == "model_saturation_check":
|
||||
raise HTTPException(
|
||||
raise ProxyHTTPRateLimitError(
|
||||
status_code=429,
|
||||
detail={
|
||||
"error": f"Model capacity reached for {model}. "
|
||||
|
|
@ -507,13 +512,15 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
"rate_limit_type": str(status["rate_limit_type"]),
|
||||
"x-litellm-priority": priority or "default",
|
||||
},
|
||||
model=resolved_model,
|
||||
llm_provider=llm_provider,
|
||||
)
|
||||
if descriptor_key == "priority_model":
|
||||
verbose_proxy_logger.debug(
|
||||
f"Enforcing priority limits for {model}, saturation: {saturation:.1%}, "
|
||||
f"priority: {priority}"
|
||||
)
|
||||
raise HTTPException(
|
||||
raise ProxyHTTPRateLimitError(
|
||||
status_code=429,
|
||||
detail={
|
||||
"error": f"Priority-based rate limit exceeded. "
|
||||
|
|
@ -531,6 +538,8 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
"x-litellm-priority": priority or "default",
|
||||
"x-litellm-saturation": f"{saturation:.2%}",
|
||||
},
|
||||
model=resolved_model,
|
||||
llm_provider=llm_provider,
|
||||
)
|
||||
|
||||
# Fail-closed guard: overall_code says OVER_LIMIT but no status
|
||||
|
|
@ -547,7 +556,7 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
f"Dynamic rate limiter: OVER_LIMIT response with unknown "
|
||||
f"descriptor_key(s) — refusing request. response={atomic_response}"
|
||||
)
|
||||
raise HTTPException(
|
||||
raise ProxyHTTPRateLimitError(
|
||||
status_code=429,
|
||||
detail={
|
||||
"error": "Rate limit exceeded",
|
||||
|
|
@ -562,6 +571,8 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
"retry-after": str(self.v3_limiter.window_size),
|
||||
"x-litellm-priority": priority or "default",
|
||||
},
|
||||
model=resolved_model,
|
||||
llm_provider=llm_provider,
|
||||
)
|
||||
|
||||
# If priority is NOT enforced (saturation below threshold) but
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue