mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(proxy/hooks): populate llm_provider on batch-rate-limit 429s
batch_rate_limiter._raise_rate_limit_error now takes a requested_model kwarg threaded from data['model'] in _check_and_increment_batch_counters. The batch-creation 429 is what gets raised when the input file's tokens/requests count would push the per-key TPM/RPM window over its limit. Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
e5d5515e57
commit
127d854ca4
1 changed files with 13 additions and 1 deletions
|
|
@ -31,6 +31,10 @@ from litellm.batches.batch_utils import (
|
|||
)
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.hooks.rate_limiter_utils import (
|
||||
ProxyHTTPRateLimitError,
|
||||
resolve_llm_provider_for_rate_limit,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from opentelemetry.trace import Span as _Span
|
||||
|
|
@ -104,6 +108,7 @@ class _PROXY_BatchRateLimiter(CustomLogger):
|
|||
descriptors: List["RateLimitDescriptor"],
|
||||
batch_usage: BatchFileUsage,
|
||||
limit_type: str,
|
||||
requested_model: Optional[str] = None,
|
||||
) -> None:
|
||||
"""Raise HTTPException for rate limit exceeded."""
|
||||
from datetime import datetime
|
||||
|
|
@ -148,7 +153,10 @@ class _PROXY_BatchRateLimiter(CustomLogger):
|
|||
f"Limit resets at: {reset_time_formatted}"
|
||||
)
|
||||
|
||||
raise HTTPException(
|
||||
resolved_model, llm_provider = resolve_llm_provider_for_rate_limit(
|
||||
requested_model
|
||||
)
|
||||
raise ProxyHTTPRateLimitError(
|
||||
status_code=429,
|
||||
detail=detail,
|
||||
headers={
|
||||
|
|
@ -156,6 +164,8 @@ class _PROXY_BatchRateLimiter(CustomLogger):
|
|||
"rate_limit_type": limit_type,
|
||||
"reset_at": reset_time_formatted,
|
||||
},
|
||||
model=resolved_model,
|
||||
llm_provider=llm_provider,
|
||||
)
|
||||
|
||||
async def _check_and_increment_batch_counters(
|
||||
|
|
@ -197,6 +207,7 @@ class _PROXY_BatchRateLimiter(CustomLogger):
|
|||
)
|
||||
|
||||
if rate_limit_response["overall_code"] == "OVER_LIMIT":
|
||||
requested_model = data.get("model") if data else None
|
||||
for status in rate_limit_response["statuses"]:
|
||||
if status["code"] == "OVER_LIMIT":
|
||||
self._raise_rate_limit_error(
|
||||
|
|
@ -204,6 +215,7 @@ class _PROXY_BatchRateLimiter(CustomLogger):
|
|||
descriptors,
|
||||
batch_usage,
|
||||
status["rate_limit_type"],
|
||||
requested_model=requested_model,
|
||||
)
|
||||
|
||||
async def count_input_file_usage(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue