mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
refactor(proxy/hooks): raise ProxyRateLimitError from dynamic rate limiters
Replaces HTTPException(status_code=429, ...) call sites in the v1 and v3 dynamic rate limiters (project-level TPM/RPM allocation, model-saturation checks, priority-based limits, fail-closed guards) with ProxyRateLimitError. The v3 limiter still imports HTTPException for an unrelated bare 'except HTTPException:' branch. LIT-2968 Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
96af71264a
commit
be5b496ad8
2 changed files with 7 additions and 12 deletions
|
|
@ -6,14 +6,13 @@ import asyncio
|
|||
import os
|
||||
from typing import List, Optional, Tuple, Union
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
import litellm
|
||||
from litellm import ModelResponse, Router
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.caching.caching import DualCache
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
from litellm.types.router import ModelGroupInfo
|
||||
from litellm.types.utils import CallTypesLiteral
|
||||
from litellm.utils import get_utc_datetime
|
||||
|
|
@ -218,8 +217,7 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
|
|||
)
|
||||
### CHECK TPM ###
|
||||
if available_tpm is not None and available_tpm == 0:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
raise ProxyRateLimitError(
|
||||
detail={
|
||||
"error": "Key={} over available TPM={}. Model TPM={}, Active keys={}".format(
|
||||
user_api_key_dict.api_key,
|
||||
|
|
@ -231,8 +229,7 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
|
|||
)
|
||||
### CHECK RPM ###
|
||||
elif available_rpm is not None and available_rpm == 0:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
raise ProxyRateLimitError(
|
||||
detail={
|
||||
"error": "Key={} over available RPM={}. Model RPM={}, Active keys={}".format(
|
||||
user_api_key_dict.api_key,
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ from litellm._logging import verbose_proxy_logger
|
|||
from litellm.caching.caching import DualCache
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
from litellm.proxy.hooks.parallel_request_limiter_v3 import (
|
||||
RateLimitDescriptor,
|
||||
RateLimitDescriptorRateLimitObject,
|
||||
|
|
@ -492,8 +493,7 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
continue
|
||||
descriptor_key = status["descriptor_key"]
|
||||
if descriptor_key == "model_saturation_check":
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
raise ProxyRateLimitError(
|
||||
detail={
|
||||
"error": f"Model capacity reached for {model}. "
|
||||
f"Priority: {priority}, "
|
||||
|
|
@ -513,8 +513,7 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
f"Enforcing priority limits for {model}, saturation: {saturation:.1%}, "
|
||||
f"priority: {priority}"
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
raise ProxyRateLimitError(
|
||||
detail={
|
||||
"error": f"Priority-based rate limit exceeded. "
|
||||
f"Model: {model}, "
|
||||
|
|
@ -547,8 +546,7 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
f"Dynamic rate limiter: OVER_LIMIT response with unknown "
|
||||
f"descriptor_key(s) — refusing request. response={atomic_response}"
|
||||
)
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
raise ProxyRateLimitError(
|
||||
detail={
|
||||
"error": "Rate limit exceeded",
|
||||
"descriptor_key": (
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue