mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
feat(proxy/hooks): wire rate_limit_type onto every limiter raise site
Each refactored proxy hook now populates rate_limit_type with the dimension that actually tripped the limit, so downstream consumers (custom callbacks, prometheus exporters via the StandardLoggingPayload) can split key/team/user rate-limit failures by cause: - parallel_request_limiter (v1): detect dimension from current vs. limit in the post-cache branch (concurrent_requests > tokens > requests, matches the boolean condition order). Base case (current is None, one limit set to 0) picks the most-specific zero. raise_rate_limit_error() helper accepts an explicit rate_limit_type kwarg with CONCURRENT_REQUESTS default (matches every existing internal call site, including the global-limit branch). - parallel_request_limiter (v3): forward status["rate_limit_type"] through map_v3_rate_limit_type() so "max_parallel_requests" → CONCURRENT_REQUESTS for the public field while the raw v3 jargon stays on the HTTP header for wire-format backward compat. - dynamic_rate_limiter (v1): TPM-zero → TOKENS, RPM-zero → REQUESTS. Pass data["model"] through so callbacks see the model that hit the limit (addresses the secondary "provider missing" complaint in the original Slack thread, partially — the model is what dashboards typically split on). - dynamic_rate_limiter (v3): forward status["rate_limit_type"] via map_v3_rate_limit_type() at every raise site (model_saturation_check, priority_model, fail-closed unknown-descriptor guard). Also pass model. - batch_rate_limiter: limit_type is hard-typed "requests"|"tokens" — map directly without going through the helper's None branch. - max_budget_limiter, max_budget_per_session_limiter: BUDGET. - max_iterations_limiter: MAX_ITERATIONS. Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
9778f94ec2
commit
48dcd101c9
8 changed files with 79 additions and 7 deletions
|
|
@ -29,7 +29,7 @@ from litellm.batches.batch_utils import (
|
|||
_get_file_content_as_dictionary,
|
||||
_get_models_from_batch_input_file_content,
|
||||
)
|
||||
from litellm.exceptions import RateLimitErrorCategory
|
||||
from litellm.exceptions import RateLimitErrorCategory, RateLimitType
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
|
|
@ -158,6 +158,14 @@ class _PROXY_BatchRateLimiter(CustomLogger):
|
|||
"reset_at": reset_time_formatted,
|
||||
},
|
||||
category=RateLimitErrorCategory.LITELLM_BATCH_RATE_LIMIT,
|
||||
# The batch limiter's `limit_type` arg is a hard-typed string —
|
||||
# always either "requests" or "tokens" — so we map it directly
|
||||
# onto the public enum without hitting the helper's None branch.
|
||||
rate_limit_type=(
|
||||
RateLimitType.TOKENS
|
||||
if limit_type == "tokens"
|
||||
else RateLimitType.REQUESTS
|
||||
),
|
||||
)
|
||||
|
||||
async def _check_and_increment_batch_counters(
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ from litellm._logging import verbose_proxy_logger
|
|||
from litellm.caching.caching import DualCache
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.exceptions import RateLimitType
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
from litellm.types.router import ModelGroupInfo
|
||||
from litellm.types.utils import CallTypesLiteral
|
||||
|
|
@ -226,6 +227,8 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
|
|||
active_projects,
|
||||
)
|
||||
},
|
||||
rate_limit_type=RateLimitType.TOKENS,
|
||||
model=data.get("model"),
|
||||
)
|
||||
### CHECK RPM ###
|
||||
elif available_rpm is not None and available_rpm == 0:
|
||||
|
|
@ -238,6 +241,8 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
|
|||
active_projects,
|
||||
)
|
||||
},
|
||||
rate_limit_type=RateLimitType.REQUESTS,
|
||||
model=data.get("model"),
|
||||
)
|
||||
elif available_rpm is not None or available_tpm is not None:
|
||||
## UPDATE CACHE WITH ACTIVE PROJECT
|
||||
|
|
|
|||
|
|
@ -14,7 +14,10 @@ from litellm._logging import verbose_proxy_logger
|
|||
from litellm.caching.caching import DualCache
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import (
|
||||
ProxyRateLimitError,
|
||||
map_v3_rate_limit_type,
|
||||
)
|
||||
from litellm.proxy.hooks.parallel_request_limiter_v3 import (
|
||||
RateLimitDescriptor,
|
||||
RateLimitDescriptorRateLimitObject,
|
||||
|
|
@ -507,6 +510,10 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
"rate_limit_type": str(status["rate_limit_type"]),
|
||||
"x-litellm-priority": priority or "default",
|
||||
},
|
||||
rate_limit_type=map_v3_rate_limit_type(
|
||||
status["rate_limit_type"]
|
||||
),
|
||||
model=model,
|
||||
)
|
||||
if descriptor_key == "priority_model":
|
||||
verbose_proxy_logger.debug(
|
||||
|
|
@ -530,6 +537,10 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
"x-litellm-priority": priority or "default",
|
||||
"x-litellm-saturation": f"{saturation:.2%}",
|
||||
},
|
||||
rate_limit_type=map_v3_rate_limit_type(
|
||||
status["rate_limit_type"]
|
||||
),
|
||||
model=model,
|
||||
)
|
||||
|
||||
# Fail-closed guard: overall_code says OVER_LIMIT but no status
|
||||
|
|
@ -556,6 +567,10 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
|
|||
str(offending["rate_limit_type"]) if offending else "unknown"
|
||||
),
|
||||
},
|
||||
rate_limit_type=map_v3_rate_limit_type(
|
||||
offending["rate_limit_type"] if offending else None
|
||||
),
|
||||
model=model,
|
||||
headers={
|
||||
"retry-after": str(self.v3_limiter.window_size),
|
||||
"x-litellm-priority": priority or "default",
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ from litellm._logging import verbose_proxy_logger
|
|||
from litellm.caching.caching import DualCache
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.exceptions import RateLimitType
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
|
||||
|
||||
|
|
@ -64,7 +65,10 @@ class _PROXY_MaxBudgetLimiter(CustomLogger):
|
|||
|
||||
# CHECK IF REQUEST ALLOWED
|
||||
if curr_spend >= max_budget:
|
||||
raise ProxyRateLimitError(detail="Max budget limit reached.")
|
||||
raise ProxyRateLimitError(
|
||||
detail="Max budget limit reached.",
|
||||
rate_limit_type=RateLimitType.BUDGET,
|
||||
)
|
||||
except HTTPException as e:
|
||||
raise e
|
||||
except Exception as e:
|
||||
|
|
|
|||
|
|
@ -21,6 +21,7 @@ from litellm import DualCache
|
|||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.exceptions import RateLimitType
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
|
||||
if TYPE_CHECKING:
|
||||
|
|
@ -117,6 +118,7 @@ class _PROXY_MaxBudgetPerSessionHandler(CustomLogger):
|
|||
f"Current spend: ${current_spend:.4f}, "
|
||||
f"max_budget_per_session: ${max_budget:.2f}."
|
||||
),
|
||||
rate_limit_type=RateLimitType.BUDGET,
|
||||
)
|
||||
|
||||
return None
|
||||
|
|
|
|||
|
|
@ -17,6 +17,7 @@ from litellm import DualCache
|
|||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.exceptions import RateLimitType
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
|
||||
if TYPE_CHECKING:
|
||||
|
|
@ -120,6 +121,7 @@ class _PROXY_MaxIterationsHandler(CustomLogger):
|
|||
f"Max iterations exceeded for session {session_id}. "
|
||||
f"Current count: {current_count}, max_iterations: {max_iterations}."
|
||||
),
|
||||
rate_limit_type=RateLimitType.MAX_ITERATIONS,
|
||||
)
|
||||
|
||||
verbose_proxy_logger.debug(
|
||||
|
|
|
|||
|
|
@ -16,6 +16,7 @@ from litellm.proxy.auth.auth_utils import (
|
|||
get_key_model_rpm_limit,
|
||||
get_key_model_tpm_limit,
|
||||
)
|
||||
from litellm.exceptions import RateLimitType
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
|
||||
if TYPE_CHECKING:
|
||||
|
|
@ -71,9 +72,21 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
|
|||
)
|
||||
if current is None:
|
||||
if max_parallel_requests == 0 or tpm_limit == 0 or rpm_limit == 0:
|
||||
# base case
|
||||
# base case — at least one dimension is set to 0 (effectively
|
||||
# disabled). Pick the most specific dimension as the
|
||||
# rate_limit_type so dashboards can attribute the failure to
|
||||
# the right cap. Order matters: max_parallel_requests is
|
||||
# listed first because it's the rarest 0 in practice and the
|
||||
# most actionable signal.
|
||||
if max_parallel_requests == 0:
|
||||
triggered_type = RateLimitType.CONCURRENT_REQUESTS
|
||||
elif tpm_limit == 0:
|
||||
triggered_type = RateLimitType.TOKENS
|
||||
else:
|
||||
triggered_type = RateLimitType.REQUESTS
|
||||
self.raise_rate_limit_error(
|
||||
additional_details=f"{CommonProxyErrors.max_parallel_request_limit_reached.value}. Hit limit for {rate_limit_type}. Current limits: max_parallel_requests: {max_parallel_requests}, tpm_limit: {tpm_limit}, rpm_limit: {rpm_limit}"
|
||||
additional_details=f"{CommonProxyErrors.max_parallel_request_limit_reached.value}. Hit limit for {rate_limit_type}. Current limits: max_parallel_requests: {max_parallel_requests}, tpm_limit: {tpm_limit}, rpm_limit: {rpm_limit}",
|
||||
rate_limit_type=triggered_type,
|
||||
)
|
||||
new_val = {
|
||||
"current_requests": 1,
|
||||
|
|
@ -95,9 +108,19 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
|
|||
values_to_update_in_cache.append((request_count_api_key, new_val))
|
||||
|
||||
else:
|
||||
# Detect which dimension actually tripped the limit so we can
|
||||
# surface the right rate_limit_type. Order matches the boolean
|
||||
# condition above (concurrent → tpm → rpm) — first match wins.
|
||||
if int(current["current_requests"]) >= max_parallel_requests:
|
||||
triggered_type = RateLimitType.CONCURRENT_REQUESTS
|
||||
elif current["current_tpm"] >= tpm_limit:
|
||||
triggered_type = RateLimitType.TOKENS
|
||||
else:
|
||||
triggered_type = RateLimitType.REQUESTS
|
||||
raise ProxyRateLimitError(
|
||||
detail=f"LiteLLM Rate Limit Handler for rate limit type = {rate_limit_type}. {CommonProxyErrors.max_parallel_request_limit_reached.value}. current rpm: {current['current_rpm']}, rpm limit: {rpm_limit}, current tpm: {current['current_tpm']}, tpm limit: {tpm_limit}, current max_parallel_requests: {current['current_requests']}, max_parallel_requests: {max_parallel_requests}",
|
||||
headers={"retry-after": str(self.time_to_next_minute())},
|
||||
rate_limit_type=triggered_type,
|
||||
)
|
||||
|
||||
await self.internal_usage_cache.async_batch_set_cache(
|
||||
|
|
@ -121,7 +144,9 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
|
|||
return seconds_to_next_minute
|
||||
|
||||
def raise_rate_limit_error(
|
||||
self, additional_details: Optional[str] = None
|
||||
self,
|
||||
additional_details: Optional[str] = None,
|
||||
rate_limit_type: Optional[RateLimitType] = None,
|
||||
) -> NoReturn:
|
||||
"""
|
||||
Raise a 429 with a retry-after header for litellm-proxy parallel-request limits.
|
||||
|
|
@ -132,6 +157,12 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
|
|||
:class:`litellm.RateLimitError` (so callers can catch by category) and a
|
||||
:class:`fastapi.HTTPException` (so the FastAPI dispatcher serializes it
|
||||
correctly with status 429 and the supplied headers).
|
||||
|
||||
``rate_limit_type`` defaults to ``CONCURRENT_REQUESTS`` because every
|
||||
existing internal caller of this helper hits the parallel-request cap
|
||||
(the global-limit branch in ``async_pre_call_hook`` and the
|
||||
all-zeros base case in ``check_key_in_limits``). Callers that know
|
||||
the dimension exactly should pass it explicitly.
|
||||
"""
|
||||
# additional_details is optional; build the detail with a None-guard
|
||||
# so callers that pass nothing don't get the literal string "None"
|
||||
|
|
@ -142,6 +173,7 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
|
|||
raise ProxyRateLimitError(
|
||||
detail=error_message,
|
||||
headers={"retry-after": str(self.time_to_next_minute())},
|
||||
rate_limit_type=rate_limit_type or RateLimitType.CONCURRENT_REQUESTS,
|
||||
)
|
||||
|
||||
async def get_all_cache_objects(
|
||||
|
|
|
|||
|
|
@ -32,7 +32,10 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
|||
)
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth.auth_utils import get_model_rate_limit_from_metadata
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
|
||||
from litellm.proxy.common_utils.proxy_rate_limit_error import (
|
||||
ProxyRateLimitError,
|
||||
map_v3_rate_limit_type,
|
||||
)
|
||||
from litellm.types.caching import RedisPipelineIncrementOperation
|
||||
from litellm.types.llms.openai import BaseLiteLLMOpenAIResponseObject
|
||||
from litellm.types.utils import ModelResponse, Usage
|
||||
|
|
@ -1875,6 +1878,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
"rate_limit_type": str(status["rate_limit_type"]),
|
||||
"reset_at": reset_time_formatted,
|
||||
},
|
||||
rate_limit_type=map_v3_rate_limit_type(status["rate_limit_type"]),
|
||||
)
|
||||
|
||||
async def async_pre_call_hook(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue