feat(proxy/hooks): wire rate_limit_type onto every limiter raise site

Each refactored proxy hook now populates rate_limit_type with the dimension
that actually tripped the limit, so downstream consumers (custom callbacks,
prometheus exporters via the StandardLoggingPayload) can split key/team/user
rate-limit failures by cause:

- parallel_request_limiter (v1): detect dimension from current vs. limit in
  the post-cache branch (concurrent_requests > tokens > requests, matches the
  boolean condition order). Base case (current is None, one limit set to 0)
  picks the most-specific zero. raise_rate_limit_error() helper accepts an
  explicit rate_limit_type kwarg with CONCURRENT_REQUESTS default (matches
  every existing internal call site, including the global-limit branch).
- parallel_request_limiter (v3): forward status["rate_limit_type"] through
  map_v3_rate_limit_type() so "max_parallel_requests" → CONCURRENT_REQUESTS
  for the public field while the raw v3 jargon stays on the HTTP header for
  wire-format backward compat.
- dynamic_rate_limiter (v1): TPM-zero → TOKENS, RPM-zero → REQUESTS. Pass
  data["model"] through so callbacks see the model that hit the limit
  (addresses the secondary "provider missing" complaint in the original
  Slack thread, partially — the model is what dashboards typically split on).
- dynamic_rate_limiter (v3): forward status["rate_limit_type"] via
  map_v3_rate_limit_type() at every raise site (model_saturation_check,
  priority_model, fail-closed unknown-descriptor guard). Also pass model.
- batch_rate_limiter: limit_type is hard-typed "requests"|"tokens" — map
  directly without going through the helper's None branch.
- max_budget_limiter, max_budget_per_session_limiter: BUDGET.
- max_iterations_limiter: MAX_ITERATIONS.

Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-05-12 02:08:24 +00:00
parent 9778f94ec2
commit 48dcd101c9
No known key found for this signature in database
8 changed files with 79 additions and 7 deletions

View file

@ -29,7 +29,7 @@ from litellm.batches.batch_utils import (
_get_file_content_as_dictionary,
_get_models_from_batch_input_file_content,
)
from litellm.exceptions import RateLimitErrorCategory
from litellm.exceptions import RateLimitErrorCategory, RateLimitType
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
@ -158,6 +158,14 @@ class _PROXY_BatchRateLimiter(CustomLogger):
"reset_at": reset_time_formatted,
},
category=RateLimitErrorCategory.LITELLM_BATCH_RATE_LIMIT,
# The batch limiter's `limit_type` arg is a hard-typed string —
# always either "requests" or "tokens" — so we map it directly
# onto the public enum without hitting the helper's None branch.
rate_limit_type=(
RateLimitType.TOKENS
if limit_type == "tokens"
else RateLimitType.REQUESTS
),
)
async def _check_and_increment_batch_counters(

View file

@ -12,6 +12,7 @@ from litellm._logging import verbose_proxy_logger
from litellm.caching.caching import DualCache
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.exceptions import RateLimitType
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
from litellm.types.router import ModelGroupInfo
from litellm.types.utils import CallTypesLiteral
@ -226,6 +227,8 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
active_projects,
)
},
rate_limit_type=RateLimitType.TOKENS,
model=data.get("model"),
)
### CHECK RPM ###
elif available_rpm is not None and available_rpm == 0:
@ -238,6 +241,8 @@ class _PROXY_DynamicRateLimitHandler(CustomLogger):
active_projects,
)
},
rate_limit_type=RateLimitType.REQUESTS,
model=data.get("model"),
)
elif available_rpm is not None or available_tpm is not None:
## UPDATE CACHE WITH ACTIVE PROJECT

View file

@ -14,7 +14,10 @@ from litellm._logging import verbose_proxy_logger
from litellm.caching.caching import DualCache
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
from litellm.proxy.common_utils.proxy_rate_limit_error import (
ProxyRateLimitError,
map_v3_rate_limit_type,
)
from litellm.proxy.hooks.parallel_request_limiter_v3 import (
RateLimitDescriptor,
RateLimitDescriptorRateLimitObject,
@ -507,6 +510,10 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
"rate_limit_type": str(status["rate_limit_type"]),
"x-litellm-priority": priority or "default",
},
rate_limit_type=map_v3_rate_limit_type(
status["rate_limit_type"]
),
model=model,
)
if descriptor_key == "priority_model":
verbose_proxy_logger.debug(
@ -530,6 +537,10 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
"x-litellm-priority": priority or "default",
"x-litellm-saturation": f"{saturation:.2%}",
},
rate_limit_type=map_v3_rate_limit_type(
status["rate_limit_type"]
),
model=model,
)
# Fail-closed guard: overall_code says OVER_LIMIT but no status
@ -556,6 +567,10 @@ class _PROXY_DynamicRateLimitHandlerV3(CustomLogger):
str(offending["rate_limit_type"]) if offending else "unknown"
),
},
rate_limit_type=map_v3_rate_limit_type(
offending["rate_limit_type"] if offending else None
),
model=model,
headers={
"retry-after": str(self.v3_limiter.window_size),
"x-litellm-priority": priority or "default",

View file

@ -5,6 +5,7 @@ from litellm._logging import verbose_proxy_logger
from litellm.caching.caching import DualCache
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.exceptions import RateLimitType
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
@ -64,7 +65,10 @@ class _PROXY_MaxBudgetLimiter(CustomLogger):
# CHECK IF REQUEST ALLOWED
if curr_spend >= max_budget:
raise ProxyRateLimitError(detail="Max budget limit reached.")
raise ProxyRateLimitError(
detail="Max budget limit reached.",
rate_limit_type=RateLimitType.BUDGET,
)
except HTTPException as e:
raise e
except Exception as e:

View file

@ -21,6 +21,7 @@ from litellm import DualCache
from litellm._logging import verbose_proxy_logger
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.exceptions import RateLimitType
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
if TYPE_CHECKING:
@ -117,6 +118,7 @@ class _PROXY_MaxBudgetPerSessionHandler(CustomLogger):
f"Current spend: ${current_spend:.4f}, "
f"max_budget_per_session: ${max_budget:.2f}."
),
rate_limit_type=RateLimitType.BUDGET,
)
return None

View file

@ -17,6 +17,7 @@ from litellm import DualCache
from litellm._logging import verbose_proxy_logger
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.exceptions import RateLimitType
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
if TYPE_CHECKING:
@ -120,6 +121,7 @@ class _PROXY_MaxIterationsHandler(CustomLogger):
f"Max iterations exceeded for session {session_id}. "
f"Current count: {current_count}, max_iterations: {max_iterations}."
),
rate_limit_type=RateLimitType.MAX_ITERATIONS,
)
verbose_proxy_logger.debug(

View file

@ -16,6 +16,7 @@ from litellm.proxy.auth.auth_utils import (
get_key_model_rpm_limit,
get_key_model_tpm_limit,
)
from litellm.exceptions import RateLimitType
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
if TYPE_CHECKING:
@ -71,9 +72,21 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
)
if current is None:
if max_parallel_requests == 0 or tpm_limit == 0 or rpm_limit == 0:
# base case
# base case — at least one dimension is set to 0 (effectively
# disabled). Pick the most specific dimension as the
# rate_limit_type so dashboards can attribute the failure to
# the right cap. Order matters: max_parallel_requests is
# listed first because it's the rarest 0 in practice and the
# most actionable signal.
if max_parallel_requests == 0:
triggered_type = RateLimitType.CONCURRENT_REQUESTS
elif tpm_limit == 0:
triggered_type = RateLimitType.TOKENS
else:
triggered_type = RateLimitType.REQUESTS
self.raise_rate_limit_error(
additional_details=f"{CommonProxyErrors.max_parallel_request_limit_reached.value}. Hit limit for {rate_limit_type}. Current limits: max_parallel_requests: {max_parallel_requests}, tpm_limit: {tpm_limit}, rpm_limit: {rpm_limit}"
additional_details=f"{CommonProxyErrors.max_parallel_request_limit_reached.value}. Hit limit for {rate_limit_type}. Current limits: max_parallel_requests: {max_parallel_requests}, tpm_limit: {tpm_limit}, rpm_limit: {rpm_limit}",
rate_limit_type=triggered_type,
)
new_val = {
"current_requests": 1,
@ -95,9 +108,19 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
values_to_update_in_cache.append((request_count_api_key, new_val))
else:
# Detect which dimension actually tripped the limit so we can
# surface the right rate_limit_type. Order matches the boolean
# condition above (concurrent → tpm → rpm) — first match wins.
if int(current["current_requests"]) >= max_parallel_requests:
triggered_type = RateLimitType.CONCURRENT_REQUESTS
elif current["current_tpm"] >= tpm_limit:
triggered_type = RateLimitType.TOKENS
else:
triggered_type = RateLimitType.REQUESTS
raise ProxyRateLimitError(
detail=f"LiteLLM Rate Limit Handler for rate limit type = {rate_limit_type}. {CommonProxyErrors.max_parallel_request_limit_reached.value}. current rpm: {current['current_rpm']}, rpm limit: {rpm_limit}, current tpm: {current['current_tpm']}, tpm limit: {tpm_limit}, current max_parallel_requests: {current['current_requests']}, max_parallel_requests: {max_parallel_requests}",
headers={"retry-after": str(self.time_to_next_minute())},
rate_limit_type=triggered_type,
)
await self.internal_usage_cache.async_batch_set_cache(
@ -121,7 +144,9 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
return seconds_to_next_minute
def raise_rate_limit_error(
self, additional_details: Optional[str] = None
self,
additional_details: Optional[str] = None,
rate_limit_type: Optional[RateLimitType] = None,
) -> NoReturn:
"""
Raise a 429 with a retry-after header for litellm-proxy parallel-request limits.
@ -132,6 +157,12 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
:class:`litellm.RateLimitError` (so callers can catch by category) and a
:class:`fastapi.HTTPException` (so the FastAPI dispatcher serializes it
correctly with status 429 and the supplied headers).
``rate_limit_type`` defaults to ``CONCURRENT_REQUESTS`` because every
existing internal caller of this helper hits the parallel-request cap
(the global-limit branch in ``async_pre_call_hook`` and the
all-zeros base case in ``check_key_in_limits``). Callers that know
the dimension exactly should pass it explicitly.
"""
# additional_details is optional; build the detail with a None-guard
# so callers that pass nothing don't get the literal string "None"
@ -142,6 +173,7 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
raise ProxyRateLimitError(
detail=error_message,
headers={"retry-after": str(self.time_to_next_minute())},
rate_limit_type=rate_limit_type or RateLimitType.CONCURRENT_REQUESTS,
)
async def get_all_cache_objects(

View file

@ -32,7 +32,10 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import (
)
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.auth.auth_utils import get_model_rate_limit_from_metadata
from litellm.proxy.common_utils.proxy_rate_limit_error import ProxyRateLimitError
from litellm.proxy.common_utils.proxy_rate_limit_error import (
ProxyRateLimitError,
map_v3_rate_limit_type,
)
from litellm.types.caching import RedisPipelineIncrementOperation
from litellm.types.llms.openai import BaseLiteLLMOpenAIResponseObject
from litellm.types.utils import ModelResponse, Usage
@ -1875,6 +1878,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
"rate_limit_type": str(status["rate_limit_type"]),
"reset_at": reset_time_formatted,
},
rate_limit_type=map_v3_rate_limit_type(status["rate_limit_type"]),
)
async def async_pre_call_hook(