feat(rate-limiting): add per-deployment tag rate limiting hook

Enforces token, request, dollar, and concurrency limits scoped to a
request tag (end_user_id by default), configured per deployment under
model_info.tag_rate_limits and admitted once per routing hop. Supports
chain-wide and per-deployment-scoped buckets, team-aliased routing
groups, and per-entry scoping via enabled_for/disabled_for/
apply_to_key_alias/apply_to_models.

Registers as the model_based_tag_rate_limits_hook callback and reuses
the identity extraction, policy fingerprinting, and bucket-key hashing
primitives from tag_rate_limits_shared.py.
This commit is contained in:
Deepanshu 2026-08-25 22:01:46 -04:00
parent 2c9dc8baa4
commit 8b227a8ff2
4 changed files with 6492 additions and 0 deletions

View file

@ -119,6 +119,7 @@ _custom_logger_compatible_callbacks_literal = Literal[
"litellm_agent",
"dynamic_rate_limiter",
"dynamic_rate_limiter_v3",
"model_based_tag_rate_limits_hook",
"langsmith",
"prometheus",
"otel",
@ -391,6 +392,7 @@ cache: Optional["Cache"] = None # cache object <- use this - https://docs.litel
default_in_memory_ttl: Optional[float] = None
default_redis_ttl: Optional[float] = None
default_redis_batch_cache_expiry: Optional[float] = None
model_based_tag_rate_limits_max_in_memory_cache_size: Optional[int] = None
model_alias_map: Dict[str, str] = {}
model_group_settings: Optional["ModelGroupSettings"] = None
max_budget: float = 0.0 # set the max budget across all providers

View file

@ -4447,6 +4447,26 @@ def _init_custom_logger_compatible_class(
dynamic_rate_limiter_obj_v3.update_variables(llm_router=llm_router)
_in_memory_loggers.append(dynamic_rate_limiter_obj_v3)
return dynamic_rate_limiter_obj_v3
elif logging_integration == "model_based_tag_rate_limits_hook":
from litellm.proxy.hooks.model_based_tag_rate_limits_hook import (
_PROXY_ModelBasedTagRateLimitsHook, # pyright: ignore[reportPrivateUsage] # resolved by name like every other opt-in callback here
)
for callback in _in_memory_loggers:
if isinstance(callback, _PROXY_ModelBasedTagRateLimitsHook):
return callback
if internal_usage_cache is None:
raise Exception(f"Internal Error: Cache cannot be empty - internal_usage_cache={internal_usage_cache}")
model_based_tag_rate_limits_hook_obj: Final = _PROXY_ModelBasedTagRateLimitsHook(
internal_usage_cache=internal_usage_cache
)
if llm_router is not None and isinstance(llm_router, litellm.Router):
model_based_tag_rate_limits_hook_obj.update_variables(llm_router=llm_router)
_in_memory_loggers.append(model_based_tag_rate_limits_hook_obj)
return model_based_tag_rate_limits_hook_obj
elif logging_integration == "langtrace":
if "LANGTRACE_API_KEY" not in os.environ:
raise ValueError("LANGTRACE_API_KEY not found in environment variables")
@ -4874,6 +4894,15 @@ def get_custom_logger_compatible_class(
if isinstance(callback, _PROXY_DynamicRateLimitHandlerV3):
return callback
elif logging_integration == "model_based_tag_rate_limits_hook":
from litellm.proxy.hooks.model_based_tag_rate_limits_hook import (
_PROXY_ModelBasedTagRateLimitsHook, # pyright: ignore[reportPrivateUsage] # resolved by name like every other opt-in callback here
)
for callback in _in_memory_loggers:
if isinstance(callback, _PROXY_ModelBasedTagRateLimitsHook):
return callback
elif logging_integration == "langtrace":
from litellm.integrations.opentelemetry import OpenTelemetry

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff