diff --git a/litellm/__init__.py b/litellm/__init__.py index 27cef1ce229..3f68dd0cbd1 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -392,6 +392,7 @@ cache: Optional["Cache"] = None # cache object <- use this - https://docs.litel default_in_memory_ttl: Optional[float] = None default_redis_ttl: Optional[float] = None default_redis_batch_cache_expiry: Optional[float] = None +tag_rate_limiter_max_in_memory_cache_size: Optional[int] = None model_alias_map: Dict[str, str] = {} model_group_settings: Optional["ModelGroupSettings"] = None max_budget: float = 0.0 # set the max budget across all providers diff --git a/litellm/proxy/hooks/tag_rate_limiter.py b/litellm/proxy/hooks/tag_rate_limiter.py index b55c472f89d..96645f4b600 100644 --- a/litellm/proxy/hooks/tag_rate_limiter.py +++ b/litellm/proxy/hooks/tag_rate_limiter.py @@ -9,8 +9,10 @@ from itertools import groupby from types import MappingProxyType from typing import TYPE_CHECKING, Final, Literal, NamedTuple, TypeAlias +import litellm from litellm._logging import verbose_proxy_logger from litellm.caching.dual_cache import DualCache +from litellm.caching.in_memory_cache import InMemoryCache from litellm.exceptions import RateLimitType from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.core_helpers import ( @@ -624,7 +626,20 @@ class _PROXY_TagRateLimiter( # pyright: ignore[reportUnusedClass] # only refer # ceiling could evict an unrelated, authentication-bound counter and # exceed a limit nothing here configured. The real Redis connection # (if any) is still shared, so cross-instance correctness is unaffected. - isolated_dual_cache: Final = DualCache(redis_cache=internal_usage_cache.redis_cache) + # + # This cache's own 200-item default is still shared across every + # distinct tag value this hook sees. Deployments rate-limiting on a + # high-cardinality tag_id (e.g. per end user) without Redis can raise + # `litellm_settings.tag_rate_limiter_max_in_memory_cache_size` so + # active buckets aren't evicted before their period elapses. 0 would + # disable this hook's in-memory cache outright, so it's rejected here + # in favor of the safe default. + configured_max_cache_size: Final = litellm.tag_rate_limiter_max_in_memory_cache_size + max_cache_size: Final = configured_max_cache_size if configured_max_cache_size else None + isolated_dual_cache: Final = DualCache( + in_memory_cache=InMemoryCache(max_size_in_memory=max_cache_size), + redis_cache=internal_usage_cache.redis_cache, + ) self.internal_usage_cache = InternalUsageCache(dual_cache=isolated_dual_cache) self._v3 = _PROXY_MaxParallelRequestsHandler_v3(self.internal_usage_cache, time_provider=time_provider) self._time_provider = time_provider or datetime.now diff --git a/tests/test_litellm/proxy/hooks/test_tag_rate_limiter.py b/tests/test_litellm/proxy/hooks/test_tag_rate_limiter.py index c0c943c4b5f..7119a0c154d 100644 --- a/tests/test_litellm/proxy/hooks/test_tag_rate_limiter.py +++ b/tests/test_litellm/proxy/hooks/test_tag_rate_limiter.py @@ -2181,3 +2181,95 @@ async def test_flooding_tag_buckets_does_not_evict_the_shared_cache_authenticati ) assert await shared_cache.async_get_cache(key="authentication_bound_counter") == "do-not-evict" + + +def _single_request_per_minute_router() -> "litellm.Router": + return litellm.Router( + model_list=[ + _deployment( + "grp", + "dep-1", + { + "request_limits": { + "limits": [{"name": "per_minute", "tag_id": "end_user_id", "limit": 1, "period_seconds": 60}] + } + }, + ) + ] + ) + + +@pytest.mark.asyncio +async def test_max_in_memory_cache_size_setting_lets_high_cardinality_tags_avoid_early_eviction( + time_controller, monkeypatch +): + """ + This hook's own isolated cache still defaults to 200 items, shared across + every distinct tag value it sees. A deployment rate-limiting on a + high-cardinality tag_id (e.g. per end user) without Redis can raise + `litellm_settings.tag_rate_limiter_max_in_memory_cache_size` so an + earlier bucket survives churn from later, unrelated tag values: with + limit=1, a still-live bucket rejects a second request instead of having + been evicted back to a fresh count of 0. + """ + monkeypatch.setattr(litellm, "tag_rate_limiter_max_in_memory_cache_size", 500) + + limiter = _PROXY_TagRateLimiter(internal_usage_cache=DualCache(), time_provider=time_controller.now) + router = _single_request_per_minute_router() + limiter.update_variables(llm_router=router) + healthy = router.model_list + + await limiter.async_filter_deployments( + model="grp", + healthy_deployments=healthy, + messages=None, + request_kwargs={"metadata": {"tags": ["end_user_id:early-user"]}}, + ) + + for i in range(250): + await limiter.async_filter_deployments( + model="grp", + healthy_deployments=healthy, + messages=None, + request_kwargs={"metadata": {"tags": [f"end_user_id:flood-{i}"]}}, + ) + + with pytest.raises(ProxyRateLimitError): + await limiter.async_filter_deployments( + model="grp", + healthy_deployments=healthy, + messages=None, + request_kwargs={"metadata": {"tags": ["end_user_id:early-user"]}}, + ) + + +@pytest.mark.asyncio +async def test_max_in_memory_cache_size_of_zero_falls_back_to_the_safe_default(time_controller, monkeypatch): + """ + 0 would hit InMemoryCache.set_cache's own `max_size_in_memory == 0` + short-circuit and silently disable this hook's in-memory cache outright, + so it must be rejected in favor of the safe 200-item default rather than + passed straight through: a limit=1 bucket must still reject a second, + immediate request for the same tag. + """ + monkeypatch.setattr(litellm, "tag_rate_limiter_max_in_memory_cache_size", 0) + + limiter = _PROXY_TagRateLimiter(internal_usage_cache=DualCache(), time_provider=time_controller.now) + router = _single_request_per_minute_router() + limiter.update_variables(llm_router=router) + healthy = router.model_list + + await limiter.async_filter_deployments( + model="grp", + healthy_deployments=healthy, + messages=None, + request_kwargs={"metadata": {"tags": ["end_user_id:u1"]}}, + ) + + with pytest.raises(ProxyRateLimitError): + await limiter.async_filter_deployments( + model="grp", + healthy_deployments=healthy, + messages=None, + request_kwargs={"metadata": {"tags": ["end_user_id:u1"]}}, + )