feat(rate-limiting): make the tag rate limiter's in-memory cache size configurable

The isolated in-memory cache added for this hook still defaults to 200
entries. A deployment rate-limiting on a high-cardinality tag_id without
Redis can churn past that cap, evicting an active counter before its
period elapses. litellm_settings.tag_rate_limiter_max_in_memory_cache_size
lets that ceiling be raised; 0 is rejected in favor of the safe default
since it would disable the in-memory cache outright.
This commit is contained in:
Deepanshu 2026-08-17 09:21:54 -04:00
parent e6c504645b
commit fe9e36a0bc
3 changed files with 109 additions and 1 deletions

View file

@ -392,6 +392,7 @@ cache: Optional["Cache"] = None # cache object <- use this - https://docs.litel
default_in_memory_ttl: Optional[float] = None
default_redis_ttl: Optional[float] = None
default_redis_batch_cache_expiry: Optional[float] = None
tag_rate_limiter_max_in_memory_cache_size: Optional[int] = None
model_alias_map: Dict[str, str] = {}
model_group_settings: Optional["ModelGroupSettings"] = None
max_budget: float = 0.0 # set the max budget across all providers

View file

@ -9,8 +9,10 @@ from itertools import groupby
from types import MappingProxyType
from typing import TYPE_CHECKING, Final, Literal, NamedTuple, TypeAlias
import litellm
from litellm._logging import verbose_proxy_logger
from litellm.caching.dual_cache import DualCache
from litellm.caching.in_memory_cache import InMemoryCache
from litellm.exceptions import RateLimitType
from litellm.integrations.custom_logger import CustomLogger
from litellm.litellm_core_utils.core_helpers import (
@ -624,7 +626,20 @@ class _PROXY_TagRateLimiter( # pyright: ignore[reportUnusedClass] # only refer
# ceiling could evict an unrelated, authentication-bound counter and
# exceed a limit nothing here configured. The real Redis connection
# (if any) is still shared, so cross-instance correctness is unaffected.
isolated_dual_cache: Final = DualCache(redis_cache=internal_usage_cache.redis_cache)
#
# This cache's own 200-item default is still shared across every
# distinct tag value this hook sees. Deployments rate-limiting on a
# high-cardinality tag_id (e.g. per end user) without Redis can raise
# `litellm_settings.tag_rate_limiter_max_in_memory_cache_size` so
# active buckets aren't evicted before their period elapses. 0 would
# disable this hook's in-memory cache outright, so it's rejected here
# in favor of the safe default.
configured_max_cache_size: Final = litellm.tag_rate_limiter_max_in_memory_cache_size
max_cache_size: Final = configured_max_cache_size if configured_max_cache_size else None
isolated_dual_cache: Final = DualCache(
in_memory_cache=InMemoryCache(max_size_in_memory=max_cache_size),
redis_cache=internal_usage_cache.redis_cache,
)
self.internal_usage_cache = InternalUsageCache(dual_cache=isolated_dual_cache)
self._v3 = _PROXY_MaxParallelRequestsHandler_v3(self.internal_usage_cache, time_provider=time_provider)
self._time_provider = time_provider or datetime.now

View file

@ -2181,3 +2181,95 @@ async def test_flooding_tag_buckets_does_not_evict_the_shared_cache_authenticati
)
assert await shared_cache.async_get_cache(key="authentication_bound_counter") == "do-not-evict"
def _single_request_per_minute_router() -> "litellm.Router":
return litellm.Router(
model_list=[
_deployment(
"grp",
"dep-1",
{
"request_limits": {
"limits": [{"name": "per_minute", "tag_id": "end_user_id", "limit": 1, "period_seconds": 60}]
}
},
)
]
)
@pytest.mark.asyncio
async def test_max_in_memory_cache_size_setting_lets_high_cardinality_tags_avoid_early_eviction(
time_controller, monkeypatch
):
"""
This hook's own isolated cache still defaults to 200 items, shared across
every distinct tag value it sees. A deployment rate-limiting on a
high-cardinality tag_id (e.g. per end user) without Redis can raise
`litellm_settings.tag_rate_limiter_max_in_memory_cache_size` so an
earlier bucket survives churn from later, unrelated tag values: with
limit=1, a still-live bucket rejects a second request instead of having
been evicted back to a fresh count of 0.
"""
monkeypatch.setattr(litellm, "tag_rate_limiter_max_in_memory_cache_size", 500)
limiter = _PROXY_TagRateLimiter(internal_usage_cache=DualCache(), time_provider=time_controller.now)
router = _single_request_per_minute_router()
limiter.update_variables(llm_router=router)
healthy = router.model_list
await limiter.async_filter_deployments(
model="grp",
healthy_deployments=healthy,
messages=None,
request_kwargs={"metadata": {"tags": ["end_user_id:early-user"]}},
)
for i in range(250):
await limiter.async_filter_deployments(
model="grp",
healthy_deployments=healthy,
messages=None,
request_kwargs={"metadata": {"tags": [f"end_user_id:flood-{i}"]}},
)
with pytest.raises(ProxyRateLimitError):
await limiter.async_filter_deployments(
model="grp",
healthy_deployments=healthy,
messages=None,
request_kwargs={"metadata": {"tags": ["end_user_id:early-user"]}},
)
@pytest.mark.asyncio
async def test_max_in_memory_cache_size_of_zero_falls_back_to_the_safe_default(time_controller, monkeypatch):
"""
0 would hit InMemoryCache.set_cache's own `max_size_in_memory == 0`
short-circuit and silently disable this hook's in-memory cache outright,
so it must be rejected in favor of the safe 200-item default rather than
passed straight through: a limit=1 bucket must still reject a second,
immediate request for the same tag.
"""
monkeypatch.setattr(litellm, "tag_rate_limiter_max_in_memory_cache_size", 0)
limiter = _PROXY_TagRateLimiter(internal_usage_cache=DualCache(), time_provider=time_controller.now)
router = _single_request_per_minute_router()
limiter.update_variables(llm_router=router)
healthy = router.model_list
await limiter.async_filter_deployments(
model="grp",
healthy_deployments=healthy,
messages=None,
request_kwargs={"metadata": {"tags": ["end_user_id:u1"]}},
)
with pytest.raises(ProxyRateLimitError):
await limiter.async_filter_deployments(
model="grp",
healthy_deployments=healthy,
messages=None,
request_kwargs={"metadata": {"tags": ["end_user_id:u1"]}},
)