mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
feat(rate-limiting): make the tag rate limiter's in-memory cache size configurable
The isolated in-memory cache added for this hook still defaults to 200 entries. A deployment rate-limiting on a high-cardinality tag_id without Redis can churn past that cap, evicting an active counter before its period elapses. litellm_settings.tag_rate_limiter_max_in_memory_cache_size lets that ceiling be raised; 0 is rejected in favor of the safe default since it would disable the in-memory cache outright.
This commit is contained in:
parent
e6c504645b
commit
fe9e36a0bc
3 changed files with 109 additions and 1 deletions
|
|
@ -392,6 +392,7 @@ cache: Optional["Cache"] = None # cache object <- use this - https://docs.litel
|
|||
default_in_memory_ttl: Optional[float] = None
|
||||
default_redis_ttl: Optional[float] = None
|
||||
default_redis_batch_cache_expiry: Optional[float] = None
|
||||
tag_rate_limiter_max_in_memory_cache_size: Optional[int] = None
|
||||
model_alias_map: Dict[str, str] = {}
|
||||
model_group_settings: Optional["ModelGroupSettings"] = None
|
||||
max_budget: float = 0.0 # set the max budget across all providers
|
||||
|
|
|
|||
|
|
@ -9,8 +9,10 @@ from itertools import groupby
|
|||
from types import MappingProxyType
|
||||
from typing import TYPE_CHECKING, Final, Literal, NamedTuple, TypeAlias
|
||||
|
||||
import litellm
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.caching.dual_cache import DualCache
|
||||
from litellm.caching.in_memory_cache import InMemoryCache
|
||||
from litellm.exceptions import RateLimitType
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.litellm_core_utils.core_helpers import (
|
||||
|
|
@ -624,7 +626,20 @@ class _PROXY_TagRateLimiter( # pyright: ignore[reportUnusedClass] # only refer
|
|||
# ceiling could evict an unrelated, authentication-bound counter and
|
||||
# exceed a limit nothing here configured. The real Redis connection
|
||||
# (if any) is still shared, so cross-instance correctness is unaffected.
|
||||
isolated_dual_cache: Final = DualCache(redis_cache=internal_usage_cache.redis_cache)
|
||||
#
|
||||
# This cache's own 200-item default is still shared across every
|
||||
# distinct tag value this hook sees. Deployments rate-limiting on a
|
||||
# high-cardinality tag_id (e.g. per end user) without Redis can raise
|
||||
# `litellm_settings.tag_rate_limiter_max_in_memory_cache_size` so
|
||||
# active buckets aren't evicted before their period elapses. 0 would
|
||||
# disable this hook's in-memory cache outright, so it's rejected here
|
||||
# in favor of the safe default.
|
||||
configured_max_cache_size: Final = litellm.tag_rate_limiter_max_in_memory_cache_size
|
||||
max_cache_size: Final = configured_max_cache_size if configured_max_cache_size else None
|
||||
isolated_dual_cache: Final = DualCache(
|
||||
in_memory_cache=InMemoryCache(max_size_in_memory=max_cache_size),
|
||||
redis_cache=internal_usage_cache.redis_cache,
|
||||
)
|
||||
self.internal_usage_cache = InternalUsageCache(dual_cache=isolated_dual_cache)
|
||||
self._v3 = _PROXY_MaxParallelRequestsHandler_v3(self.internal_usage_cache, time_provider=time_provider)
|
||||
self._time_provider = time_provider or datetime.now
|
||||
|
|
|
|||
|
|
@ -2181,3 +2181,95 @@ async def test_flooding_tag_buckets_does_not_evict_the_shared_cache_authenticati
|
|||
)
|
||||
|
||||
assert await shared_cache.async_get_cache(key="authentication_bound_counter") == "do-not-evict"
|
||||
|
||||
|
||||
def _single_request_per_minute_router() -> "litellm.Router":
|
||||
return litellm.Router(
|
||||
model_list=[
|
||||
_deployment(
|
||||
"grp",
|
||||
"dep-1",
|
||||
{
|
||||
"request_limits": {
|
||||
"limits": [{"name": "per_minute", "tag_id": "end_user_id", "limit": 1, "period_seconds": 60}]
|
||||
}
|
||||
},
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_max_in_memory_cache_size_setting_lets_high_cardinality_tags_avoid_early_eviction(
|
||||
time_controller, monkeypatch
|
||||
):
|
||||
"""
|
||||
This hook's own isolated cache still defaults to 200 items, shared across
|
||||
every distinct tag value it sees. A deployment rate-limiting on a
|
||||
high-cardinality tag_id (e.g. per end user) without Redis can raise
|
||||
`litellm_settings.tag_rate_limiter_max_in_memory_cache_size` so an
|
||||
earlier bucket survives churn from later, unrelated tag values: with
|
||||
limit=1, a still-live bucket rejects a second request instead of having
|
||||
been evicted back to a fresh count of 0.
|
||||
"""
|
||||
monkeypatch.setattr(litellm, "tag_rate_limiter_max_in_memory_cache_size", 500)
|
||||
|
||||
limiter = _PROXY_TagRateLimiter(internal_usage_cache=DualCache(), time_provider=time_controller.now)
|
||||
router = _single_request_per_minute_router()
|
||||
limiter.update_variables(llm_router=router)
|
||||
healthy = router.model_list
|
||||
|
||||
await limiter.async_filter_deployments(
|
||||
model="grp",
|
||||
healthy_deployments=healthy,
|
||||
messages=None,
|
||||
request_kwargs={"metadata": {"tags": ["end_user_id:early-user"]}},
|
||||
)
|
||||
|
||||
for i in range(250):
|
||||
await limiter.async_filter_deployments(
|
||||
model="grp",
|
||||
healthy_deployments=healthy,
|
||||
messages=None,
|
||||
request_kwargs={"metadata": {"tags": [f"end_user_id:flood-{i}"]}},
|
||||
)
|
||||
|
||||
with pytest.raises(ProxyRateLimitError):
|
||||
await limiter.async_filter_deployments(
|
||||
model="grp",
|
||||
healthy_deployments=healthy,
|
||||
messages=None,
|
||||
request_kwargs={"metadata": {"tags": ["end_user_id:early-user"]}},
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_max_in_memory_cache_size_of_zero_falls_back_to_the_safe_default(time_controller, monkeypatch):
|
||||
"""
|
||||
0 would hit InMemoryCache.set_cache's own `max_size_in_memory == 0`
|
||||
short-circuit and silently disable this hook's in-memory cache outright,
|
||||
so it must be rejected in favor of the safe 200-item default rather than
|
||||
passed straight through: a limit=1 bucket must still reject a second,
|
||||
immediate request for the same tag.
|
||||
"""
|
||||
monkeypatch.setattr(litellm, "tag_rate_limiter_max_in_memory_cache_size", 0)
|
||||
|
||||
limiter = _PROXY_TagRateLimiter(internal_usage_cache=DualCache(), time_provider=time_controller.now)
|
||||
router = _single_request_per_minute_router()
|
||||
limiter.update_variables(llm_router=router)
|
||||
healthy = router.model_list
|
||||
|
||||
await limiter.async_filter_deployments(
|
||||
model="grp",
|
||||
healthy_deployments=healthy,
|
||||
messages=None,
|
||||
request_kwargs={"metadata": {"tags": ["end_user_id:u1"]}},
|
||||
)
|
||||
|
||||
with pytest.raises(ProxyRateLimitError):
|
||||
await limiter.async_filter_deployments(
|
||||
model="grp",
|
||||
healthy_deployments=healthy,
|
||||
messages=None,
|
||||
request_kwargs={"metadata": {"tags": ["end_user_id:u1"]}},
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue