diff --git a/litellm/proxy/hooks/parallel_request_limiter_v3.py b/litellm/proxy/hooks/parallel_request_limiter_v3.py index 486e3cf88a4..89559ff72ae 100644 --- a/litellm/proxy/hooks/parallel_request_limiter_v3.py +++ b/litellm/proxy/hooks/parallel_request_limiter_v3.py @@ -1341,7 +1341,7 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger): else: limit_value = rate_limit.get("tokens_per_unit") inc_amount = int(increment_amounts.get("tokens", 0) or 0) - if limit_value is None: + if limit_value is None or inc_amount < 0: continue counter_key = self.create_rate_limit_keys(descriptor_key, descriptor_value, rlt) # Counter-key TTL and window_size are conceptually distinct diff --git a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py index 84e41711176..7b3f00a55a8 100644 --- a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py +++ b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py @@ -5055,3 +5055,17 @@ async def test_atomic_check_with_zero_increment_still_enforces_token_limit(): ) == 150 ) + + negative_increment: Dict[str, int] = {"requests": -1, "tokens": -50} + refund_attempt = await handler.atomic_check_and_increment_by_n( + descriptors=[descriptor], + increments=[negative_increment], + ) + assert refund_attempt["overall_code"] == "OK" + assert refund_attempt["statuses"] == [] + assert ( + await handler.internal_usage_cache.async_get_cache( + key=counter_key, litellm_parent_otel_span=None, local_only=True + ) + == 150 + )