diff --git a/litellm/proxy/hooks/parallel_request_limiter.py b/litellm/proxy/hooks/parallel_request_limiter.py index a6dde585309..4ad2062a4e2 100644 --- a/litellm/proxy/hooks/parallel_request_limiter.py +++ b/litellm/proxy/hooks/parallel_request_limiter.py @@ -699,6 +699,8 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger): self.print_verbose(e) async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time): + if is_batch_line_item_event(kwargs): + return try: self.print_verbose("Inside Max Parallel Request Failure Hook") litellm_parent_otel_span: Final[Span | None] = _get_parent_otel_span_from_kwargs(kwargs=kwargs) diff --git a/litellm/proxy/hooks/parallel_request_limiter_v3.py b/litellm/proxy/hooks/parallel_request_limiter_v3.py index d5b31624064..981c0b81a07 100644 --- a/litellm/proxy/hooks/parallel_request_limiter_v3.py +++ b/litellm/proxy/hooks/parallel_request_limiter_v3.py @@ -4681,6 +4681,8 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger): whose partial usage was recovered settles the reservation at that usage instead of refunding it. """ + if is_batch_line_item_event(kwargs): + return from litellm.litellm_core_utils.core_helpers import ( _get_parent_otel_span_from_kwargs, ) diff --git a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter.py b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter.py index 0e2683dcbfd..cbc16d734a1 100644 --- a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter.py +++ b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter.py @@ -84,3 +84,29 @@ async def test_async_log_success_event_counts_non_chat_response_tokens(response_ f"expected 50 tokens counted for {scope_id}, " f"got {current['current_tpm']}" ) + + +@pytest.mark.asyncio +async def test_async_log_failure_event_skips_batch_line_item_events(): + """Failed line children were never admitted by the limiter, so the failure + hook must not decrement request counters they never incremented.""" + parallel_request_handler = _PROXY_MaxParallelRequestsHandler( + internal_usage_cache=InternalUsageCache(DualCache()) + ) + local_cache = parallel_request_handler.internal_usage_cache.dual_cache.in_memory_cache + + await parallel_request_handler.async_log_failure_event( + kwargs={ + "exception": "litellm.APIError: upstream 500", + "litellm_params": { + "batch_parent_id": "batch_1", + "metadata": {"user_api_key": hash_token("sk-line-item")}, + }, + "model": "gpt-3.5-turbo", + }, + response_obj=None, + start_time=datetime.now(), + end_time=datetime.now(), + ) + + assert local_cache.cache_dict == {} diff --git a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py index 904b5d67556..ea8d5016463 100644 --- a/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py +++ b/tests/test_litellm/proxy/hooks/test_parallel_request_limiter_v3.py @@ -7177,3 +7177,32 @@ async def test_async_log_success_event_skips_batch_line_item_events(): ) assert local_cache.in_memory_cache.cache_dict == {} + + +@pytest.mark.asyncio +async def test_async_log_failure_event_skips_batch_line_item_events(): + """Failed line children were never admitted by the limiter, so the failure + hook must not release slots or refund TPM they never reserved.""" + local_cache = DualCache() + handler = _PROXY_MaxParallelRequestsHandler(internal_usage_cache=InternalUsageCache(local_cache)) + stash = get_or_create_request_stash() + stash.owner_litellm_call_id = "call-line-item" + stash.reserved_tokens = 50 + stash.reserved_scopes = frozenset({("api_key", hash_token("sk-line-item"))}) + + await handler.async_log_failure_event( + kwargs={ + "litellm_call_id": "call-line-item", + "standard_logging_object": {"metadata": {"user_api_key_hash": hash_token("sk-line-item")}}, + "litellm_params": { + "batch_parent_id": "batch_1", + "metadata": {"user_api_key_hash": hash_token("sk-line-item"), "model_group": "gpt-3.5-turbo"}, + }, + "model": "gpt-3.5-turbo", + }, + response_obj=None, + start_time=datetime.now(), + end_time=datetime.now(), + ) + + assert local_cache.in_memory_cache.cache_dict == {}