fix(batches): skip failed line item events in parallel request limiter failure hooks

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
yucheng 2026-09-23 15:48:58 +00:00
parent 67d4471a37
commit ee8dcfc20d
4 changed files with 59 additions and 0 deletions

View file

@ -699,6 +699,8 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
self.print_verbose(e)
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
if is_batch_line_item_event(kwargs):
return
try:
self.print_verbose("Inside Max Parallel Request Failure Hook")
litellm_parent_otel_span: Final[Span | None] = _get_parent_otel_span_from_kwargs(kwargs=kwargs)

View file

@ -4681,6 +4681,8 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
whose partial usage was recovered settles the reservation at that
usage instead of refunding it.
"""
if is_batch_line_item_event(kwargs):
return
from litellm.litellm_core_utils.core_helpers import (
_get_parent_otel_span_from_kwargs,
)

View file

@ -84,3 +84,29 @@ async def test_async_log_success_event_counts_non_chat_response_tokens(response_
f"expected 50 tokens counted for {scope_id}, "
f"got {current['current_tpm']}"
)
@pytest.mark.asyncio
async def test_async_log_failure_event_skips_batch_line_item_events():
"""Failed line children were never admitted by the limiter, so the failure
hook must not decrement request counters they never incremented."""
parallel_request_handler = _PROXY_MaxParallelRequestsHandler(
internal_usage_cache=InternalUsageCache(DualCache())
)
local_cache = parallel_request_handler.internal_usage_cache.dual_cache.in_memory_cache
await parallel_request_handler.async_log_failure_event(
kwargs={
"exception": "litellm.APIError: upstream 500",
"litellm_params": {
"batch_parent_id": "batch_1",
"metadata": {"user_api_key": hash_token("sk-line-item")},
},
"model": "gpt-3.5-turbo",
},
response_obj=None,
start_time=datetime.now(),
end_time=datetime.now(),
)
assert local_cache.cache_dict == {}

View file

@ -7177,3 +7177,32 @@ async def test_async_log_success_event_skips_batch_line_item_events():
)
assert local_cache.in_memory_cache.cache_dict == {}
@pytest.mark.asyncio
async def test_async_log_failure_event_skips_batch_line_item_events():
"""Failed line children were never admitted by the limiter, so the failure
hook must not release slots or refund TPM they never reserved."""
local_cache = DualCache()
handler = _PROXY_MaxParallelRequestsHandler(internal_usage_cache=InternalUsageCache(local_cache))
stash = get_or_create_request_stash()
stash.owner_litellm_call_id = "call-line-item"
stash.reserved_tokens = 50
stash.reserved_scopes = frozenset({("api_key", hash_token("sk-line-item"))})
await handler.async_log_failure_event(
kwargs={
"litellm_call_id": "call-line-item",
"standard_logging_object": {"metadata": {"user_api_key_hash": hash_token("sk-line-item")}},
"litellm_params": {
"batch_parent_id": "batch_1",
"metadata": {"user_api_key_hash": hash_token("sk-line-item"), "model_group": "gpt-3.5-turbo"},
},
"model": "gpt-3.5-turbo",
},
response_obj=None,
start_time=datetime.now(),
end_time=datetime.now(),
)
assert local_cache.in_memory_cache.cache_dict == {}