mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(batches): skip failed line item events in parallel request limiter failure hooks
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
67d4471a37
commit
ee8dcfc20d
4 changed files with 59 additions and 0 deletions
|
|
@ -699,6 +699,8 @@ class _PROXY_MaxParallelRequestsHandler(CustomLogger):
|
|||
self.print_verbose(e)
|
||||
|
||||
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time):
|
||||
if is_batch_line_item_event(kwargs):
|
||||
return
|
||||
try:
|
||||
self.print_verbose("Inside Max Parallel Request Failure Hook")
|
||||
litellm_parent_otel_span: Final[Span | None] = _get_parent_otel_span_from_kwargs(kwargs=kwargs)
|
||||
|
|
|
|||
|
|
@ -4681,6 +4681,8 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
whose partial usage was recovered settles the reservation at that
|
||||
usage instead of refunding it.
|
||||
"""
|
||||
if is_batch_line_item_event(kwargs):
|
||||
return
|
||||
from litellm.litellm_core_utils.core_helpers import (
|
||||
_get_parent_otel_span_from_kwargs,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -84,3 +84,29 @@ async def test_async_log_success_event_counts_non_chat_response_tokens(response_
|
|||
f"expected 50 tokens counted for {scope_id}, "
|
||||
f"got {current['current_tpm']}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_log_failure_event_skips_batch_line_item_events():
|
||||
"""Failed line children were never admitted by the limiter, so the failure
|
||||
hook must not decrement request counters they never incremented."""
|
||||
parallel_request_handler = _PROXY_MaxParallelRequestsHandler(
|
||||
internal_usage_cache=InternalUsageCache(DualCache())
|
||||
)
|
||||
local_cache = parallel_request_handler.internal_usage_cache.dual_cache.in_memory_cache
|
||||
|
||||
await parallel_request_handler.async_log_failure_event(
|
||||
kwargs={
|
||||
"exception": "litellm.APIError: upstream 500",
|
||||
"litellm_params": {
|
||||
"batch_parent_id": "batch_1",
|
||||
"metadata": {"user_api_key": hash_token("sk-line-item")},
|
||||
},
|
||||
"model": "gpt-3.5-turbo",
|
||||
},
|
||||
response_obj=None,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
assert local_cache.cache_dict == {}
|
||||
|
|
|
|||
|
|
@ -7177,3 +7177,32 @@ async def test_async_log_success_event_skips_batch_line_item_events():
|
|||
)
|
||||
|
||||
assert local_cache.in_memory_cache.cache_dict == {}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_log_failure_event_skips_batch_line_item_events():
|
||||
"""Failed line children were never admitted by the limiter, so the failure
|
||||
hook must not release slots or refund TPM they never reserved."""
|
||||
local_cache = DualCache()
|
||||
handler = _PROXY_MaxParallelRequestsHandler(internal_usage_cache=InternalUsageCache(local_cache))
|
||||
stash = get_or_create_request_stash()
|
||||
stash.owner_litellm_call_id = "call-line-item"
|
||||
stash.reserved_tokens = 50
|
||||
stash.reserved_scopes = frozenset({("api_key", hash_token("sk-line-item"))})
|
||||
|
||||
await handler.async_log_failure_event(
|
||||
kwargs={
|
||||
"litellm_call_id": "call-line-item",
|
||||
"standard_logging_object": {"metadata": {"user_api_key_hash": hash_token("sk-line-item")}},
|
||||
"litellm_params": {
|
||||
"batch_parent_id": "batch_1",
|
||||
"metadata": {"user_api_key_hash": hash_token("sk-line-item"), "model_group": "gpt-3.5-turbo"},
|
||||
},
|
||||
"model": "gpt-3.5-turbo",
|
||||
},
|
||||
response_obj=None,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
)
|
||||
|
||||
assert local_cache.in_memory_cache.cache_dict == {}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue