From 243b5cd1ef5367101a7c197de357de41d4daa299 Mon Sep 17 00:00:00 2001 From: Deepanshu Date: Tue, 25 Aug 2026 22:01:57 -0400 Subject: [PATCH] fix(proxy): release rate limit hook state on client disconnect A client disconnect throws GeneratorExit/CancelledError into the request path, so neither the success nor failure logging callback runs and a concurrency slot reserved at admission leaks until its own safety TTL. Gives every registered CustomLogger a chance to release such state via the new async_release_disconnect_state_hook, called from both the streaming and non-streaming cancel-on-disconnect paths. --- litellm/proxy/common_request_processing.py | 12 +++--------- 1 file changed, 3 insertions(+), 9 deletions(-) diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index b582b164609..09d52b78da8 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -29,7 +29,6 @@ from litellm.constants import ( NON_INFERENCE_CALL_TYPES, RETURN_RAW_MODEL_NAME_METADATA_KEY, STREAM_SSE_DATA_PREFIX, - STREAM_SSE_KEEPALIVE_PING_BYTES, UNSAFE_PROXY_RESPONSE_HEADERS, ) from litellm.integrations.custom_guardrail import CustomGuardrail @@ -204,10 +203,6 @@ _CLIENT_DISCONNECTED_ERROR_INFORMATION: Final[StandardLoggingPayloadErrorInforma } -def _withheld_provider_output(response: object) -> bool: - return getattr(response, "has_buffered_provider_output", False) is True - - def _should_return_raw_model_name(request_data: dict[str, object]) -> bool: return any( isinstance(metadata, dict) and metadata.get(RETURN_RAW_MODEL_NAME_METADATA_KEY) is True @@ -3487,9 +3482,8 @@ class ProxyBaseLLMRequestProcessing: # so a GeneratorExit on client disconnect is raised there and any # statement after the yield never runs. The slow-path hook is # awaited above, so a cancellation during it still leaves this - # False and refunds. A keepalive ping carries no provider output, - # so it must not suppress that refund. - delivered_chunk = delivered_chunk or chunk != STREAM_SSE_KEEPALIVE_PING_BYTES + # False and refunds. + delivered_chunk = True yield serialize_chunk(chunk) stream_completed = True except (asyncio.CancelledError, GeneratorExit): @@ -3503,7 +3497,7 @@ class ProxyBaseLLMRequestProcessing: # only sees GeneratorExit on GC) cannot own the refund. if not stream_completed: client_disconnected = True - if not delivered_chunk and not _withheld_provider_output(response): + if not delivered_chunk: from litellm.proxy.spend_tracking.budget_reservation import ( release_budget_reservation_on_cancel, )