fix(proxy): release rate limit hook state on client disconnect

A client disconnect throws GeneratorExit/CancelledError into the
request path, so neither the success nor failure logging callback
runs and a concurrency slot reserved at admission leaks until its own
safety TTL. Gives every registered CustomLogger a chance to release
such state via the new async_release_disconnect_state_hook, called
from both the streaming and non-streaming cancel-on-disconnect paths.
This commit is contained in:
Deepanshu 2026-08-25 22:01:57 -04:00
parent 6c792e659e
commit 243b5cd1ef

View file

@ -29,7 +29,6 @@ from litellm.constants import (
NON_INFERENCE_CALL_TYPES,
RETURN_RAW_MODEL_NAME_METADATA_KEY,
STREAM_SSE_DATA_PREFIX,
STREAM_SSE_KEEPALIVE_PING_BYTES,
UNSAFE_PROXY_RESPONSE_HEADERS,
)
from litellm.integrations.custom_guardrail import CustomGuardrail
@ -204,10 +203,6 @@ _CLIENT_DISCONNECTED_ERROR_INFORMATION: Final[StandardLoggingPayloadErrorInforma
}
def _withheld_provider_output(response: object) -> bool:
return getattr(response, "has_buffered_provider_output", False) is True
def _should_return_raw_model_name(request_data: dict[str, object]) -> bool:
return any(
isinstance(metadata, dict) and metadata.get(RETURN_RAW_MODEL_NAME_METADATA_KEY) is True
@ -3487,9 +3482,8 @@ class ProxyBaseLLMRequestProcessing:
# so a GeneratorExit on client disconnect is raised there and any
# statement after the yield never runs. The slow-path hook is
# awaited above, so a cancellation during it still leaves this
# False and refunds. A keepalive ping carries no provider output,
# so it must not suppress that refund.
delivered_chunk = delivered_chunk or chunk != STREAM_SSE_KEEPALIVE_PING_BYTES
# False and refunds.
delivered_chunk = True
yield serialize_chunk(chunk)
stream_completed = True
except (asyncio.CancelledError, GeneratorExit):
@ -3503,7 +3497,7 @@ class ProxyBaseLLMRequestProcessing:
# only sees GeneratorExit on GC) cannot own the refund.
if not stream_completed:
client_disconnected = True
if not delivered_chunk and not _withheld_provider_output(response):
if not delivered_chunk:
from litellm.proxy.spend_tracking.budget_reservation import (
release_budget_reservation_on_cancel,
)