From cf6252c3e441520174dcaedeadd54effb8d03f91 Mon Sep 17 00:00:00 2001 From: Tai An Date: Thu, 28 May 2026 12:20:27 -0700 Subject: [PATCH] fix(passthrough): swallow flush replay errors and map Anthropic overloaded_error to 529 (#29187) Signed-off-by: Tai An --- litellm/litellm_core_utils/litellm_logging.py | 43 +++++++++++++++---- litellm/llms/anthropic/chat/handler.py | 10 ++++- 2 files changed, 44 insertions(+), 9 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 97266096ef9..aa6254442fb 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -1993,10 +1993,24 @@ class Logging(LiteLLMLoggingBaseClass): 3. Log the complete streaming response (trigger success handler) This is used for passthrough endpoints """ - complete_streaming_response = self._flush_passthrough_collected_chunks_helper( - raw_bytes=raw_bytes, - provider_config=provider_config, - ) + try: + complete_streaming_response = ( + self._flush_passthrough_collected_chunks_helper( + raw_bytes=raw_bytes, + provider_config=provider_config, + ) + ) + except Exception as e: + # The flush is a logging/spend-tracking path. Provider errors + # surfaced inside buffered chunks (e.g. mid-stream `overloaded_error` + # for Anthropic / Bedrock) must not become unhandled exceptions — + # the upstream stream is already closed and the client response is + # done by the time we get here. + verbose_logger.warning( + "flush_passthrough_collected_chunks: skipping success-logging — " + f"provider config raised during chunk replay: {type(e).__name__}: {e}" + ) + return if complete_streaming_response is not None: self.success_handler(result=complete_streaming_response) @@ -2007,10 +2021,23 @@ class Logging(LiteLLMLoggingBaseClass): raw_bytes: List[bytes], provider_config: "BasePassthroughConfig", ): - complete_streaming_response = self._flush_passthrough_collected_chunks_helper( - raw_bytes=raw_bytes, - provider_config=provider_config, - ) + try: + complete_streaming_response = ( + self._flush_passthrough_collected_chunks_helper( + raw_bytes=raw_bytes, + provider_config=provider_config, + ) + ) + except Exception as e: + # See `flush_passthrough_collected_chunks` above — the flush is a + # logging/spend-tracking task scheduled via `asyncio.create_task`, + # so a bare raise here surfaces as `Task exception was never + # retrieved` and gives users no signal about the real cause. + verbose_logger.warning( + "async_flush_passthrough_collected_chunks: skipping success-logging — " + f"provider config raised during chunk replay: {type(e).__name__}: {e}" + ) + return if complete_streaming_response is not None: await self.async_success_handler(result=complete_streaming_response) diff --git a/litellm/llms/anthropic/chat/handler.py b/litellm/llms/anthropic/chat/handler.py index 2fb29b32a61..04845bc9f40 100644 --- a/litellm/llms/anthropic/chat/handler.py +++ b/litellm/llms/anthropic/chat/handler.py @@ -1018,12 +1018,20 @@ class ModelResponseIterator: elif type_chunk == "error": """ {"type":"error","error":{"details":null,"type":"api_error","message":"Internal server error"} } + + Anthropic does not return an HTTP status code in the in-stream + error event. Use `error.type` to pick a sensible default so + clients can distinguish retryable conditions: `overloaded_error` + is 529 per Anthropic's public conventions, everything else + stays 500. """ _error_dict = chunk.get("error", {}) or {} message = _error_dict.get("message", None) or str(chunk) + _error_type = _error_dict.get("type", "") or "" + _status_code = 529 if _error_type == "overloaded_error" else 500 raise AnthropicError( message=message, - status_code=500, # it looks like Anthropic API does not return a status code in the chunk error - default to 500 + status_code=_status_code, ) text, tool_use = self._handle_json_mode_chunk(text=text, tool_use=tool_use)