fix(passthrough): swallow flush replay errors and map Anthropic overloaded_error to 529 (#29187)

Signed-off-by: Tai An <antai12232931@outlook.com>
This commit is contained in:
Tai An 2026-05-28 12:20:27 -07:00
parent 928f09f8a4
commit cf6252c3e4
2 changed files with 44 additions and 9 deletions

View file

@ -1993,10 +1993,24 @@ class Logging(LiteLLMLoggingBaseClass):
3. Log the complete streaming response (trigger success handler)
This is used for passthrough endpoints
"""
complete_streaming_response = self._flush_passthrough_collected_chunks_helper(
raw_bytes=raw_bytes,
provider_config=provider_config,
)
try:
complete_streaming_response = (
self._flush_passthrough_collected_chunks_helper(
raw_bytes=raw_bytes,
provider_config=provider_config,
)
)
except Exception as e:
# The flush is a logging/spend-tracking path. Provider errors
# surfaced inside buffered chunks (e.g. mid-stream `overloaded_error`
# for Anthropic / Bedrock) must not become unhandled exceptions —
# the upstream stream is already closed and the client response is
# done by the time we get here.
verbose_logger.warning(
"flush_passthrough_collected_chunks: skipping success-logging — "
f"provider config raised during chunk replay: {type(e).__name__}: {e}"
)
return
if complete_streaming_response is not None:
self.success_handler(result=complete_streaming_response)
@ -2007,10 +2021,23 @@ class Logging(LiteLLMLoggingBaseClass):
raw_bytes: List[bytes],
provider_config: "BasePassthroughConfig",
):
complete_streaming_response = self._flush_passthrough_collected_chunks_helper(
raw_bytes=raw_bytes,
provider_config=provider_config,
)
try:
complete_streaming_response = (
self._flush_passthrough_collected_chunks_helper(
raw_bytes=raw_bytes,
provider_config=provider_config,
)
)
except Exception as e:
# See `flush_passthrough_collected_chunks` above — the flush is a
# logging/spend-tracking task scheduled via `asyncio.create_task`,
# so a bare raise here surfaces as `Task exception was never
# retrieved` and gives users no signal about the real cause.
verbose_logger.warning(
"async_flush_passthrough_collected_chunks: skipping success-logging — "
f"provider config raised during chunk replay: {type(e).__name__}: {e}"
)
return
if complete_streaming_response is not None:
await self.async_success_handler(result=complete_streaming_response)

View file

@ -1018,12 +1018,20 @@ class ModelResponseIterator:
elif type_chunk == "error":
"""
{"type":"error","error":{"details":null,"type":"api_error","message":"Internal server error"} }
Anthropic does not return an HTTP status code in the in-stream
error event. Use `error.type` to pick a sensible default so
clients can distinguish retryable conditions: `overloaded_error`
is 529 per Anthropic's public conventions, everything else
stays 500.
"""
_error_dict = chunk.get("error", {}) or {}
message = _error_dict.get("message", None) or str(chunk)
_error_type = _error_dict.get("type", "") or ""
_status_code = 529 if _error_type == "overloaded_error" else 500
raise AnthropicError(
message=message,
status_code=500, # it looks like Anthropic API does not return a status code in the chunk error - default to 500
status_code=_status_code,
)
text, tool_use = self._handle_json_mode_chunk(text=text, tool_use=tool_use)