fix(router): eagerly fetch deferred stream to surface HTTP errors in fallback path

Providers like Vertex AI and Bedrock defer their HTTP call until the first
__anext__ on the returned CustomStreamWrapper (completion_stream=None,
make_call set). Errors raised inside __anext__ (e.g. 429, 503) escape the
_acompletion try/except block, so fail_calls is never incremented, deployment
cooldown does not fire, and the standard fallback chain is bypassed.

Call fetch_stream() on the wrapper before delegating to
_acompletion_streaming_iterator when completion_stream is None and make_call
is set. Any HTTP error now propagates through _acompletion's except block,
increments fail_calls, and enters the normal retry/fallback chain.

Strip Content-Length, Transfer-Encoding, Content-Encoding, and Content-Type
from exception headers at the same point to prevent HTTP framing mismatches
when LiteLLM builds its own error response body.

Add a re-raise guard in _acompletion_streaming_iterator (async and sync paths)
so MidStreamFallbackError with already-generated content re-raises to the
caller instead of silently injecting a continuation prompt into a fresh request
to a fallback model.

Apply logging cleanup in async_function_with_fallbacks_common_utils: use
%s-style formatting and exc_info=True instead of f-strings with
traceback.format_exc().
This commit is contained in:
Deepanshu 2026-07-01 09:08:25 -04:00
parent cbeaf86c8d
commit 0f79f103a5
2 changed files with 393 additions and 525 deletions

View file

@ -2095,6 +2095,9 @@ class Router:
async for item in model_response:
yield item
except MidStreamFallbackError as e:
if not e.is_pre_first_chunk and e.generated_content:
raise
from litellm.main import stream_chunk_builder
complete_response_object = stream_chunk_builder(chunks=model_response.chunks)
@ -2113,24 +2116,7 @@ class Router:
"content_policy_fallbacks", self.content_policy_fallbacks
)
initial_kwargs["original_function"] = self._acompletion
if e.is_pre_first_chunk or not e.generated_content:
# No content was generated before the error (e.g. a
# rate-limit 429 on the very first chunk). Retry with
# the original messages — adding a continuation prompt
# would waste tokens and confuse the model.
initial_kwargs["messages"] = messages
else:
initial_kwargs["messages"] = messages + [
{
"role": "system",
"content": "You are a helpful assistant. You are given a message and you need to respond to it. You are also given a generated content. You need to respond to the message in continuation of the generated content. Do not repeat the same content. Your response should be in continuation of this text: ",
},
{
"role": "assistant",
"content": e.generated_content,
"prefix": True,
},
]
initial_kwargs["messages"] = messages
self._update_kwargs_before_fallbacks(model=model_group, kwargs=initial_kwargs)
fallback_response = await self.async_function_with_fallbacks_common_utils(
e=e,
@ -2650,6 +2636,9 @@ class Router:
for item in model_response:
yield item
except MidStreamFallbackError as e:
if not e.is_pre_first_chunk and e.generated_content:
raise
from litellm.main import stream_chunk_builder
complete_response_object = stream_chunk_builder(chunks=model_response.chunks)
@ -2669,20 +2658,7 @@ class Router:
router_self.content_policy_fallbacks,
)
initial_kwargs["original_function"] = router_self._completion
if e.is_pre_first_chunk or not e.generated_content:
initial_kwargs["messages"] = messages
else:
initial_kwargs["messages"] = messages + [
{
"role": "system",
"content": "You are a helpful assistant. You are given a message and you need to respond to it. You are also given a generated content. You need to respond to the message in continuation of the generated content. Do not repeat the same content. Your response should be in continuation of this text: ",
},
{
"role": "assistant",
"content": e.generated_content,
"prefix": True,
},
]
initial_kwargs["messages"] = messages
router_self._update_kwargs_before_fallbacks(model=model_group, kwargs=initial_kwargs)
fallback_response = router_self.function_with_fallbacks(
**initial_kwargs,
@ -2890,6 +2866,21 @@ class Router:
)
if isinstance(response, CustomStreamWrapper):
if response.completion_stream is None and response.make_call is not None:
try:
await response.fetch_stream()
except Exception as fetch_err:
_headers = getattr(fetch_err, "headers", None)
if isinstance(_headers, dict):
_framing_headers = frozenset(
{"content-length", "transfer-encoding", "content-encoding", "content-type"}
)
setattr(
fetch_err,
"headers",
{k: v for k, v in _headers.items() if k.lower() not in _framing_headers},
)
raise fetch_err
return await self._acompletion_streaming_iterator(
model_response=response,
messages=messages,
@ -6117,7 +6108,7 @@ class Router:
"""
Common utilities for async_function_with_fallbacks
"""
verbose_router_logger.debug("Traceback%s", traceback.format_exc())
verbose_router_logger.debug("Traceback", exc_info=True)
original_exception = e
fallback_model_group = None
original_model_group: str | None = kwargs.get("model") # type: ignore
@ -6333,15 +6324,17 @@ class Router:
except Exception as new_exception:
parent_otel_span = _get_parent_otel_span_from_kwargs(kwargs)
fallback_failure_exception_str = redact_string(str(new_exception))
cooldown_info = await _async_get_cooldown_deployments_with_debug_info(
litellm_router_instance=self,
parent_otel_span=parent_otel_span,
)
verbose_router_logger.error(
"litellm.router.py::async_function_with_fallbacks() - Error occurred while trying to do fallbacks - {}\n{}\n\nDebug Information:\nCooldown Deployments={}".format(
fallback_failure_exception_str,
redact_string(traceback.format_exc()),
await _async_get_cooldown_deployments_with_debug_info(
litellm_router_instance=self,
parent_otel_span=parent_otel_span,
),
)
"litellm.router.py::async_function_with_fallbacks() - "
"Error occurred while trying to do fallbacks - %s\n"
"Debug Information:\nCooldown Deployments=%s",
fallback_failure_exception_str,
cooldown_info,
exc_info=True,
)
if hasattr(original_exception, "message") and litellm.expose_router_debug_in_errors:

File diff suppressed because it is too large Load diff