From a634cb9b62ecfe7e37de8715bf2e2cd763b2a8b2 Mon Sep 17 00:00:00 2001 From: shivam Date: Wed, 29 Jul 2026 00:30:33 +0000 Subject: [PATCH 1/4] fix(responses): initialize completed_response and _hidden_params on the completion bridge iterator LiteLLMCompletionStreamingIterator subclasses ResponsesAPIStreamingIterator but never runs the base constructor, so it lacks completed_response and _hidden_params. When a provider errors before the first content chunk, the router reads completed_response directly in _extract_partial_responses_usage and raises AttributeError, and the missing _hidden_params makes the proxy pre-wrap the iterator for header attachment so the Responses mid-stream fallback path is skipped entirely. Initialize both on the bridge iterator so provider errors surface and fallbacks run. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../streaming_iterator.py | 7 +++- ...st_router_aresponses_streaming_fallback.py | 40 +++++++++++++++++++ 2 files changed, 46 insertions(+), 1 deletion(-) diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index cf69654d15d..2aded792f73 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -67,7 +67,12 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): self.request_input: str | ResponseInputParam = request_input self.responses_api_request: ResponsesAPIOptionalRequestParams = responses_api_request self.custom_llm_provider: str | None = custom_llm_provider - self.litellm_metadata: dict | None = litellm_metadata or {} + self.litellm_metadata = litellm_metadata or {} + self.completed_response: Any | None = None + _wrapper_hidden_params = getattr(litellm_custom_stream_wrapper, "_hidden_params", None) + self._hidden_params: dict[str, Any] = ( + dict(_wrapper_hidden_params) if isinstance(_wrapper_hidden_params, dict) else {} + ) # Store lightweight dict snapshots for stream_chunk_builder to reduce # repeated Pydantic attribute access in end-of-stream assembly. self.collected_chat_completion_chunks: list[dict[str, Any]] = [] diff --git a/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py b/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py index 2fb7bdfceb5..a1ac020e209 100644 --- a/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py +++ b/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py @@ -483,3 +483,43 @@ async def test_aresponses_client_error_event_skips_fallback(): assert exc_info.value.status_code == 400 mock_fallback.assert_not_awaited() + + +# -------- regression: real bridge iterator honors the base contract (LIT-4912) -------- + + +def test_extract_partial_responses_usage_real_bridge_iterator_pre_first_chunk(): + """ + Regression for LIT-4912. + + When a provider errors before the first content chunk, the completion-bridge + iterator reaches _extract_partial_responses_usage with no collected chunks. + A real LiteLLMCompletionStreamingIterator (not a MagicMock standing in for + it) must expose the base-class contract so usage extraction returns None + instead of raising AttributeError on completed_response, and must carry + _hidden_params so the router does not pre-wrap it for header attachment and + thereby skip the mid-stream fallback path entirely. + """ + from litellm.responses.litellm_completion_transformation.streaming_iterator import ( + LiteLLMCompletionStreamingIterator, + ) + from litellm.responses.streaming_iterator import BaseResponsesAPIStreamingIterator + + class _FakeStreamWrapper: + def __init__(self) -> None: + self.logging_obj = MagicMock() + self._hidden_params = {"model_id": "deployment-123"} + + iterator = LiteLLMCompletionStreamingIterator( + model="anthropic/claude-3-5-sonnet-latest", + litellm_custom_stream_wrapper=_FakeStreamWrapper(), + request_input="hi", + responses_api_request={}, + ) + + assert isinstance(iterator, BaseResponsesAPIStreamingIterator) + assert iterator.completed_response is None + assert iterator._hidden_params == {"model_id": "deployment-123"} + assert not iterator.collected_chat_completion_chunks + + assert Router._extract_partial_responses_usage(iterator) is None From 32e43221b93e094597eaa28a8fec3d4edc507581 Mon Sep 17 00:00:00 2001 From: yassin Date: Wed, 2 Sep 2026 15:48:31 +0000 Subject: [PATCH 2/4] fix: preserve bridge iterator hidden params Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../streaming_iterator.py | 7 ++-- ...st_router_aresponses_streaming_fallback.py | 39 +++++++++---------- 2 files changed, 21 insertions(+), 25 deletions(-) diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index bd934ae6899..ad6cf09aaae 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -89,10 +89,9 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): self.request_input: str | ResponseInputParam = request_input self.responses_api_request: ResponsesAPIOptionalRequestParams = responses_api_request self.custom_llm_provider: str | None = custom_llm_provider - self.litellm_metadata = litellm_metadata or {} - self.completed_response: Any | None = None - _wrapper_hidden_params = getattr(litellm_custom_stream_wrapper, "_hidden_params", None) - self._hidden_params: dict[str, Any] = ( + self.litellm_metadata: dict | None = litellm_metadata or {} + _wrapper_hidden_params: Final = getattr(litellm_custom_stream_wrapper, "_hidden_params", None) + self._hidden_params: dict[str, object] = ( dict(_wrapper_hidden_params) if isinstance(_wrapper_hidden_params, dict) else {} ) # Store lightweight dict snapshots for stream_chunk_builder to reduce diff --git a/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py b/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py index 5446334e7fb..08112c2c04c 100644 --- a/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py +++ b/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py @@ -10,12 +10,12 @@ Targets the four helpers introduced on Router: - _aresponses_streaming_iterator """ -from typing import Any, AsyncIterator, List +from collections.abc import AsyncIterator +from typing import Any from unittest.mock import AsyncMock, MagicMock, patch import pytest - from litellm import Router from litellm.types.llms.openai import ( ResponseAPIUsage, @@ -46,9 +46,7 @@ def _make_router() -> Router: ) -def _make_completed_event( - input_tokens: int, output_tokens: int, total_tokens: int -) -> ResponseCompletedEvent: +def _make_completed_event(input_tokens: int, output_tokens: int, total_tokens: int) -> ResponseCompletedEvent: response = ResponsesAPIResponse.model_construct( usage=ResponseAPIUsage( input_tokens=input_tokens, @@ -145,9 +143,7 @@ def test_combine_responses_fallback_usage_passthrough_for_unknown_event(): def test_build_responses_continuation_input_from_string(): - out = Router._build_responses_continuation_input( - "Hello world", "partial assistant text" - ) + out = Router._build_responses_continuation_input("Hello world", "partial assistant text") assert len(out) == 3 assert out[0]["role"] == "user" assert out[0]["content"][0]["text"] == "Hello world" @@ -157,7 +153,7 @@ def test_build_responses_continuation_input_from_string(): def test_build_responses_continuation_input_from_list_preserves_items(): - existing: List[Any] = [ + existing: list[Any] = [ { "type": "message", "role": "user", @@ -227,9 +223,7 @@ async def test_aresponses_streaming_iterator_passthrough(): router = _make_router() source = _FakeSource() - wrapper = await router._aresponses_streaming_iterator( - source, initial_kwargs={"model": "primary"} - ) + wrapper = await router._aresponses_streaming_iterator(source, initial_kwargs={"model": "primary"}) assert isinstance(wrapper, BaseResponsesAPIStreamingIterator) collected = [ev async for ev in wrapper] @@ -276,15 +270,18 @@ async def test_aresponses_with_streaming_fallbacks_wraps_streaming_iterator(): async def fake_original(**_kwargs): return streaming_iter - with patch.object( - router, - "_ageneric_api_call_with_fallbacks", - new=AsyncMock(return_value=streaming_iter), - ), patch.object( - router, - "_aresponses_streaming_iterator", - new=AsyncMock(return_value=wrapped), - ) as mock_wrap: + with ( + patch.object( + router, + "_ageneric_api_call_with_fallbacks", + new=AsyncMock(return_value=streaming_iter), + ), + patch.object( + router, + "_aresponses_streaming_iterator", + new=AsyncMock(return_value=wrapped), + ) as mock_wrap, + ): out = await router._aresponses_with_streaming_fallbacks( original_function=fake_original, model="primary", From 730d13ccd19eb4b1abba67ef9ee35e77dc885262 Mon Sep 17 00:00:00 2001 From: yassin Date: Wed, 2 Sep 2026 16:00:56 +0000 Subject: [PATCH 3/4] fix: satisfy lint and coverage gates Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../streaming_iterator.py | 6 ++++-- .../test_router_aresponses_streaming_fallback.py | 12 ++++++++++++ 2 files changed, 16 insertions(+), 2 deletions(-) diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index ad6cf09aaae..f71ff3c6660 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -91,8 +91,10 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): self.custom_llm_provider: str | None = custom_llm_provider self.litellm_metadata: dict | None = litellm_metadata or {} _wrapper_hidden_params: Final = getattr(litellm_custom_stream_wrapper, "_hidden_params", None) - self._hidden_params: dict[str, object] = ( - dict(_wrapper_hidden_params) if isinstance(_wrapper_hidden_params, dict) else {} + self._hidden_params: dict[str, object] = ( # mutable-ok: downstream response metadata requires a mutable dict + dict(_wrapper_hidden_params) # mutable-ok: downstream response metadata requires a mutable dict + if isinstance(_wrapper_hidden_params, dict) + else {} # mutable-ok: downstream response metadata requires a mutable dict ) # Store lightweight dict snapshots for stream_chunk_builder to reduce # repeated Pydantic attribute access in end-of-stream assembly. diff --git a/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py b/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py index 08112c2c04c..a538db85146 100644 --- a/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py +++ b/tests/router_unit_tests/test_router_aresponses_streaming_fallback.py @@ -533,6 +533,10 @@ def test_extract_partial_responses_usage_real_bridge_iterator_pre_first_chunk(): self.logging_obj = MagicMock() self._hidden_params = {"model_id": "deployment-123"} + class _FakeStreamWrapperWithoutHiddenParams: + def __init__(self) -> None: + self.logging_obj = MagicMock() + iterator = LiteLLMCompletionStreamingIterator( model="anthropic/claude-3-5-sonnet-latest", litellm_custom_stream_wrapper=_FakeStreamWrapper(), @@ -545,4 +549,12 @@ def test_extract_partial_responses_usage_real_bridge_iterator_pre_first_chunk(): assert iterator._hidden_params == {"model_id": "deployment-123"} assert not iterator.collected_chat_completion_chunks + iterator_without_hidden_params = LiteLLMCompletionStreamingIterator( + model="anthropic/claude-3-5-sonnet-latest", + litellm_custom_stream_wrapper=_FakeStreamWrapperWithoutHiddenParams(), + request_input="hi", + responses_api_request={}, + ) + + assert iterator_without_hidden_params._hidden_params == {} assert Router._extract_partial_responses_usage(iterator) is None From 658b7781095a331a0ed1b54b45d938041a0b8539 Mon Sep 17 00:00:00 2001 From: yassin Date: Wed, 2 Sep 2026 16:10:37 +0000 Subject: [PATCH 4/4] style: single mutable-ok on bridge hidden params Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../streaming_iterator.py | 8 ++------ 1 file changed, 2 insertions(+), 6 deletions(-) diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index f71ff3c6660..fc1c1c74bc6 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -90,12 +90,8 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): self.responses_api_request: ResponsesAPIOptionalRequestParams = responses_api_request self.custom_llm_provider: str | None = custom_llm_provider self.litellm_metadata: dict | None = litellm_metadata or {} - _wrapper_hidden_params: Final = getattr(litellm_custom_stream_wrapper, "_hidden_params", None) - self._hidden_params: dict[str, object] = ( # mutable-ok: downstream response metadata requires a mutable dict - dict(_wrapper_hidden_params) # mutable-ok: downstream response metadata requires a mutable dict - if isinstance(_wrapper_hidden_params, dict) - else {} # mutable-ok: downstream response metadata requires a mutable dict - ) + _hp: Final = getattr(litellm_custom_stream_wrapper, "_hidden_params", None) + self._hidden_params: dict[str, object] = dict(_hp) if isinstance(_hp, dict) else {} # mutable-ok: router header # Store lightweight dict snapshots for stream_chunk_builder to reduce # repeated Pydantic attribute access in end-of-stream assembly. self.collected_chat_completion_chunks: list[dict[str, object]] = []