From 58b360cf464f81194104dbb728b76b7e7112a652 Mon Sep 17 00:00:00 2001 From: rain <1504569896@qq.com> Date: Mon, 14 Sep 2026 02:45:32 +0800 Subject: [PATCH 1/8] fix(responses bridge): rebuild empty output[] from streamed output_item.done events Some providers (e.g. ChatGPT subscription backend) emit a terminal response.completed event whose response.output array is empty even though the answer was fully streamed via response.output_item.done events. The Responses-to-completion bridge then raises 'Unknown items in responses API response: []' and the caller loses the answer (#41009). - BaseResponsesAPIStreamingIterator now records output_item.done items keyed by output_index and exposes get_streamed_output_items(). - ResponsesToCompletionBridgeHandler._collect_response_from_stream (sync+async) rebuilds response.output from those items when the terminal payload has an empty output array, instead of failing downstream. Closes #41009 --- .../handler.py | 13 +++ litellm/responses/streaming_iterator.py | 22 +++++ ...est_responses_api_bridge_recover_output.py | 91 +++++++++++++++++++ 3 files changed, 126 insertions(+) create mode 100644 tests/test_litellm/test_responses_api_bridge_recover_output.py diff --git a/litellm/completion_extras/litellm_responses_transformation/handler.py b/litellm/completion_extras/litellm_responses_transformation/handler.py index f494d6610a1..0d61914c2ba 100644 --- a/litellm/completion_extras/litellm_responses_transformation/handler.py +++ b/litellm/completion_extras/litellm_responses_transformation/handler.py @@ -87,6 +87,14 @@ class ResponsesToCompletionBridgeHandler: response: Final = self._coerce_response_object(response_obj, hidden_params) if not isinstance(response, ResponsesAPIResponse): raise ValueError("Stream completed response is invalid") + # Some providers emit a terminal response.completed event with an empty output + # array even though output_item.done events streamed the full answer. Rebuild + # the output from those streamed items before failing on empty choices. + get_streamed_items: Final = getattr(stream_iter, "get_streamed_output_items", None) + if len(response.output) == 0 and callable(get_streamed_items): + streamed_items: Final = get_streamed_items() + if streamed_items: + response.output = streamed_items # type: ignore[assignment] return response async def _collect_response_from_stream_async(self, stream_iter: Any) -> "ResponsesAPIResponse": @@ -102,6 +110,11 @@ class ResponsesToCompletionBridgeHandler: response: Final = self._coerce_response_object(response_obj, hidden_params) if not isinstance(response, ResponsesAPIResponse): raise ValueError("Stream completed response is invalid") + get_streamed_items: Final = getattr(stream_iter, "get_streamed_output_items", None) + if len(response.output) == 0 and callable(get_streamed_items): + streamed_items: Final = get_streamed_items() + if streamed_items: + response.output = streamed_items # type: ignore[assignment] return response def validate_input_kwargs(self, kwargs: dict) -> ResponsesToCompletionBridgeHandlerInputKwargs: diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index 195214b077c..eafe0ae5ffd 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -293,6 +293,10 @@ class BaseResponsesAPIStreamingIterator: self._completed_response_logged = False self._completed_response_cache_hit: bool | None = None self._persist_completed_response_before_logging = True + # output_item.done events seen during the stream, keyed by output_index. + # Some providers emit a terminal response.completed event with an empty + # output array even though items were streamed; these are used to rebuild it. + self._streamed_output_items: dict[int, object] = {} self._stream_created_time: float = time.time() # track request context for hooks @@ -400,6 +404,14 @@ class BaseResponsesAPIStreamingIterator: custom_llm_provider=self.custom_llm_provider, model_id=_stream_model_id, ) + if _event_type == ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE: + _done_item: Final[object] = getattr(openai_responses_api_chunk, "item", None) + if _done_item is not None: + _output_index: Final = getattr(openai_responses_api_chunk, "output_index", None) + _index: Final = ( + _output_index if isinstance(_output_index, int) else len(self._streamed_output_items) + ) + self._streamed_output_items.setdefault(_index, _done_item) elif _event_type == ResponsesAPIStreamEvents.OUTPUT_TEXT_ANNOTATION_ADDED: _annotation: Final[object] = getattr(openai_responses_api_chunk, "annotation", None) if _annotation is not None: @@ -502,6 +514,16 @@ class BaseResponsesAPIStreamingIterator: self._handle_failure(e) raise + def get_streamed_output_items(self) -> list[object]: + """ + Output items received via response.output_item.done events, ordered by output_index. + + Used to rebuild a terminal response.completed payload whose output array is empty. + """ + if not self._streamed_output_items: + return [] + return [item for _, item in sorted(self._streamed_output_items.items())] + def _log_completed_response(self, *, is_async: bool) -> None: if self._completed_response_logged: return diff --git a/tests/test_litellm/test_responses_api_bridge_recover_output.py b/tests/test_litellm/test_responses_api_bridge_recover_output.py new file mode 100644 index 00000000000..7b38e6a6ce5 --- /dev/null +++ b/tests/test_litellm/test_responses_api_bridge_recover_output.py @@ -0,0 +1,91 @@ +from litellm.completion_extras.litellm_responses_transformation.handler import ( + ResponsesToCompletionBridgeHandler, +) +from litellm.types.llms.openai import ResponsesAPIResponse + + +class _CompletedEvent: + def __init__(self, response): + self.response = response + + +class _FakeResponsesStream: + """ + Simulates a Responses API stream whose terminal response.completed event carries + an empty output array even though output_item.done events streamed the answer + (see issue #41009). + """ + + def __init__(self, response, streamed_items): + self._emitted = False + self._response = response + self._streamed_items = streamed_items + self.completed_response = None + self._hidden_params = {"headers": {"x-test": "1"}} + + def __iter__(self): + return self + + def __next__(self): + if not self._emitted: + self._emitted = True + self.completed_response = _CompletedEvent(self._response) + return {"type": "response.completed"} + raise StopIteration + + def get_streamed_output_items(self): + return self._streamed_items + + +def test_collect_response_from_stream_rebuilds_empty_output_from_streamed_items(): + handler = ResponsesToCompletionBridgeHandler() + response = ResponsesAPIResponse.model_construct( + id="resp-1", + created_at=0, + output=[], + object="response", + model="gpt-5.6-sol", + ) + streamed_items = [ + { + "id": "msg_1", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + {"type": "output_text", "text": "Hi! How can I help?", "annotations": []} + ], + } + ] + stream = _FakeResponsesStream(response, streamed_items) + + collected = handler._collect_response_from_stream(stream) + + assert len(collected.output) == 1 + assert "Hi! How can I help?" in str(collected.output[0]) + + +def test_collect_response_from_stream_keeps_nonempty_output_untouched(): + handler = ResponsesToCompletionBridgeHandler() + output = [ + { + "id": "msg_0", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "from terminal event", "annotations": []}], + } + ] + response = ResponsesAPIResponse.model_construct( + id="resp-2", + created_at=0, + output=output, + object="response", + model="gpt-5.2", + ) + stream = _FakeResponsesStream(response, [{"id": "msg_1", "type": "message"}]) + + collected = handler._collect_response_from_stream(stream) + + assert len(collected.output) == 1 + assert "from terminal event" in str(collected.output[0]) From c147c9f96908f4f65440ec4009c725368812d38f Mon Sep 17 00:00:00 2001 From: rain <1504569896@qq.com> Date: Mon, 14 Sep 2026 19:32:04 +0800 Subject: [PATCH 2/8] fix(review): collision-safe streamed index, named suppression, merge tests into mapped file Review follow-up for #41014: - Replace the len()-based fallback index with max(known)+1 so a missing or duplicate output_index can never collide with a real index and silently discard a streamed item. - Replace the banned '# type: ignore' with a named pyright suppression (reportAssignmentType) carrying a reason, per repo directive (LIT009). - Move the regression tests into the existing mapped test file (tests/test_litellm/test_responses_api_bridge_non_stream.py) instead of a separate file, and add the missing async coverage for _collect_response_from_stream_async. --- .../handler.py | 4 +- litellm/responses/streaming_iterator.py | 18 +++- .../test_responses_api_bridge_non_stream.py | 80 ++++++++++++++++ ...est_responses_api_bridge_recover_output.py | 91 ------------------- 4 files changed, 96 insertions(+), 97 deletions(-) delete mode 100644 tests/test_litellm/test_responses_api_bridge_recover_output.py diff --git a/litellm/completion_extras/litellm_responses_transformation/handler.py b/litellm/completion_extras/litellm_responses_transformation/handler.py index 0d61914c2ba..b2567b558b1 100644 --- a/litellm/completion_extras/litellm_responses_transformation/handler.py +++ b/litellm/completion_extras/litellm_responses_transformation/handler.py @@ -94,7 +94,7 @@ class ResponsesToCompletionBridgeHandler: if len(response.output) == 0 and callable(get_streamed_items): streamed_items: Final = get_streamed_items() if streamed_items: - response.output = streamed_items # type: ignore[assignment] + response.output = streamed_items # pyright: ignore[reportAssignmentType] # streamed items are the typed SSE output items return response async def _collect_response_from_stream_async(self, stream_iter: Any) -> "ResponsesAPIResponse": @@ -114,7 +114,7 @@ class ResponsesToCompletionBridgeHandler: if len(response.output) == 0 and callable(get_streamed_items): streamed_items: Final = get_streamed_items() if streamed_items: - response.output = streamed_items # type: ignore[assignment] + response.output = streamed_items # pyright: ignore[reportAssignmentType] # streamed items are the typed SSE output items return response def validate_input_kwargs(self, kwargs: dict) -> ResponsesToCompletionBridgeHandlerInputKwargs: diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index eafe0ae5ffd..8c7ba9350d8 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -408,10 +408,20 @@ class BaseResponsesAPIStreamingIterator: _done_item: Final[object] = getattr(openai_responses_api_chunk, "item", None) if _done_item is not None: _output_index: Final = getattr(openai_responses_api_chunk, "output_index", None) - _index: Final = ( - _output_index if isinstance(_output_index, int) else len(self._streamed_output_items) - ) - self._streamed_output_items.setdefault(_index, _done_item) + if ( + isinstance(_output_index, int) + and not isinstance(_output_index, bool) + and _output_index not in self._streamed_output_items + ): + self._streamed_output_items[_output_index] = _done_item + else: + # missing/invalid/duplicate index: append after the highest known + # index so the fallback can never collide with a real index and + # silently discard a streamed item + _fallback_index: Final = ( + max(self._streamed_output_items.keys(), default=-1) + 1 + ) + self._streamed_output_items[_fallback_index] = _done_item elif _event_type == ResponsesAPIStreamEvents.OUTPUT_TEXT_ANNOTATION_ADDED: _annotation: Final[object] = getattr(openai_responses_api_chunk, "annotation", None) if _annotation is not None: diff --git a/tests/test_litellm/test_responses_api_bridge_non_stream.py b/tests/test_litellm/test_responses_api_bridge_non_stream.py index 617b2cfc031..c438a409f0a 100644 --- a/tests/test_litellm/test_responses_api_bridge_non_stream.py +++ b/tests/test_litellm/test_responses_api_bridge_non_stream.py @@ -65,6 +65,86 @@ def test_should_collect_response_from_stream(): assert collected.id == "resp-1" assert collected._hidden_params.get("headers") == {"x-test": "1"} +class _RecoverableResponsesStream(_FakeResponsesStream): + """Terminal response.completed carries an empty output array even though + output_item.done events streamed the full answer (issue #41009).""" + + def __init__(self, response, streamed_items): + super().__init__(response) + self._streamed_items = streamed_items + + def get_streamed_output_items(self): + return self._streamed_items + + +class _RecoverableAsyncResponsesStream(_RecoverableResponsesStream): + def __aiter__(self): + return self + + async def __anext__(self): + if not self._emitted: + self._emitted = True + self.completed_response = _CompletedEvent(self._response) + return {"type": "response.completed"} + raise StopAsyncIteration + + +def test_collect_response_from_stream_rebuilds_empty_output_from_streamed_items(): + handler = ResponsesToCompletionBridgeHandler() + response = ResponsesAPIResponse.model_construct( + id="resp-recover", + created_at=0, + output=[], + object="response", + model="gpt-5.6-sol", + ) + streamed_items = [ + { + "id": "msg_1", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + {"type": "output_text", "text": "Hi! How can I help?", "annotations": []} + ], + } + ] + stream = _RecoverableResponsesStream(response, streamed_items) + + collected = handler._collect_response_from_stream(stream) + + assert len(collected.output) == 1 + assert "Hi! How can I help?" in str(collected.output[0]) + + +@pytest.mark.asyncio +async def test_collect_response_from_stream_async_rebuilds_empty_output_from_streamed_items(): + handler = ResponsesToCompletionBridgeHandler() + response = ResponsesAPIResponse.model_construct( + id="resp-recover-async", + created_at=0, + output=[], + object="response", + model="gpt-5.6-sol", + ) + streamed_items = [ + { + "id": "msg_1", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + {"type": "output_text", "text": "Hi! How can I help?", "annotations": []} + ], + } + ] + stream = _RecoverableAsyncResponsesStream(response, streamed_items) + + collected = await handler._collect_response_from_stream_async(stream) + + assert len(collected.output) == 1 + assert "Hi! How can I help?" in str(collected.output[0]) + def create_mock_completion_response( model: str = "gpt-4", diff --git a/tests/test_litellm/test_responses_api_bridge_recover_output.py b/tests/test_litellm/test_responses_api_bridge_recover_output.py deleted file mode 100644 index 7b38e6a6ce5..00000000000 --- a/tests/test_litellm/test_responses_api_bridge_recover_output.py +++ /dev/null @@ -1,91 +0,0 @@ -from litellm.completion_extras.litellm_responses_transformation.handler import ( - ResponsesToCompletionBridgeHandler, -) -from litellm.types.llms.openai import ResponsesAPIResponse - - -class _CompletedEvent: - def __init__(self, response): - self.response = response - - -class _FakeResponsesStream: - """ - Simulates a Responses API stream whose terminal response.completed event carries - an empty output array even though output_item.done events streamed the answer - (see issue #41009). - """ - - def __init__(self, response, streamed_items): - self._emitted = False - self._response = response - self._streamed_items = streamed_items - self.completed_response = None - self._hidden_params = {"headers": {"x-test": "1"}} - - def __iter__(self): - return self - - def __next__(self): - if not self._emitted: - self._emitted = True - self.completed_response = _CompletedEvent(self._response) - return {"type": "response.completed"} - raise StopIteration - - def get_streamed_output_items(self): - return self._streamed_items - - -def test_collect_response_from_stream_rebuilds_empty_output_from_streamed_items(): - handler = ResponsesToCompletionBridgeHandler() - response = ResponsesAPIResponse.model_construct( - id="resp-1", - created_at=0, - output=[], - object="response", - model="gpt-5.6-sol", - ) - streamed_items = [ - { - "id": "msg_1", - "type": "message", - "status": "completed", - "role": "assistant", - "content": [ - {"type": "output_text", "text": "Hi! How can I help?", "annotations": []} - ], - } - ] - stream = _FakeResponsesStream(response, streamed_items) - - collected = handler._collect_response_from_stream(stream) - - assert len(collected.output) == 1 - assert "Hi! How can I help?" in str(collected.output[0]) - - -def test_collect_response_from_stream_keeps_nonempty_output_untouched(): - handler = ResponsesToCompletionBridgeHandler() - output = [ - { - "id": "msg_0", - "type": "message", - "status": "completed", - "role": "assistant", - "content": [{"type": "output_text", "text": "from terminal event", "annotations": []}], - } - ] - response = ResponsesAPIResponse.model_construct( - id="resp-2", - created_at=0, - output=output, - object="response", - model="gpt-5.2", - ) - stream = _FakeResponsesStream(response, [{"id": "msg_1", "type": "message"}]) - - collected = handler._collect_response_from_stream(stream) - - assert len(collected.output) == 1 - assert "from terminal event" in str(collected.output[0]) From 7c568dd6c30c5033cf801fc89e184b454877e26d Mon Sep 17 00:00:00 2001 From: rain <1504569896@qq.com> Date: Mon, 14 Sep 2026 20:03:03 +0800 Subject: [PATCH 3/8] style: apply ruff format to responses streaming bridge files --- litellm/responses/streaming_iterator.py | 4 +--- .../test_litellm/test_responses_api_bridge_non_stream.py | 9 +++------ 2 files changed, 4 insertions(+), 9 deletions(-) diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index 8c7ba9350d8..48d5af58aad 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -418,9 +418,7 @@ class BaseResponsesAPIStreamingIterator: # missing/invalid/duplicate index: append after the highest known # index so the fallback can never collide with a real index and # silently discard a streamed item - _fallback_index: Final = ( - max(self._streamed_output_items.keys(), default=-1) + 1 - ) + _fallback_index: Final = max(self._streamed_output_items.keys(), default=-1) + 1 self._streamed_output_items[_fallback_index] = _done_item elif _event_type == ResponsesAPIStreamEvents.OUTPUT_TEXT_ANNOTATION_ADDED: _annotation: Final[object] = getattr(openai_responses_api_chunk, "annotation", None) diff --git a/tests/test_litellm/test_responses_api_bridge_non_stream.py b/tests/test_litellm/test_responses_api_bridge_non_stream.py index c438a409f0a..358134fb5e0 100644 --- a/tests/test_litellm/test_responses_api_bridge_non_stream.py +++ b/tests/test_litellm/test_responses_api_bridge_non_stream.py @@ -65,6 +65,7 @@ def test_should_collect_response_from_stream(): assert collected.id == "resp-1" assert collected._hidden_params.get("headers") == {"x-test": "1"} + class _RecoverableResponsesStream(_FakeResponsesStream): """Terminal response.completed carries an empty output array even though output_item.done events streamed the full answer (issue #41009).""" @@ -104,9 +105,7 @@ def test_collect_response_from_stream_rebuilds_empty_output_from_streamed_items( "type": "message", "status": "completed", "role": "assistant", - "content": [ - {"type": "output_text", "text": "Hi! How can I help?", "annotations": []} - ], + "content": [{"type": "output_text", "text": "Hi! How can I help?", "annotations": []}], } ] stream = _RecoverableResponsesStream(response, streamed_items) @@ -133,9 +132,7 @@ async def test_collect_response_from_stream_async_rebuilds_empty_output_from_str "type": "message", "status": "completed", "role": "assistant", - "content": [ - {"type": "output_text", "text": "Hi! How can I help?", "annotations": []} - ], + "content": [{"type": "output_text", "text": "Hi! How can I help?", "annotations": []}], } ] stream = _RecoverableAsyncResponsesStream(response, streamed_items) From 60261b48dca90e260024a9d3e9720fd2397c1681 Mon Sep 17 00:00:00 2001 From: rain <1504569896@qq.com> Date: Tue, 15 Sep 2026 01:44:31 +0800 Subject: [PATCH 4/8] style: satisfy the type-discipline budget gate on the streamed-items accumulator --- litellm/responses/streaming_iterator.py | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index 48d5af58aad..06fd45f4855 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -296,7 +296,7 @@ class BaseResponsesAPIStreamingIterator: # output_item.done events seen during the stream, keyed by output_index. # Some providers emit a terminal response.completed event with an empty # output array even though items were streamed; these are used to rebuild it. - self._streamed_output_items: dict[int, object] = {} + self._streamed_output_items: dict[int, object] = {} # mutable-ok: streaming accumulator keyed by output_index self._stream_created_time: float = time.time() # track request context for hooks @@ -522,15 +522,17 @@ class BaseResponsesAPIStreamingIterator: self._handle_failure(e) raise - def get_streamed_output_items(self) -> list[object]: + def get_streamed_output_items(self) -> list[object]: # mutable-ok: response.output takes a real list """ Output items received via response.output_item.done events, ordered by output_index. Used to rebuild a terminal response.completed payload whose output array is empty. """ if not self._streamed_output_items: - return [] - return [item for _, item in sorted(self._streamed_output_items.items())] + return [] # mutable-ok: response.output takes a real list + return [ + item for _, item in sorted(self._streamed_output_items.items()) + ] # mutable-ok: response.output takes a real list def _log_completed_response(self, *, is_async: bool) -> None: if self._completed_response_logged: From efc90ede631f6b5529ca013bfaa9ebc2daed85d0 Mon Sep 17 00:00:00 2001 From: rain <1504569896@qq.com> Date: Tue, 15 Sep 2026 01:45:32 +0800 Subject: [PATCH 5/8] style: move suppression to the comprehension line after ruff reflow --- litellm/responses/streaming_iterator.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index 06fd45f4855..3aa1df26110 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -530,9 +530,9 @@ class BaseResponsesAPIStreamingIterator: """ if not self._streamed_output_items: return [] # mutable-ok: response.output takes a real list - return [ + return [ # mutable-ok: response.output takes a real list item for _, item in sorted(self._streamed_output_items.items()) - ] # mutable-ok: response.output takes a real list + ] def _log_completed_response(self, *, is_async: bool) -> None: if self._completed_response_logged: From af1540ac2b0ecb5341b58a09de9d4a01ae474683 Mon Sep 17 00:00:00 2001 From: Rain <152621457+Rainmemery@users.noreply.github.com> Date: Sun, 20 Sep 2026 23:05:12 +0800 Subject: [PATCH 6/8] test(responses): cover streamed-output rebuild and collision-safe fallback --- .../responses/test_streaming_iterator.py | 86 +++++++++++++++++++ 1 file changed, 86 insertions(+) diff --git a/tests/test_litellm/responses/test_streaming_iterator.py b/tests/test_litellm/responses/test_streaming_iterator.py index dbf54ec3b9b..8847b9a5692 100644 --- a/tests/test_litellm/responses/test_streaming_iterator.py +++ b/tests/test_litellm/responses/test_streaming_iterator.py @@ -942,3 +942,89 @@ def test_persist_completed_response_to_cache_survives_an_unserializable_response iterator._persist_completed_response_to_cache(is_async=False) cache.add_cache.assert_not_called() + + +def test_process_chunk_collects_output_item_done_in_order(): + """output_item.done events are keyed by index and returned in order.""" + logging_obj = _logging_obj_for_collector() + iterator = _make_collector_iterator(logging_obj=logging_obj) + + for index, item in enumerate([{"id": "msg_1"}, {"id": "msg_2"}]): + iterator._process_chunk( + json.dumps({"type": "response.output_item.done", "output_index": index, "item": item}) + ) + + assert iterator.get_streamed_output_items() == [{"id": "msg_1"}, {"id": "msg_2"}] + + +def test_process_chunk_collector_fallback_index_avoids_collision(): + """A missing/invalid index appends after the max key, never overwriting an entry.""" + logging_obj = _logging_obj_for_collector() + iterator = _make_collector_iterator(logging_obj=logging_obj) + + iterator._process_chunk( + json.dumps( + {"type": "response.output_item.done", "output_index": 5, "item": {"id": "msg_5"}} + ) + ) + iterator._process_chunk( + json.dumps({"type": "response.output_item.done", "item": {"id": "msg_fallback"}}) + ) + items = iterator.get_streamed_output_items() + assert items == [{"id": "msg_5"}, {"id": "msg_fallback"}] + + +def test_process_chunk_collector_duplicate_index_appends(): + """A repeated index is never dropped: the second item is appended after the max key.""" + logging_obj = _logging_obj_for_collector() + iterator = _make_collector_iterator(logging_obj=logging_obj) + + iterator._process_chunk( + json.dumps( + {"type": "response.output_item.done", "output_index": 1, "item": {"id": "msg_a"}} + ) + ) + iterator._process_chunk( + json.dumps( + {"type": "response.output_item.done", "output_index": 1, "item": {"id": "msg_b"}} + ) + ) + assert iterator.get_streamed_output_items() == [{"id": "msg_a"}, {"id": "msg_b"}] + + +def test_get_streamed_output_items_empty_when_no_done_events(): + logging_obj = _logging_obj_for_collector() + iterator = _make_collector_iterator(logging_obj=logging_obj) + assert iterator.get_streamed_output_items() == [] + + +def _streamed_item_config() -> Mock: + """transform_streaming_response returns output_item.done events with item/index preserved.""" + mock_config = Mock() + mock_config.transform_streaming_response.side_effect = ( + lambda model, parsed_chunk, logging_obj: Mock( + type=ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE, + item=parsed_chunk.get("item"), + output_index=parsed_chunk.get("output_index"), + ) + ) + return mock_config + + +def _make_collector_iterator(*, logging_obj: Mock) -> SyncResponsesAPIStreamingIterator: + return SyncResponsesAPIStreamingIterator( + response=Mock(headers={}), + model="gpt-4o-mini", + responses_api_provider_config=_streamed_item_config(), + logging_obj=logging_obj, + litellm_metadata={}, + custom_llm_provider="openai", + call_type="responses", + ) + + +def _logging_obj_for_collector() -> Mock: + logging_obj = Mock(spec=LiteLLMLoggingObj) + logging_obj.completion_start_time = None + logging_obj.model_call_details = {"litellm_params": {}} + return logging_obj From 6e723882ccbc5f8aa11425a33614446ca6ac1fcc Mon Sep 17 00:00:00 2001 From: rain <1504569896@qq.com> Date: Sun, 20 Sep 2026 22:59:34 +0000 Subject: [PATCH 7/8] test(responses): cover empty-output rebuild on stream completion Cover _collect_response_from_stream and _collect_response_from_stream_async for #41014: when the terminal response carries an empty output array but output_item.done events streamed the full answer, the bridge rebuilds the output from the streamed items. Includes the rebuild path, the non-empty-output short-circuit and the no-streamed-items fallback for the sync collector, plus a rebuild case for the async collector. Signed-off-by: rain <1504569896@qq.com> --- ...itellm_responses_transformation_handler.py | 128 ++++++++++++++++++ 1 file changed, 128 insertions(+) diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_handler.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_handler.py index c5d7ca96a21..0ede5203066 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_handler.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_handler.py @@ -335,3 +335,131 @@ async def test_acompletion_keeps_provider_native_model_id_through_responses( handed_model = responses_call.call_args.kwargs["model"] assert _upstream_model_for(handed_model, custom_llm_provider) == expected_upstream_model +# ------------------------------------------------------------------------ +# #41009/#41014: rebuild empty streamed output from output_item.done events +# ------------------------------------------------------------------------ + + +class _FakeCompletedResponse: + response = { + "id": "resp_41014", + "object": "response", + "created_at": 0, + "status": "completed", + "model": "gpt-5.4", + "output": [], + } + + +class _FakeStreamIter: + def __init__(self, streamed_items): + self.completed_response = _FakeCompletedResponse() + self._hidden_params = None + self._items = streamed_items + + def __iter__(self): + return iter([]) + + def get_streamed_output_items(self): + return self._items + + +class _FakeAsyncStreamIter(_FakeStreamIter): + def __aiter__(self): + return self + + async def __anext__(self): + raise StopAsyncIteration + + +def test_collect_response_from_stream_rebuilds_empty_output_from_streamed_items(): + from litellm.completion_extras.litellm_responses_transformation.handler import ( + ResponsesToCompletionBridgeHandler, + ) + + bridge = ResponsesToCompletionBridgeHandler() + streamed_items = [ + { + "type": "message", + "id": "msg_1", + "role": "assistant", + "content": [{"type": "output_text", "text": "Hello", "annotations": []}], + } + ] + response = bridge._collect_response_from_stream(_FakeStreamIter(streamed_items)) + assert len(response.output) == 1 + assert response.output[0]["id"] == "msg_1" + + +def test_collect_response_from_stream_keeps_nonempty_output_untouched(): + from litellm.completion_extras.litellm_responses_transformation.handler import ( + ResponsesToCompletionBridgeHandler, + ) + + bridge = ResponsesToCompletionBridgeHandler() + + class _NonEmptyCompleted: + response = { + "id": "resp_nonempty", + "object": "response", + "created_at": 0, + "status": "completed", + "model": "gpt-5.4", + "output": [{"type": "message", "id": "msg_0"}], + } + + class _StreamNoItems: + completed_response = _NonEmptyCompleted() + _hidden_params = None + + def __iter__(self): + return iter([]) + + response = bridge._collect_response_from_stream(_StreamNoItems()) + assert len(response.output) == 1 + assert response.output[0]["id"] == "msg_0" + + +def test_collect_response_from_stream_without_streamed_items_keeps_empty_output(): + from litellm.completion_extras.litellm_responses_transformation.handler import ( + ResponsesToCompletionBridgeHandler, + ) + + bridge = ResponsesToCompletionBridgeHandler() + + class _NoStreamedItems: + completed_response = _FakeCompletedResponse() + _hidden_params = None + + def __iter__(self): + return iter([]) + + response = bridge._collect_response_from_stream(_NoStreamedItems()) + assert len(response.output) == 0 + + +def test_collect_response_from_stream_async_rebuilds_empty_output(): + from litellm.completion_extras.litellm_responses_transformation.handler import ( + ResponsesToCompletionBridgeHandler, + ) + from litellm.types.utils import ResponsesAPIResponse + import asyncio + + async def run(): + bridge = ResponsesToCompletionBridgeHandler() + streamed_items = [ + { + "type": "message", + "id": "msg_async_1", + "role": "assistant", + "content": [{"type": "output_text", "text": "Hi", "annotations": []}], + } + ] + response = await bridge._collect_response_from_stream_async(_FakeAsyncStreamIter(streamed_items)) + assert isinstance(response, ResponsesAPIResponse) + assert len(response.output) == 1 + assert response.output[0]["id"] == "msg_async_1" + + asyncio.run(run()) + + From 9467f7e887dc5b6d700a0430df438cb482e0d76e Mon Sep 17 00:00:00 2001 From: rain <1504569896@qq.com> Date: Wed, 23 Sep 2026 18:24:52 +0800 Subject: [PATCH 8/8] fix(responses bridge): suppress the correct pyright rule on output rebuild type_check_gate flagged a net +2 reportAttributeAccessIssue: the assignment response.output = streamed_items is an attribute-access error, but the ignore comments named reportAssignmentType, so nothing was suppressed. Correct the rule name on both call sites. --- .../litellm_responses_transformation/handler.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/handler.py b/litellm/completion_extras/litellm_responses_transformation/handler.py index 151783d00aa..4984fcd1a0e 100644 --- a/litellm/completion_extras/litellm_responses_transformation/handler.py +++ b/litellm/completion_extras/litellm_responses_transformation/handler.py @@ -94,7 +94,7 @@ class ResponsesToCompletionBridgeHandler: if len(response.output) == 0 and callable(get_streamed_items): streamed_items: Final = get_streamed_items() if streamed_items: - response.output = streamed_items # pyright: ignore[reportAssignmentType] # streamed items are the typed SSE output items + response.output = streamed_items # pyright: ignore[reportAttributeAccessIssue] # assigning typed SSE items back to ResponsesAPIResponse.output return response async def _collect_response_from_stream_async(self, stream_iter: AsyncIterable[object]) -> "ResponsesAPIResponse": @@ -114,7 +114,7 @@ class ResponsesToCompletionBridgeHandler: if len(response.output) == 0 and callable(get_streamed_items): streamed_items: Final = get_streamed_items() if streamed_items: - response.output = streamed_items # pyright: ignore[reportAssignmentType] # streamed items are the typed SSE output items + response.output = streamed_items # pyright: ignore[reportAttributeAccessIssue] # assigning typed SSE items back to ResponsesAPIResponse.output return response def validate_input_kwargs(self, kwargs: dict) -> ResponsesToCompletionBridgeHandlerInputKwargs: