diff --git a/litellm/completion_extras/litellm_responses_transformation/handler.py b/litellm/completion_extras/litellm_responses_transformation/handler.py index f494d6610a1..0d61914c2ba 100644 --- a/litellm/completion_extras/litellm_responses_transformation/handler.py +++ b/litellm/completion_extras/litellm_responses_transformation/handler.py @@ -87,6 +87,14 @@ class ResponsesToCompletionBridgeHandler: response: Final = self._coerce_response_object(response_obj, hidden_params) if not isinstance(response, ResponsesAPIResponse): raise ValueError("Stream completed response is invalid") + # Some providers emit a terminal response.completed event with an empty output + # array even though output_item.done events streamed the full answer. Rebuild + # the output from those streamed items before failing on empty choices. + get_streamed_items: Final = getattr(stream_iter, "get_streamed_output_items", None) + if len(response.output) == 0 and callable(get_streamed_items): + streamed_items: Final = get_streamed_items() + if streamed_items: + response.output = streamed_items # type: ignore[assignment] return response async def _collect_response_from_stream_async(self, stream_iter: Any) -> "ResponsesAPIResponse": @@ -102,6 +110,11 @@ class ResponsesToCompletionBridgeHandler: response: Final = self._coerce_response_object(response_obj, hidden_params) if not isinstance(response, ResponsesAPIResponse): raise ValueError("Stream completed response is invalid") + get_streamed_items: Final = getattr(stream_iter, "get_streamed_output_items", None) + if len(response.output) == 0 and callable(get_streamed_items): + streamed_items: Final = get_streamed_items() + if streamed_items: + response.output = streamed_items # type: ignore[assignment] return response def validate_input_kwargs(self, kwargs: dict) -> ResponsesToCompletionBridgeHandlerInputKwargs: diff --git a/litellm/responses/streaming_iterator.py b/litellm/responses/streaming_iterator.py index 195214b077c..eafe0ae5ffd 100644 --- a/litellm/responses/streaming_iterator.py +++ b/litellm/responses/streaming_iterator.py @@ -293,6 +293,10 @@ class BaseResponsesAPIStreamingIterator: self._completed_response_logged = False self._completed_response_cache_hit: bool | None = None self._persist_completed_response_before_logging = True + # output_item.done events seen during the stream, keyed by output_index. + # Some providers emit a terminal response.completed event with an empty + # output array even though items were streamed; these are used to rebuild it. + self._streamed_output_items: dict[int, object] = {} self._stream_created_time: float = time.time() # track request context for hooks @@ -400,6 +404,14 @@ class BaseResponsesAPIStreamingIterator: custom_llm_provider=self.custom_llm_provider, model_id=_stream_model_id, ) + if _event_type == ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE: + _done_item: Final[object] = getattr(openai_responses_api_chunk, "item", None) + if _done_item is not None: + _output_index: Final = getattr(openai_responses_api_chunk, "output_index", None) + _index: Final = ( + _output_index if isinstance(_output_index, int) else len(self._streamed_output_items) + ) + self._streamed_output_items.setdefault(_index, _done_item) elif _event_type == ResponsesAPIStreamEvents.OUTPUT_TEXT_ANNOTATION_ADDED: _annotation: Final[object] = getattr(openai_responses_api_chunk, "annotation", None) if _annotation is not None: @@ -502,6 +514,16 @@ class BaseResponsesAPIStreamingIterator: self._handle_failure(e) raise + def get_streamed_output_items(self) -> list[object]: + """ + Output items received via response.output_item.done events, ordered by output_index. + + Used to rebuild a terminal response.completed payload whose output array is empty. + """ + if not self._streamed_output_items: + return [] + return [item for _, item in sorted(self._streamed_output_items.items())] + def _log_completed_response(self, *, is_async: bool) -> None: if self._completed_response_logged: return diff --git a/tests/test_litellm/test_responses_api_bridge_recover_output.py b/tests/test_litellm/test_responses_api_bridge_recover_output.py new file mode 100644 index 00000000000..7b38e6a6ce5 --- /dev/null +++ b/tests/test_litellm/test_responses_api_bridge_recover_output.py @@ -0,0 +1,91 @@ +from litellm.completion_extras.litellm_responses_transformation.handler import ( + ResponsesToCompletionBridgeHandler, +) +from litellm.types.llms.openai import ResponsesAPIResponse + + +class _CompletedEvent: + def __init__(self, response): + self.response = response + + +class _FakeResponsesStream: + """ + Simulates a Responses API stream whose terminal response.completed event carries + an empty output array even though output_item.done events streamed the answer + (see issue #41009). + """ + + def __init__(self, response, streamed_items): + self._emitted = False + self._response = response + self._streamed_items = streamed_items + self.completed_response = None + self._hidden_params = {"headers": {"x-test": "1"}} + + def __iter__(self): + return self + + def __next__(self): + if not self._emitted: + self._emitted = True + self.completed_response = _CompletedEvent(self._response) + return {"type": "response.completed"} + raise StopIteration + + def get_streamed_output_items(self): + return self._streamed_items + + +def test_collect_response_from_stream_rebuilds_empty_output_from_streamed_items(): + handler = ResponsesToCompletionBridgeHandler() + response = ResponsesAPIResponse.model_construct( + id="resp-1", + created_at=0, + output=[], + object="response", + model="gpt-5.6-sol", + ) + streamed_items = [ + { + "id": "msg_1", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + {"type": "output_text", "text": "Hi! How can I help?", "annotations": []} + ], + } + ] + stream = _FakeResponsesStream(response, streamed_items) + + collected = handler._collect_response_from_stream(stream) + + assert len(collected.output) == 1 + assert "Hi! How can I help?" in str(collected.output[0]) + + +def test_collect_response_from_stream_keeps_nonempty_output_untouched(): + handler = ResponsesToCompletionBridgeHandler() + output = [ + { + "id": "msg_0", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "from terminal event", "annotations": []}], + } + ] + response = ResponsesAPIResponse.model_construct( + id="resp-2", + created_at=0, + output=output, + object="response", + model="gpt-5.2", + ) + stream = _FakeResponsesStream(response, [{"id": "msg_1", "type": "message"}]) + + collected = handler._collect_response_from_stream(stream) + + assert len(collected.output) == 1 + assert "from terminal event" in str(collected.output[0])