mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(responses bridge): rebuild empty output[] from streamed output_item.done events
Some providers (e.g. ChatGPT subscription backend) emit a terminal response.completed event whose response.output array is empty even though the answer was fully streamed via response.output_item.done events. The Responses-to-completion bridge then raises 'Unknown items in responses API response: []' and the caller loses the answer (#41009). - BaseResponsesAPIStreamingIterator now records output_item.done items keyed by output_index and exposes get_streamed_output_items(). - ResponsesToCompletionBridgeHandler._collect_response_from_stream (sync+async) rebuilds response.output from those items when the terminal payload has an empty output array, instead of failing downstream. Closes #41009
This commit is contained in:
parent
e484a7c89c
commit
58b360cf46
3 changed files with 126 additions and 0 deletions
|
|
@ -87,6 +87,14 @@ class ResponsesToCompletionBridgeHandler:
|
|||
response: Final = self._coerce_response_object(response_obj, hidden_params)
|
||||
if not isinstance(response, ResponsesAPIResponse):
|
||||
raise ValueError("Stream completed response is invalid")
|
||||
# Some providers emit a terminal response.completed event with an empty output
|
||||
# array even though output_item.done events streamed the full answer. Rebuild
|
||||
# the output from those streamed items before failing on empty choices.
|
||||
get_streamed_items: Final = getattr(stream_iter, "get_streamed_output_items", None)
|
||||
if len(response.output) == 0 and callable(get_streamed_items):
|
||||
streamed_items: Final = get_streamed_items()
|
||||
if streamed_items:
|
||||
response.output = streamed_items # type: ignore[assignment]
|
||||
return response
|
||||
|
||||
async def _collect_response_from_stream_async(self, stream_iter: Any) -> "ResponsesAPIResponse":
|
||||
|
|
@ -102,6 +110,11 @@ class ResponsesToCompletionBridgeHandler:
|
|||
response: Final = self._coerce_response_object(response_obj, hidden_params)
|
||||
if not isinstance(response, ResponsesAPIResponse):
|
||||
raise ValueError("Stream completed response is invalid")
|
||||
get_streamed_items: Final = getattr(stream_iter, "get_streamed_output_items", None)
|
||||
if len(response.output) == 0 and callable(get_streamed_items):
|
||||
streamed_items: Final = get_streamed_items()
|
||||
if streamed_items:
|
||||
response.output = streamed_items # type: ignore[assignment]
|
||||
return response
|
||||
|
||||
def validate_input_kwargs(self, kwargs: dict) -> ResponsesToCompletionBridgeHandlerInputKwargs:
|
||||
|
|
|
|||
|
|
@ -293,6 +293,10 @@ class BaseResponsesAPIStreamingIterator:
|
|||
self._completed_response_logged = False
|
||||
self._completed_response_cache_hit: bool | None = None
|
||||
self._persist_completed_response_before_logging = True
|
||||
# output_item.done events seen during the stream, keyed by output_index.
|
||||
# Some providers emit a terminal response.completed event with an empty
|
||||
# output array even though items were streamed; these are used to rebuild it.
|
||||
self._streamed_output_items: dict[int, object] = {}
|
||||
self._stream_created_time: float = time.time()
|
||||
|
||||
# track request context for hooks
|
||||
|
|
@ -400,6 +404,14 @@ class BaseResponsesAPIStreamingIterator:
|
|||
custom_llm_provider=self.custom_llm_provider,
|
||||
model_id=_stream_model_id,
|
||||
)
|
||||
if _event_type == ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE:
|
||||
_done_item: Final[object] = getattr(openai_responses_api_chunk, "item", None)
|
||||
if _done_item is not None:
|
||||
_output_index: Final = getattr(openai_responses_api_chunk, "output_index", None)
|
||||
_index: Final = (
|
||||
_output_index if isinstance(_output_index, int) else len(self._streamed_output_items)
|
||||
)
|
||||
self._streamed_output_items.setdefault(_index, _done_item)
|
||||
elif _event_type == ResponsesAPIStreamEvents.OUTPUT_TEXT_ANNOTATION_ADDED:
|
||||
_annotation: Final[object] = getattr(openai_responses_api_chunk, "annotation", None)
|
||||
if _annotation is not None:
|
||||
|
|
@ -502,6 +514,16 @@ class BaseResponsesAPIStreamingIterator:
|
|||
self._handle_failure(e)
|
||||
raise
|
||||
|
||||
def get_streamed_output_items(self) -> list[object]:
|
||||
"""
|
||||
Output items received via response.output_item.done events, ordered by output_index.
|
||||
|
||||
Used to rebuild a terminal response.completed payload whose output array is empty.
|
||||
"""
|
||||
if not self._streamed_output_items:
|
||||
return []
|
||||
return [item for _, item in sorted(self._streamed_output_items.items())]
|
||||
|
||||
def _log_completed_response(self, *, is_async: bool) -> None:
|
||||
if self._completed_response_logged:
|
||||
return
|
||||
|
|
|
|||
|
|
@ -0,0 +1,91 @@
|
|||
from litellm.completion_extras.litellm_responses_transformation.handler import (
|
||||
ResponsesToCompletionBridgeHandler,
|
||||
)
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
|
||||
|
||||
class _CompletedEvent:
|
||||
def __init__(self, response):
|
||||
self.response = response
|
||||
|
||||
|
||||
class _FakeResponsesStream:
|
||||
"""
|
||||
Simulates a Responses API stream whose terminal response.completed event carries
|
||||
an empty output array even though output_item.done events streamed the answer
|
||||
(see issue #41009).
|
||||
"""
|
||||
|
||||
def __init__(self, response, streamed_items):
|
||||
self._emitted = False
|
||||
self._response = response
|
||||
self._streamed_items = streamed_items
|
||||
self.completed_response = None
|
||||
self._hidden_params = {"headers": {"x-test": "1"}}
|
||||
|
||||
def __iter__(self):
|
||||
return self
|
||||
|
||||
def __next__(self):
|
||||
if not self._emitted:
|
||||
self._emitted = True
|
||||
self.completed_response = _CompletedEvent(self._response)
|
||||
return {"type": "response.completed"}
|
||||
raise StopIteration
|
||||
|
||||
def get_streamed_output_items(self):
|
||||
return self._streamed_items
|
||||
|
||||
|
||||
def test_collect_response_from_stream_rebuilds_empty_output_from_streamed_items():
|
||||
handler = ResponsesToCompletionBridgeHandler()
|
||||
response = ResponsesAPIResponse.model_construct(
|
||||
id="resp-1",
|
||||
created_at=0,
|
||||
output=[],
|
||||
object="response",
|
||||
model="gpt-5.6-sol",
|
||||
)
|
||||
streamed_items = [
|
||||
{
|
||||
"id": "msg_1",
|
||||
"type": "message",
|
||||
"status": "completed",
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "output_text", "text": "Hi! How can I help?", "annotations": []}
|
||||
],
|
||||
}
|
||||
]
|
||||
stream = _FakeResponsesStream(response, streamed_items)
|
||||
|
||||
collected = handler._collect_response_from_stream(stream)
|
||||
|
||||
assert len(collected.output) == 1
|
||||
assert "Hi! How can I help?" in str(collected.output[0])
|
||||
|
||||
|
||||
def test_collect_response_from_stream_keeps_nonempty_output_untouched():
|
||||
handler = ResponsesToCompletionBridgeHandler()
|
||||
output = [
|
||||
{
|
||||
"id": "msg_0",
|
||||
"type": "message",
|
||||
"status": "completed",
|
||||
"role": "assistant",
|
||||
"content": [{"type": "output_text", "text": "from terminal event", "annotations": []}],
|
||||
}
|
||||
]
|
||||
response = ResponsesAPIResponse.model_construct(
|
||||
id="resp-2",
|
||||
created_at=0,
|
||||
output=output,
|
||||
object="response",
|
||||
model="gpt-5.2",
|
||||
)
|
||||
stream = _FakeResponsesStream(response, [{"id": "msg_1", "type": "message"}])
|
||||
|
||||
collected = handler._collect_response_from_stream(stream)
|
||||
|
||||
assert len(collected.output) == 1
|
||||
assert "from terminal event" in str(collected.output[0])
|
||||
Loading…
Add table
Reference in a new issue