From 52f7dc4cf0e4cc646015967f67acad6e4c3c3bf3 Mon Sep 17 00:00:00 2001 From: Vineeth Sai Date: Tue, 25 Aug 2026 12:50:47 -0700 Subject: [PATCH] fix(responses): stop closing a message item the tool-only stream never opened `_ensure_output_item_for_chunk` deliberately opens no message item for a tool-first chunk, so a turn that produces only tool calls has no message item to close. `return_default_done_events` emitted `output_text.done`, `content_part.done` and `output_item.done` for it anyway, referencing an item id the client never saw opened. Clients that track output items by id reject that. Skip the three message-done events when no content part was ever added and the built response carries tool calls and no text, latching the flags so `is_stream_finished()` still reports done and the stream proceeds straight to `response.completed`. Rebased onto current staging. The test file was renamed upstream from test_tool_call_streaming_transformation.py to test_streaming_iterator_transformation.py, so the three new cases move into the successor file, and this branch's helper `_chunk` is renamed `_tool_chunk` so it cannot shadow the `_chunk` that file already defines with a different signature. --- .../streaming_iterator.py | 22 +++ .../test_streaming_iterator_transformation.py | 148 ++++++++++++++++++ 2 files changed, 170 insertions(+) diff --git a/litellm/responses/litellm_completion_transformation/streaming_iterator.py b/litellm/responses/litellm_completion_transformation/streaming_iterator.py index 92bbca9ee5b..0677d141a58 100644 --- a/litellm/responses/litellm_completion_transformation/streaming_iterator.py +++ b/litellm/responses/litellm_completion_transformation/streaming_iterator.py @@ -743,9 +743,31 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator): ), ) + def _message_item_was_never_opened(self, litellm_complete_object: ModelResponse) -> bool: + """True when this turn produced tool calls only, so no message item exists. + + ``_ensure_output_item_for_chunk`` deliberately opens no message item for a + tool-first chunk, and a turn with no assistant text has nothing to put in one. + Emitting the message ``.done`` events anyway would reference an item the client + never saw opened, which clients that track output items by id reject. + """ + if self.sent_content_part_added_event: + return False + message: Final = litellm_complete_object.choices[0].message + if getattr(message, "content", None) or getattr(message, "reasoning_content", None): + return False + return bool(getattr(message, "tool_calls", None)) + def return_default_done_events( self, litellm_complete_object: ModelResponse ) -> BaseLiteLLMOpenAIResponseObject | None: + if self._message_item_was_never_opened(litellm_complete_object): + # Latch the flags so is_stream_finished() still reports done and the + # stream proceeds straight to response.completed. + self.sent_output_text_done_event = True + self.sent_output_content_part_done_event = True + self.sent_output_item_done_event = True + return None if self.sent_output_text_done_event is False: self.sent_output_text_done_event = True return self.create_output_text_done_event(litellm_complete_object) diff --git a/tests/test_litellm/responses/litellm_completion_transformation/test_streaming_iterator_transformation.py b/tests/test_litellm/responses/litellm_completion_transformation/test_streaming_iterator_transformation.py index 823f656ddc5..01ead031312 100644 --- a/tests/test_litellm/responses/litellm_completion_transformation/test_streaming_iterator_transformation.py +++ b/tests/test_litellm/responses/litellm_completion_transformation/test_streaming_iterator_transformation.py @@ -523,3 +523,151 @@ async def test_streaming_response_id_falls_back_when_upstream_yields_nothing(): assert response_ids assert len(set(response_ids)) == 1 assert response_ids[0].startswith("resp_") + + +class _FakeChatCompletionStream: + """Minimal stand-in for CustomStreamWrapper: yields chunks, carries logging_obj.""" + + def __init__(self, chunks): + self._chunks = iter(chunks) + self.logging_obj = None + + def __iter__(self): + return self + + def __next__(self): + return next(self._chunks) + + +def _tool_chunk(content=None, tool_calls=None, finish_reason=None): + return ModelResponseStream( + id="chatcmpl-a83572f7", + created=123, + model="claude-sonnet-4-6", + object="chat.completion.chunk", + choices=[ + StreamingChoices( + finish_reason=finish_reason, + index=0, + delta=Delta( + role="assistant", content=content, tool_calls=tool_calls + ), + ) + ], + ) + + +_WEB_SEARCH_TOOL_CALL = [ + { + "id": "tooluse_8G8H6h8nrt79WT4b8sHNd0", + "type": "function", + "function": {"name": "web_search", "arguments": '{"query":"floppy disks"}'}, + } +] + + +def _drive(chunks): + """Run the whole stream and return (event type, item id) for every event.""" + iterator = LiteLLMCompletionStreamingIterator( + model="claude-sonnet-4-6", + litellm_custom_stream_wrapper=_FakeChatCompletionStream(chunks), + request_input="Search the web for floppy disks.", + responses_api_request={}, + ) + iterator.litellm_custom_stream_wrapper.__anext__ = AsyncMock() + return [ + ( + event.type, + getattr(event, "item_id", None) + or getattr(getattr(event, "item", None), "id", None), + ) + for event in iterator + ] + + +def _unopened_done_events(events): + """Every *.done whose item id was never introduced by a matching *.added.""" + opened = { + item_id + for event_type, item_id in events + if event_type + in ( + ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED, + ResponsesAPIStreamEvents.CONTENT_PART_ADDED, + ) + } + return [ + (event_type, item_id) + for event_type, item_id in events + if event_type + in ( + ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE, + ResponsesAPIStreamEvents.CONTENT_PART_DONE, + ResponsesAPIStreamEvents.OUTPUT_TEXT_DONE, + ) + and item_id not in opened + ] + + +def test_tool_only_turn_does_not_close_a_message_item_it_never_opened(): + """A tool call with no assistant text must not emit message .done events. + + _ensure_output_item_for_chunk opens no message item for a tool-first chunk, so + emitting output_text.done / content_part.done / output_item.done for one leaves + three events pointing at an id the client never saw. Clients that track output + items by id (e.g. the Vercel AI SDK) abort the run on that. + """ + events = _drive( + [ + _tool_chunk(content="", tool_calls=_WEB_SEARCH_TOOL_CALL), + _tool_chunk(finish_reason="tool_calls"), + ] + ) + + assert _unopened_done_events(events) == [] + # The tool call itself is still opened and closed as before. + assert ( + ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED, + "tooluse_8G8H6h8nrt79WT4b8sHNd0", + ) in events + assert ( + ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE, + "tooluse_8G8H6h8nrt79WT4b8sHNd0", + ) in events + assert events[-1][0] == ResponsesAPIStreamEvents.RESPONSE_COMPLETED + + +def test_tool_call_with_text_still_closes_its_message_item(): + """Control: text alongside the tool call keeps the full message item lifecycle.""" + events = _drive( + [ + _tool_chunk(content="Let me search."), + _tool_chunk(content="", tool_calls=_WEB_SEARCH_TOOL_CALL), + _tool_chunk(finish_reason="tool_calls"), + ] + ) + + assert _unopened_done_events(events) == [] + message_id = next( + item_id + for event_type, item_id in events + if event_type == ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED + and item_id.startswith("msg_") + ) + for done_event in ( + ResponsesAPIStreamEvents.OUTPUT_TEXT_DONE, + ResponsesAPIStreamEvents.CONTENT_PART_DONE, + ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE, + ): + assert (done_event, message_id) in events + + +def test_text_only_turn_still_closes_its_message_item(): + """Control: the ordinary no-tool-calls stream is untouched.""" + events = _drive([_tool_chunk(content="Floppy disks are obsolete."), _tool_chunk(finish_reason="stop")]) + + assert _unopened_done_events(events) == [] + assert any( + event_type == ResponsesAPIStreamEvents.OUTPUT_TEXT_DONE + for event_type, _ in events + )