fix(responses): prevent streaming tool_calls from being dropped when text + tool_calls (#17652)

When OpenAI Responses API returns both text AND tool_calls, the bridge
transformation was emitting is_finished=True after the text message completed,
causing subsequent tool_call chunks to be dropped.

The fix:
- response.output_item.done for messages no longer emits is_finished=True
- Added handler for response.completed to properly signal stream end
This commit is contained in:
Cesar Garcia 2025-12-08 23:51:59 -03:00 • committed by GitHub
parent 61e737e361
commit fd9ff90307
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 136 additions and 1 deletions

View file

@ -873,8 +873,10 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
usage=None,
)
elif output_item.get("type") == "message":
# Don't emit is_finished=True here - there may be more output items
# (e.g., tool_calls) coming after the message. Wait for response.completed.
return GenericStreamingChunk(
finish_reason="stop", is_finished=True, usage=None, text=""
finish_reason="", is_finished=False, usage=None, text=""
)
elif event_type == "response.output_text.delta":
@ -907,6 +909,12 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
)
]
)
elif event_type == "response.completed":
# Response is fully complete - now we can signal is_finished=True
# This ensures we don't prematurely end the stream before tool_calls arrive
return GenericStreamingChunk(
text="", tool_use=None, is_finished=True, finish_reason="stop", usage=None
)
else:
pass
# For any unhandled event types, create a minimal valid chunk or skip

View file

@ -444,3 +444,130 @@ def test_transform_request_single_char_keys_not_matched():
assert result_correct.get("previous_response_id") == "resp_abc"
print("✓ Single-character keys are not incorrectly matched to metadata/previous_response_id")
# =============================================================================
# Tests for issue #17246: Streaming tool_calls dropped when text + tool_calls
# =============================================================================
def test_message_done_does_not_emit_is_finished():
"""
Test that OUTPUT_ITEM_DONE for a message does NOT emit is_finished=True.
This is the core fix for issue #17246.
Before fix: message completion emitted is_finished=True, causing tool_calls
that came after to be dropped.
"""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
chunk = {
"type": "response.output_item.done",
"item": {"type": "message", "content": []}
}
result = iterator.chunk_parser(chunk)
# After the fix, message completion should NOT set is_finished=True
assert result["is_finished"] == False, "message completion should not emit is_finished=True"
assert result["finish_reason"] == "", "message completion should not emit finish_reason"
def test_response_completed_emits_is_finished():
"""
Test that response.completed DOES emit is_finished=True.
This ensures streaming ends properly after ALL output items are sent.
"""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
chunk = {"type": "response.completed"}
result = iterator.chunk_parser(chunk)
assert result["is_finished"] == True, "response.completed should emit is_finished=True"
assert result["finish_reason"] == "stop", "response.completed should emit finish_reason='stop'"
def test_function_call_done_emits_is_finished():
"""
Test that OUTPUT_ITEM_DONE for a function_call still emits is_finished=True.
This preserves existing behavior for tool_calls.
"""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
chunk = {
"type": "response.output_item.done",
"item": {
"type": "function_call",
"name": "get_weather",
"call_id": "call_123",
"arguments": '{"location": "Tokyo"}'
}
}
result = iterator.chunk_parser(chunk)
assert result["is_finished"] == True, "function_call completion should emit is_finished=True"
assert result["finish_reason"] == "tool_calls", "function_call should emit finish_reason='tool_calls'"
assert result["tool_use"] is not None, "function_call should include tool_use"
def test_text_plus_tool_calls_sequence():
"""
Test the full sequence when model returns text + tool_calls.
This is the main scenario for issue #17246.
Expected: is_finished=True should NOT appear until function_call is done,
not when message is done.
"""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
# Simulate the sequence from OpenAI Responses API
chunks = [
{"type": "response.output_text.delta", "delta": "Hello"},
{"type": "response.output_text.delta", "delta": "!"},
{"type": "response.output_item.done", "item": {"type": "message", "content": []}}, # message done
{"type": "response.output_item.added", "item": {"type": "function_call", "name": "get_weather", "call_id": "call_123"}},
{"type": "response.function_call_arguments.delta", "delta": '{"location":"Tokyo"}'},
{"type": "response.output_item.done", "item": {"type": "function_call", "name": "get_weather", "call_id": "call_123", "arguments": '{"location":"Tokyo"}'}},
{"type": "response.completed"},
]
results = [iterator.chunk_parser(chunk) for chunk in chunks]
# Check message done (index 2) does NOT have is_finished=True
message_done_result = results[2]
assert message_done_result["is_finished"] == False, "message done should not have is_finished=True"
# Check function_call done (index 5) DOES have is_finished=True
function_done_result = results[5]
assert function_done_result["is_finished"] == True, "function_call done should have is_finished=True"
assert function_done_result["finish_reason"] == "tool_calls"
# Check response.completed (index 6) also has is_finished=True
completed_result = results[6]
assert completed_result["is_finished"] == True, "response.completed should have is_finished=True"