diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 31af5a144eb..4a77a638709 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -51,7 +51,11 @@ from litellm.types.llms.openai import ( from litellm.types.utils import GenericStreamingChunk, ModelResponseStream if TYPE_CHECKING: - from openai.types.responses import ResponseInputImageParam, ResponseOutputItem + from openai.types.responses import ( + ResponseInputImageParam, + ResponseOutputItem, + ResponseOutputMessage, + ) from openai.types.responses.response_text_config_param import ( ResponseTextConfigParam as ResponseText, ) @@ -658,6 +662,102 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): return request_data + @staticmethod + def _choices_from_response_output_message( + item: "ResponseOutputMessage", + starting_index: int, + reasoning_content: str | None, + pending_reasoning_item: dict[str, Any] | None, + ) -> tuple[list[Any], int]: + """Build one Choices per content block on a ResponseOutputMessage. + + The first emitted choice carries the pending reasoning (content + items); + any subsequent content blocks emit choices with reasoning fields cleared, + matching the flush semantics of the original inline loop. + """ + from litellm.types.utils import Choices, Message + + new_choices: Final[list[Any]] = [] + current_index = starting_index + carry_reasoning_content = reasoning_content + carry_reasoning_item = pending_reasoning_item + for content in item.content: + response_text = getattr(content, "text", "") + raw_annotations = getattr(content, "annotations", None) + annotations = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format(raw_annotations) + reasoning_items_for_msg = cast( + list[ChatCompletionReasoningItem] | None, + ([carry_reasoning_item] if carry_reasoning_item is not None else None), + ) + msg = Message( + role=item.role, + content=response_text or "", + reasoning_content=carry_reasoning_content, + annotations=annotations, + reasoning_items=reasoning_items_for_msg, + ) + new_choices.append(Choices(message=msg, finish_reason="stop", index=current_index)) + carry_reasoning_content = None + carry_reasoning_item = None + current_index += 1 + return new_choices, current_index + + @staticmethod + def _merge_accumulated_tool_calls_into_choices( + choices: list[Any], + accumulated_tool_calls: list[dict[str, Any]], + reasoning_content: str | None, + pending_reasoning_item: dict[str, Any] | None, + fallback_index: int, + ) -> None: + """Attach accumulated tool_calls to the last text-message choice, or + append a new tool-only choice if no text message was produced. + + Backfills reasoning_content and reasoning_items onto the merged choice + so encrypted reasoning survives the assistant+tool_calls merge path. + """ + from litellm.types.utils import Choices, Message + + last_msg_choice = next( + ( + c + for c in reversed(choices) + if getattr(c, "message", None) is not None and not getattr(c.message, "tool_calls", None) + ), + None, + ) + if last_msg_choice is None: + reasoning_items_for_msg = cast( + list[ChatCompletionReasoningItem] | None, + ([pending_reasoning_item] if pending_reasoning_item is not None else None), + ) + msg = Message( + content=None, + tool_calls=accumulated_tool_calls, + reasoning_content=reasoning_content, + reasoning_items=reasoning_items_for_msg, + ) + choices.append(Choices(message=msg, finish_reason="tool_calls", index=fallback_index)) + return + + last_msg_choice.message.tool_calls = accumulated_tool_calls + if getattr(last_msg_choice.message, "content", None) is None: + last_msg_choice.message.content = "" + last_msg_choice.finish_reason = "tool_calls" + needs_reasoning_content_backfill = ( + reasoning_content is not None and getattr(last_msg_choice.message, "reasoning_content", None) is None + ) + if needs_reasoning_content_backfill: + last_msg_choice.message.reasoning_content = reasoning_content + needs_reasoning_items_backfill = ( + pending_reasoning_item is not None and getattr(last_msg_choice.message, "reasoning_items", None) is None + ) + if needs_reasoning_items_backfill: + last_msg_choice.message.reasoning_items = cast( + list[ChatCompletionReasoningItem] | None, + [pending_reasoning_item], + ) + @staticmethod def _convert_response_output_to_choices( output_items: Sequence[object], @@ -686,9 +786,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): except ImportError: ResponseApplyPatchToolCall = None - from litellm.types.utils import Choices, Message - - choices: Final[list[Choices]] = [] + choices: Final[list[Any]] = [] index = 0 reasoning_content: str | None = None pending_reasoning_item: _BuiltReasoningItem | None = None @@ -708,35 +806,15 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): reasoning_content = " ".join(s["text"] for s in pending_reasoning_item["summary"] if s.get("text")) elif isinstance(item, ResponseOutputMessage): - for content in item.content: - response_text = getattr(content, "text", "") - # Extract annotations from content if present - raw_annotations = getattr(content, "annotations", None) - annotations = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format( - raw_annotations - ) - msg = Message( - role=item.role, - content=response_text if response_text else "", - reasoning_content=reasoning_content, - annotations=annotations, - reasoning_items=cast( - list[ChatCompletionReasoningItem] | None, - ([pending_reasoning_item] if pending_reasoning_item is not None else None), - ), - ) - - choices.append( - Choices( - message=msg, - finish_reason="stop", - index=index, - ) - ) - - reasoning_content = None # flush - pending_reasoning_item = None # flush - index += 1 + new_choices, index = LiteLLMResponsesTransformationHandler._choices_from_response_output_message( + item=item, + starting_index=index, + reasoning_content=reasoning_content, + pending_reasoning_item=pending_reasoning_item, + ) + choices.extend(new_choices) + reasoning_content = None + pending_reasoning_item = None elif isinstance(item, ResponseFunctionToolCall): from litellm.responses.litellm_completion_transformation.transformation import ( @@ -784,20 +862,14 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): else: pass # don't fail request if item in list is not supported - # If we accumulated tool calls, create a single choice with all of them if accumulated_tool_calls: - msg = Message( - content=None, - tool_calls=accumulated_tool_calls, + LiteLLMResponsesTransformationHandler._merge_accumulated_tool_calls_into_choices( + choices=choices, + accumulated_tool_calls=accumulated_tool_calls, reasoning_content=reasoning_content, - reasoning_items=cast( - list[ChatCompletionReasoningItem] | None, - ([pending_reasoning_item] if pending_reasoning_item is not None else None), - ), + pending_reasoning_item=pending_reasoning_item, + fallback_index=index, ) - choices.append(Choices(message=msg, finish_reason="tool_calls", index=index)) - reasoning_content = None - pending_reasoning_item = None return choices diff --git a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index d0f9bad795d..f688c014ff7 100644 --- a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -1,9 +1,7 @@ -import datetime import json import os -import unittest -from typing import TYPE_CHECKING, Final, List, Literal, Optional, Tuple, get_args -from unittest.mock import ANY, MagicMock, Mock, patch +from typing import TYPE_CHECKING, Final, Literal, get_args +from unittest.mock import MagicMock, Mock, patch import httpx import pytest @@ -55,7 +53,6 @@ def test_convert_chat_completion_messages_to_responses_api_image_input(): assert user_content in response_str assert user_image in response_str - print("response: ", response) assert response[0]["content"][1]["image_url"] == user_image @@ -146,8 +143,6 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_imag ), "image_url should be a flat string, not a nested object" assert "detail" in image_item, "detail field should be present" - print("✓ Tool result with image correctly transformed to Responses API format") - def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text(): """ @@ -224,10 +219,6 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text text_item["text"] == "15 degrees" ), f"Expected text '15 degrees', got '{text_item.get('text')}'" - print( - "✓ Tool result with text correctly transformed to use input_text for Responses API format" - ) - def test_openai_responses_chunk_parser_reasoning_summary(): from litellm.completion_extras.litellm_responses_transformation.transformation import ( @@ -512,8 +503,6 @@ and I learn to carry this small calm home.""" # Check reasoning content assert choice.message.reasoning_content == reasoning_summary.text - print("✓ transform_response correctly handled reasoning items and output messages") - def _make_empty_responses_api_response(model: str = "gpt-5.4"): from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse @@ -983,10 +972,6 @@ def test_transform_request_single_char_keys_not_matched(): assert result_correct.get("metadata") == {"user_id": "123"} assert result_correct.get("previous_response_id") == "resp_abc" - print( - "✓ Single-character keys are not incorrectly matched to metadata/previous_response_id" - ) - # ============================================================================= # Tests for issue #17246: Streaming tool_calls dropped when text + tool_calls @@ -1390,8 +1375,6 @@ def test_tool_message_output_uses_input_text_not_output_text(): ), f"Expected input_text, got {output[0].get('type')}" assert output[0]["text"] == '{"temperature": 15, "condition": "sunny"}' - print("✓ Tool message output correctly uses input_text type") - def test_multiple_tool_calls_in_single_choice(): """ @@ -1533,8 +1516,6 @@ def test_multiple_tool_calls_in_single_choice(): assert tool_calls[2]["id"] == "call_horoscope" assert tool_calls[2]["function"]["name"] == "get_horoscope" - print("✓ Multiple tool calls are correctly grouped in a single choice") - def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): """ @@ -1547,7 +1528,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): When flag is enabled (flag=True or env var), summary="detailed" is added. """ - import litellm from litellm.completion_extras.litellm_responses_transformation.transformation import ( LiteLLMResponsesTransformationHandler, ) @@ -1576,9 +1556,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): "summary" not in result ), f"Summary should NOT be present by default for effort={effort}" - print( - f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}' (no summary by default)" - ) # Test 2: With flag enabled - summary IS added litellm.reasoning_auto_summary = True @@ -1592,9 +1569,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): result["summary"] == "detailed" ), f"Summary should be 'detailed' when flag is enabled for effort={effort}" - print( - f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}', summary='detailed' (flag enabled)" - ) # Test 3: With env var enabled (flag disabled) - summary IS added litellm.reasoning_auto_summary = False @@ -1604,7 +1578,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): assert ( result["summary"] == "detailed" ), "Summary should be 'detailed' when env var is enabled" - print("✓ LITELLM_REASONING_AUTO_SUMMARY env var works correctly") # Test 4: Dict input is passed through as-is (no modification) litellm.reasoning_auto_summary = False @@ -1615,11 +1588,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): result_dict = handler._map_reasoning_effort(dict_input) assert result_dict["effort"] == "high" assert result_dict["summary"] == "custom_summary" - print("✓ Dict input is passed through without modification") - - print( - "✓ All reasoning_effort behaviors work correctly with flag/env var control" - ) finally: # Restore original values @@ -1812,10 +1780,6 @@ def test_transform_response_preserves_annotations(): assert result.usage.completion_tokens == 20 assert result.usage.total_tokens == 30 - print( - "✓ Annotations from Responses API are correctly preserved in Chat Completions format" - ) - def test_apply_patch_tool_call_converted_to_chat_completion_tool_call(): """ @@ -2109,10 +2073,6 @@ def test_multi_tool_call_stream_no_premature_finish(): f"— only response.completed should terminate the stream" ) - print( - "✓ Multi-tool-call stream completes without premature finish_reason termination" - ) - # ============================================================================= # Tests for issue #21331: Parallel tool call indices in streaming @@ -2425,10 +2385,6 @@ def test_parallel_tool_calls_comprehensive_streaming_integration(): 1, }, f"Parallel tool calls must have distinct indices {{0, 1}}, got: {set(added_tool_call_indices)}" - print( - "✓ Parallel tool calls with split argument deltas stream correctly end-to-end" - ) - def test_map_optional_params_preserves_reasoning_summary(): """Test that reasoning_effort dict with summary field is preserved. @@ -2911,6 +2867,165 @@ def test_reasoning_items_streaming_emitted_on_response_completed(): assert ri["summary"][0]["text"] == summary_text +def test_reasoning_items_preserved_when_merged_with_tool_calls(): + """ + Regression: when a Responses turn contains [message, reasoning, function_call] + (reasoning item arriving AFTER the assistant message), the merge path that + attaches accumulated tool_calls onto the last message choice must also backfill + the structured ``reasoning_items`` (with ``encrypted_content``) onto that + message; otherwise the encrypted reasoning payload needed to round-trip on the + next turn is silently dropped. The flattened ``reasoning_content`` string is + already backfilled by the existing code; ``reasoning_items`` currently is not. + """ + from unittest.mock import Mock + + from openai.types.responses import ( + ResponseFunctionToolCall, + ResponseOutputMessage, + ResponseOutputText, + ) + from openai.types.responses.response_reasoning_item import ( + ResponseReasoningItem, + Summary, + ) + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + from litellm.types.llms.openai import ( + InputTokensDetails, + OutputTokensDetails, + ResponseAPIUsage, + ResponsesAPIResponse, + ) + from litellm.types.utils import ModelResponse, Usage + + handler = LiteLLMResponsesTransformationHandler() + + encrypted = "gAAAAABpw5xyz789FAKE==" + summary_text = "deciding which city" + preamble = "Let me check the weather." + + output_message = ResponseOutputMessage( + id="msg_merge001", + content=[ + ResponseOutputText( + annotations=[], + text=preamble, + type="output_text", + logprobs=[], + ) + ], + role="assistant", + status="completed", + type="message", + ) + reasoning_item = ResponseReasoningItem( + id="rs_merge001", + summary=[Summary(text=summary_text, type="summary_text")], + type="reasoning", + content=None, + encrypted_content=encrypted, + status=None, + ) + function_call_item = ResponseFunctionToolCall( + id="fc_merge001", + type="function_call", + status="completed", + arguments='{"city": "Paris"}', + call_id="call_paris_merge", + name="get_weather", + ) + + usage = ResponseAPIUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + audio_tokens=None, cached_tokens=0, text_tokens=None + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None), + total_tokens=30, + cost=None, + ) + raw_response = ResponsesAPIResponse( + id="resp_merge001", + created_at=1234567890, + error=None, + incomplete_details=None, + instructions=None, + metadata={}, + model="gpt-5-mini", + object="response", + output=[output_message, reasoning_item, function_call_item], + parallel_tool_calls=True, + temperature=1.0, + tool_choice="auto", + tools=[], + top_p=1.0, + max_output_tokens=None, + previous_response_id=None, + reasoning={"effort": "low", "summary": "detailed"}, + status="completed", + text={"format": {"type": "text"}, "verbosity": "medium"}, + truncation="disabled", + usage=usage, + user=None, + store=True, + background=False, + billing={"payer": "developer"}, + max_tool_calls=None, + prompt_cache_key=None, + safety_identifier=None, + service_tier="default", + top_logprobs=0, + ) + model_response = ModelResponse( + id="chatcmpl-merge001", + created=1234567890, + model=None, + object="chat.completion", + system_fingerprint=None, + choices=[], + usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0), + ) + + result = handler.transform_response( + model="gpt-5-mini", + raw_response=raw_response, + model_response=model_response, + logging_obj=Mock(), + request_data={"model": "gpt-5-mini"}, + messages=[{"role": "user", "content": "What's the weather in Paris?"}], + optional_params={}, + litellm_params={}, + encoding=Mock(), + ) + + assert len(result.choices) == 1, ( + f"merge must collapse into a single choice, got {len(result.choices)}" + ) + choice = result.choices[0] + msg = choice.message + + assert msg.tool_calls is not None and len(msg.tool_calls) == 1 + assert msg.tool_calls[0]["function"]["name"] == "get_weather" + assert msg.tool_calls[0]["function"]["arguments"] == '{"city": "Paris"}' + assert choice.finish_reason == "tool_calls" + + assert msg.content == preamble, "assistant preamble text must be preserved" + + assert msg.reasoning_content == summary_text, ( + "reasoning_content must be backfilled onto the merged choice" + ) + + assert msg.reasoning_items is not None, ( + "reasoning_items must be backfilled onto the merged choice so encrypted_content " + "can round-trip to the provider on the next turn" + ) + assert len(msg.reasoning_items) == 1 + assert msg.reasoning_items[0]["encrypted_content"] == encrypted + + def test_streaming_function_call_tool_id_for_degenerate_call_id(): """In streaming, Bedrock Mantle's function_call event carries a unique ``id`` (``fc_...``) and a non-unique, index-based ``call_id`` (``call_0``). For that @@ -2945,6 +3060,320 @@ def test_streaming_function_call_tool_id_for_degenerate_call_id(): assert stream_tool_id("fc_2", "call_tokyo") == "call_tokyo" +def _build_message_plus_tool_call_response( + output_items, + model="gpt-4o", +): + """Helper that wraps output_items into a minimal ResponsesAPIResponse and returns the + transformed ModelResponse produced by LiteLLMResponsesTransformationHandler.""" + from unittest.mock import Mock + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + from litellm.types.llms.openai import ( + InputTokensDetails, + OutputTokensDetails, + ResponseAPIUsage, + ResponsesAPIResponse, + ) + from litellm.types.utils import ModelResponse, Usage + + usage = ResponseAPIUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails(cached_tokens=0), + output_tokens=20, + output_tokens_details=OutputTokensDetails(reasoning_tokens=0), + total_tokens=30, + ) + + raw_response = ResponsesAPIResponse( + id="resp_bug18401", + created_at=1234567890, + error=None, + incomplete_details=None, + instructions=None, + metadata={}, + model=model, + object="response", + output=output_items, + parallel_tool_calls=True, + temperature=1.0, + tool_choice="auto", + tools=[], + top_p=1.0, + max_output_tokens=None, + previous_response_id=None, + reasoning=None, + status="completed", + text=None, + truncation="disabled", + usage=usage, + user=None, + store=True, + background=False, + ) + + model_response = ModelResponse( + id="chatcmpl-bug18401", + created=1234567890, + model=None, + object="chat.completion", + choices=[], + usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0), + ) + + handler = LiteLLMResponsesTransformationHandler() + return handler.transform_response( + model=model, + raw_response=raw_response, + model_response=model_response, + logging_obj=Mock(), + request_data={"model": model}, + messages=[{"role": "user", "content": "test"}], + optional_params={}, + litellm_params={}, + encoding=Mock(), + ) + + +def _make_output_message(text, message_id="msg_bug18401"): + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + + return ResponseOutputMessage( + id=message_id, + content=[ + ResponseOutputText( + annotations=[], + text=text, + type="output_text", + logprobs=[], + ) + ], + role="assistant", + status="completed", + type="message", + ) + + +def _make_function_tool_call(call_id, name, arguments, tc_id="fc_bug18401"): + from openai.types.responses import ResponseFunctionToolCall + + return ResponseFunctionToolCall( + id=tc_id, + type="function_call", + status="completed", + arguments=arguments, + call_id=call_id, + name=name, + ) + + +def test_message_plus_function_call_merged_into_single_choice(): + """Regression: a Responses turn that contains both a message and a function_call must + collapse into a single Chat Completions choice, because Chat Completions clients only + read choices[0]. Prior to the fix the bridge emitted two choices (content in [0], + tool_calls in [1]) which caused tools to be silently ignored downstream.""" + result = _build_message_plus_tool_call_response( + output_items=[ + _make_output_message("Fetching the weather now."), + _make_function_tool_call( + call_id="call_paris", + name="get_weather", + arguments='{"location": "Paris"}', + ), + ] + ) + + assert len(result.choices) == 1, f"Expected 1 choice, got {len(result.choices)}" + choice = result.choices[0] + assert choice.finish_reason == "tool_calls" + assert choice.message.content == "Fetching the weather now." + tool_calls = choice.message.tool_calls + assert tool_calls is not None and len(tool_calls) == 1 + assert tool_calls[0]["id"] == "call_paris" + assert tool_calls[0]["function"]["name"] == "get_weather" + assert tool_calls[0]["function"]["arguments"] == '{"location": "Paris"}' + + +def test_tool_only_turn_unchanged(): + """Guard against regressing the pre-fix behaviour for tool-only turns: when the + Responses output has no assistant message, a fresh Choice must still be appended + with the accumulated tool_calls and finish_reason='tool_calls'.""" + result = _build_message_plus_tool_call_response( + output_items=[ + _make_function_tool_call( + call_id="call_only", + name="get_weather", + arguments='{"location": "Tokyo"}', + ), + ] + ) + + assert len(result.choices) == 1 + choice = result.choices[0] + assert choice.finish_reason == "tool_calls" + tool_calls = choice.message.tool_calls + assert tool_calls is not None and len(tool_calls) == 1 + assert tool_calls[0]["id"] == "call_only" + + +def test_message_only_turn_unchanged(): + """Guard: a message-only Responses turn must still produce a single choice with + finish_reason='stop' after the merge fix (no accumulated_tool_calls path taken).""" + result = _build_message_plus_tool_call_response( + output_items=[_make_output_message("Just a plain answer.")] + ) + + assert len(result.choices) == 1 + choice = result.choices[0] + assert choice.finish_reason == "stop" + assert choice.message.content == "Just a plain answer." + assert not choice.message.tool_calls + + +def test_multiple_function_calls_after_message_merged(): + """When a message is followed by multiple parallel function_calls, all of them must + land on the single message-carrying choice (Chat Completions groups them all under + one message).""" + result = _build_message_plus_tool_call_response( + output_items=[ + _make_output_message("Fetching two cities in parallel."), + _make_function_tool_call( + call_id="call_paris", + name="get_weather", + arguments='{"location": "Paris"}', + tc_id="fc_1", + ), + _make_function_tool_call( + call_id="call_tokyo", + name="get_weather", + arguments='{"location": "Tokyo"}', + tc_id="fc_2", + ), + ] + ) + + assert len(result.choices) == 1 + choice = result.choices[0] + assert choice.finish_reason == "tool_calls" + assert choice.message.content == "Fetching two cities in parallel." + tool_calls = choice.message.tool_calls + assert tool_calls is not None and len(tool_calls) == 2 + assert {tc["id"] for tc in tool_calls} == {"call_paris", "call_tokyo"} + assert all(tc["function"]["name"] == "get_weather" for tc in tool_calls) + + +def test_multiple_message_items_tool_calls_collapse_onto_last(): + """When the Responses output contains two assistant messages before a function_call, + the accumulated tool_calls must attach to the LAST (most recent) message choice, not + the first, because that is the one the model was on when it decided to call the tool. + The earlier message choice must remain a plain text turn with finish_reason='stop'.""" + result = _build_message_plus_tool_call_response( + output_items=[ + _make_output_message("first", message_id="msg_first"), + _make_output_message("second", message_id="msg_second"), + _make_function_tool_call( + call_id="call_after_second", + name="get_weather", + arguments='{"location": "Paris"}', + ), + ] + ) + + assert len(result.choices) == 2 + + first_choice = next(c for c in result.choices if c.message.content == "first") + second_choice = next(c for c in result.choices if c.message.content == "second") + + assert first_choice.finish_reason == "stop" + assert not first_choice.message.tool_calls + + assert second_choice.finish_reason == "tool_calls" + tool_calls = second_choice.message.tool_calls + assert tool_calls is not None and len(tool_calls) == 1 + assert tool_calls[0]["id"] == "call_after_second" + + +def test_streaming_text_plus_function_call_lands_on_choice_index_zero(): + """Streaming counterpart to the merged-choice fix: when a message and a function_call + arrive in the same Responses turn, both the content deltas and the tool_call deltas + must be emitted on choice index 0, and the terminal response.completed chunk must + carry finish_reason='tool_calls'. This mirrors the non-streaming merge behaviour so + that Chat Completions clients that only read choices[0] see both the text and the + tool call.""" + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + OpenAiResponsesToChatCompletionStreamIterator, + ) + + iterator = OpenAiResponsesToChatCompletionStreamIterator( + streaming_response=None, sync_stream=True + ) + + chunks = [ + {"type": "response.output_text.delta", "delta": "Fetching "}, + {"type": "response.output_text.delta", "delta": "the weather."}, + { + "type": "response.output_item.done", + "item": {"type": "message", "content": []}, + }, + { + "type": "response.output_item.added", + "item": { + "type": "function_call", + "name": "get_weather", + "call_id": "call_paris", + "id": "fc_1", + }, + }, + { + "type": "response.function_call_arguments.delta", + "delta": '{"location":"Paris"}', + }, + { + "type": "response.output_item.done", + "item": { + "type": "function_call", + "name": "get_weather", + "call_id": "call_paris", + "arguments": '{"location":"Paris"}', + }, + }, + { + "type": "response.completed", + "response": { + "output": [ + {"type": "message"}, + {"type": "function_call", "call_id": "call_paris"}, + ] + }, + }, + ] + + results = [iterator.chunk_parser(chunk) for chunk in chunks] + + for idx, r in enumerate(results): + assert len(r.choices) == 1, f"chunk {idx} produced {len(r.choices)} choices" + assert r.choices[0].index == 0, ( + f"chunk {idx} landed on index {r.choices[0].index}, expected 0" + ) + + text_deltas = [r.choices[0].delta.content for r in results[:2]] + assert text_deltas == ["Fetching ", "the weather."] + + tool_added = results[3].choices[0].delta.tool_calls + assert tool_added and tool_added[0]["function"]["name"] == "get_weather" + + tool_arg_delta = results[4].choices[0].delta.tool_calls + assert ( + tool_arg_delta + and tool_arg_delta[0]["function"]["arguments"] == '{"location":"Paris"}' + ) + + terminal = results[-1] + assert terminal.choices[0].finish_reason == "tool_calls" + + def test_streaming_chunks_share_one_chat_completion_id(): """Every chunk of one streamed chat completion must carry the same ``id``, per the OpenAI spec. The bridge builds a fresh ``ModelResponseStream`` per Responses event, @@ -3567,8 +3996,8 @@ async def test_acompletion_bridge_normalizes_tool_choice_on_the_wire( def _make_incomplete_responses_api_response( - incomplete_reason: Optional[str], - output: "List[ResponseOutputItem]", + incomplete_reason: str | None, + output: "list[ResponseOutputItem]", status: Literal["completed", "incomplete"] = "incomplete", empty_incomplete_details: bool = False, ) -> "ResponsesAPIResponse":