From 510363fbadd374a72cdffbbdb605ca70576fdf9f Mon Sep 17 00:00:00 2001 From: Stephen Chin Date: Thu, 6 Aug 2026 18:48:12 +0000 Subject: [PATCH 1/5] fix(responses): merge assistant message and tool calls into one choice Cherry-picked from upstream commit eda94a7e1e829744c8ea8fba97f5fdb5e8dd05b4. When a Responses API turn produces both an assistant text message and one or more function/tool calls, the chat-completions bridge was putting the tool calls on a separate choice from the message. Cursor agent mode and other chat-completions clients only look at the first choice with tool_calls, so a text-plus-tool-call turn silently dropped the tool call. This finds the last message choice without existing tool_calls and attaches the accumulated tool calls to it instead of appending a new choice, falling back to a fresh choice only when no bare message choice exists. Adds regression tests for message+tool-call merging, tool-only and message-only turns staying unchanged, multiple function calls after a message, multiple message items collapsing tool calls onto the last one, and the streaming path landing text-plus-function-call on choice index 0. --- .../transformation.py | 47 ++- ...responses_transformation_transformation.py | 314 ++++++++++++++++++ 2 files changed, 352 insertions(+), 9 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 31af5a144eb..620dbde2312 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -784,18 +784,47 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): else: pass # don't fail request if item in list is not supported - # If we accumulated tool calls, create a single choice with all of them if accumulated_tool_calls: - msg = Message( - content=None, - tool_calls=accumulated_tool_calls, - reasoning_content=reasoning_content, - reasoning_items=cast( - list[ChatCompletionReasoningItem] | None, - ([pending_reasoning_item] if pending_reasoning_item is not None else None), + last_msg_choice = next( + ( + c + for c in reversed(choices) + if getattr(c, "message", None) is not None and not getattr(c.message, "tool_calls", None) ), + None, ) - choices.append(Choices(message=msg, finish_reason="tool_calls", index=index)) + if last_msg_choice is not None: + last_msg_choice.message.tool_calls = accumulated_tool_calls + if getattr(last_msg_choice.message, "content", None) is None: + last_msg_choice.message.content = "" + last_msg_choice.finish_reason = "tool_calls" + if ( + reasoning_content is not None + and getattr(last_msg_choice.message, "reasoning_content", None) is None + ): + last_msg_choice.message.reasoning_content = reasoning_content + # Mirror the reasoning_content backfill above: reasoning_items carries + # encrypted_content, needed to round-trip reasoning to the provider on + # the next turn, and was being dropped on this merge path. + if ( + pending_reasoning_item is not None + and getattr(last_msg_choice.message, "reasoning_items", None) is None + ): + last_msg_choice.message.reasoning_items = cast( + list[ChatCompletionReasoningItem] | None, + [pending_reasoning_item], + ) + else: + msg = Message( + content=None, + tool_calls=accumulated_tool_calls, + reasoning_content=reasoning_content, + reasoning_items=cast( + list[ChatCompletionReasoningItem] | None, + ([pending_reasoning_item] if pending_reasoning_item is not None else None), + ), + ) + choices.append(Choices(message=msg, finish_reason="tool_calls", index=index)) reasoning_content = None pending_reasoning_item = None diff --git a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index d0f9bad795d..eedbdf3c600 100644 --- a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -2945,6 +2945,320 @@ def test_streaming_function_call_tool_id_for_degenerate_call_id(): assert stream_tool_id("fc_2", "call_tokyo") == "call_tokyo" +def _build_message_plus_tool_call_response( + output_items, + model="gpt-4o", +): + """Helper that wraps output_items into a minimal ResponsesAPIResponse and returns the + transformed ModelResponse produced by LiteLLMResponsesTransformationHandler.""" + from unittest.mock import Mock + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + from litellm.types.llms.openai import ( + InputTokensDetails, + OutputTokensDetails, + ResponseAPIUsage, + ResponsesAPIResponse, + ) + from litellm.types.utils import ModelResponse, Usage + + usage = ResponseAPIUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails(cached_tokens=0), + output_tokens=20, + output_tokens_details=OutputTokensDetails(reasoning_tokens=0), + total_tokens=30, + ) + + raw_response = ResponsesAPIResponse( + id="resp_bug18401", + created_at=1234567890, + error=None, + incomplete_details=None, + instructions=None, + metadata={}, + model=model, + object="response", + output=output_items, + parallel_tool_calls=True, + temperature=1.0, + tool_choice="auto", + tools=[], + top_p=1.0, + max_output_tokens=None, + previous_response_id=None, + reasoning=None, + status="completed", + text=None, + truncation="disabled", + usage=usage, + user=None, + store=True, + background=False, + ) + + model_response = ModelResponse( + id="chatcmpl-bug18401", + created=1234567890, + model=None, + object="chat.completion", + choices=[], + usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0), + ) + + handler = LiteLLMResponsesTransformationHandler() + return handler.transform_response( + model=model, + raw_response=raw_response, + model_response=model_response, + logging_obj=Mock(), + request_data={"model": model}, + messages=[{"role": "user", "content": "test"}], + optional_params={}, + litellm_params={}, + encoding=Mock(), + ) + + +def _make_output_message(text, message_id="msg_bug18401"): + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + + return ResponseOutputMessage( + id=message_id, + content=[ + ResponseOutputText( + annotations=[], + text=text, + type="output_text", + logprobs=[], + ) + ], + role="assistant", + status="completed", + type="message", + ) + + +def _make_function_tool_call(call_id, name, arguments, tc_id="fc_bug18401"): + from openai.types.responses import ResponseFunctionToolCall + + return ResponseFunctionToolCall( + id=tc_id, + type="function_call", + status="completed", + arguments=arguments, + call_id=call_id, + name=name, + ) + + +def test_message_plus_function_call_merged_into_single_choice(): + """Regression: a Responses turn that contains both a message and a function_call must + collapse into a single Chat Completions choice, because Chat Completions clients only + read choices[0]. Prior to the fix the bridge emitted two choices (content in [0], + tool_calls in [1]) which caused tools to be silently ignored downstream.""" + result = _build_message_plus_tool_call_response( + output_items=[ + _make_output_message("Fetching the weather now."), + _make_function_tool_call( + call_id="call_paris", + name="get_weather", + arguments='{"location": "Paris"}', + ), + ] + ) + + assert len(result.choices) == 1, f"Expected 1 choice, got {len(result.choices)}" + choice = result.choices[0] + assert choice.finish_reason == "tool_calls" + assert choice.message.content == "Fetching the weather now." + tool_calls = choice.message.tool_calls + assert tool_calls is not None and len(tool_calls) == 1 + assert tool_calls[0]["id"] == "call_paris" + assert tool_calls[0]["function"]["name"] == "get_weather" + assert tool_calls[0]["function"]["arguments"] == '{"location": "Paris"}' + + +def test_tool_only_turn_unchanged(): + """Guard against regressing the pre-fix behaviour for tool-only turns: when the + Responses output has no assistant message, a fresh Choice must still be appended + with the accumulated tool_calls and finish_reason='tool_calls'.""" + result = _build_message_plus_tool_call_response( + output_items=[ + _make_function_tool_call( + call_id="call_only", + name="get_weather", + arguments='{"location": "Tokyo"}', + ), + ] + ) + + assert len(result.choices) == 1 + choice = result.choices[0] + assert choice.finish_reason == "tool_calls" + tool_calls = choice.message.tool_calls + assert tool_calls is not None and len(tool_calls) == 1 + assert tool_calls[0]["id"] == "call_only" + + +def test_message_only_turn_unchanged(): + """Guard: a message-only Responses turn must still produce a single choice with + finish_reason='stop' after the merge fix (no accumulated_tool_calls path taken).""" + result = _build_message_plus_tool_call_response( + output_items=[_make_output_message("Just a plain answer.")] + ) + + assert len(result.choices) == 1 + choice = result.choices[0] + assert choice.finish_reason == "stop" + assert choice.message.content == "Just a plain answer." + assert not choice.message.tool_calls + + +def test_multiple_function_calls_after_message_merged(): + """When a message is followed by multiple parallel function_calls, all of them must + land on the single message-carrying choice (Chat Completions groups them all under + one message).""" + result = _build_message_plus_tool_call_response( + output_items=[ + _make_output_message("Fetching two cities in parallel."), + _make_function_tool_call( + call_id="call_paris", + name="get_weather", + arguments='{"location": "Paris"}', + tc_id="fc_1", + ), + _make_function_tool_call( + call_id="call_tokyo", + name="get_weather", + arguments='{"location": "Tokyo"}', + tc_id="fc_2", + ), + ] + ) + + assert len(result.choices) == 1 + choice = result.choices[0] + assert choice.finish_reason == "tool_calls" + assert choice.message.content == "Fetching two cities in parallel." + tool_calls = choice.message.tool_calls + assert tool_calls is not None and len(tool_calls) == 2 + assert {tc["id"] for tc in tool_calls} == {"call_paris", "call_tokyo"} + assert all(tc["function"]["name"] == "get_weather" for tc in tool_calls) + + +def test_multiple_message_items_tool_calls_collapse_onto_last(): + """When the Responses output contains two assistant messages before a function_call, + the accumulated tool_calls must attach to the LAST (most recent) message choice, not + the first, because that is the one the model was on when it decided to call the tool. + The earlier message choice must remain a plain text turn with finish_reason='stop'.""" + result = _build_message_plus_tool_call_response( + output_items=[ + _make_output_message("first", message_id="msg_first"), + _make_output_message("second", message_id="msg_second"), + _make_function_tool_call( + call_id="call_after_second", + name="get_weather", + arguments='{"location": "Paris"}', + ), + ] + ) + + assert len(result.choices) == 2 + + first_choice = next(c for c in result.choices if c.message.content == "first") + second_choice = next(c for c in result.choices if c.message.content == "second") + + assert first_choice.finish_reason == "stop" + assert not first_choice.message.tool_calls + + assert second_choice.finish_reason == "tool_calls" + tool_calls = second_choice.message.tool_calls + assert tool_calls is not None and len(tool_calls) == 1 + assert tool_calls[0]["id"] == "call_after_second" + + +def test_streaming_text_plus_function_call_lands_on_choice_index_zero(): + """Streaming counterpart to the merged-choice fix: when a message and a function_call + arrive in the same Responses turn, both the content deltas and the tool_call deltas + must be emitted on choice index 0, and the terminal response.completed chunk must + carry finish_reason='tool_calls'. This mirrors the non-streaming merge behaviour so + that Chat Completions clients that only read choices[0] see both the text and the + tool call.""" + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + OpenAiResponsesToChatCompletionStreamIterator, + ) + + iterator = OpenAiResponsesToChatCompletionStreamIterator( + streaming_response=None, sync_stream=True + ) + + chunks = [ + {"type": "response.output_text.delta", "delta": "Fetching "}, + {"type": "response.output_text.delta", "delta": "the weather."}, + { + "type": "response.output_item.done", + "item": {"type": "message", "content": []}, + }, + { + "type": "response.output_item.added", + "item": { + "type": "function_call", + "name": "get_weather", + "call_id": "call_paris", + "id": "fc_1", + }, + }, + { + "type": "response.function_call_arguments.delta", + "delta": '{"location":"Paris"}', + }, + { + "type": "response.output_item.done", + "item": { + "type": "function_call", + "name": "get_weather", + "call_id": "call_paris", + "arguments": '{"location":"Paris"}', + }, + }, + { + "type": "response.completed", + "response": { + "output": [ + {"type": "message"}, + {"type": "function_call", "call_id": "call_paris"}, + ] + }, + }, + ] + + results = [iterator.chunk_parser(chunk) for chunk in chunks] + + for idx, r in enumerate(results): + assert len(r.choices) == 1, f"chunk {idx} produced {len(r.choices)} choices" + assert r.choices[0].index == 0, ( + f"chunk {idx} landed on index {r.choices[0].index}, expected 0" + ) + + text_deltas = [r.choices[0].delta.content for r in results[:2]] + assert text_deltas == ["Fetching ", "the weather."] + + tool_added = results[3].choices[0].delta.tool_calls + assert tool_added and tool_added[0]["function"]["name"] == "get_weather" + + tool_arg_delta = results[4].choices[0].delta.tool_calls + assert ( + tool_arg_delta + and tool_arg_delta[0]["function"]["arguments"] == '{"location":"Paris"}' + ) + + terminal = results[-1] + assert terminal.choices[0].finish_reason == "tool_calls" + + def test_streaming_chunks_share_one_chat_completion_id(): """Every chunk of one streamed chat completion must carry the same ``id``, per the OpenAI spec. The bridge builds a fresh ``ModelResponseStream`` per Responses event, From 8d9f30b63e7aaff4366df519cc96ed03fd0fff5b Mon Sep 17 00:00:00 2001 From: Stephen Chin Date: Sun, 19 Jul 2026 09:28:42 -0700 Subject: [PATCH 2/5] fix(responses): preserve reasoning_items on merge I found a follow-on gap in the same merge path from my earlier fix. When reasoning arrives after the assistant message in the output list (message, reasoning, function_call), the merge block that attaches accumulated tool_calls onto the last message choice backfilled reasoning_content but not reasoning_items. That silently dropped encrypted_content, which the provider needs to round-trip reasoning on the next turn. I added a symmetric backfill for reasoning_items right next to the existing reasoning_content backfill, and kept the regression test that exercises the message-then-reasoning-then-function_call ordering. Also cleaned up unused imports, a duplicate import, and leftover print statements in the test file while I was in there. --- ...responses_transformation_transformation.py | 209 ++++++++++++++---- 1 file changed, 169 insertions(+), 40 deletions(-) diff --git a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index eedbdf3c600..682eab957be 100644 --- a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -1,11 +1,9 @@ -import datetime import json import os import unittest from typing import TYPE_CHECKING, Final, List, Literal, Optional, Tuple, get_args from unittest.mock import ANY, MagicMock, Mock, patch -import httpx import pytest import litellm @@ -55,7 +53,6 @@ def test_convert_chat_completion_messages_to_responses_api_image_input(): assert user_content in response_str assert user_image in response_str - print("response: ", response) assert response[0]["content"][1]["image_url"] == user_image @@ -146,8 +143,6 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_imag ), "image_url should be a flat string, not a nested object" assert "detail" in image_item, "detail field should be present" - print("✓ Tool result with image correctly transformed to Responses API format") - def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text(): """ @@ -224,10 +219,6 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text text_item["text"] == "15 degrees" ), f"Expected text '15 degrees', got '{text_item.get('text')}'" - print( - "✓ Tool result with text correctly transformed to use input_text for Responses API format" - ) - def test_openai_responses_chunk_parser_reasoning_summary(): from litellm.completion_extras.litellm_responses_transformation.transformation import ( @@ -512,8 +503,6 @@ and I learn to carry this small calm home.""" # Check reasoning content assert choice.message.reasoning_content == reasoning_summary.text - print("✓ transform_response correctly handled reasoning items and output messages") - def _make_empty_responses_api_response(model: str = "gpt-5.4"): from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse @@ -983,10 +972,6 @@ def test_transform_request_single_char_keys_not_matched(): assert result_correct.get("metadata") == {"user_id": "123"} assert result_correct.get("previous_response_id") == "resp_abc" - print( - "✓ Single-character keys are not incorrectly matched to metadata/previous_response_id" - ) - # ============================================================================= # Tests for issue #17246: Streaming tool_calls dropped when text + tool_calls @@ -1390,8 +1375,6 @@ def test_tool_message_output_uses_input_text_not_output_text(): ), f"Expected input_text, got {output[0].get('type')}" assert output[0]["text"] == '{"temperature": 15, "condition": "sunny"}' - print("✓ Tool message output correctly uses input_text type") - def test_multiple_tool_calls_in_single_choice(): """ @@ -1533,8 +1516,6 @@ def test_multiple_tool_calls_in_single_choice(): assert tool_calls[2]["id"] == "call_horoscope" assert tool_calls[2]["function"]["name"] == "get_horoscope" - print("✓ Multiple tool calls are correctly grouped in a single choice") - def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): """ @@ -1547,7 +1528,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): When flag is enabled (flag=True or env var), summary="detailed" is added. """ - import litellm from litellm.completion_extras.litellm_responses_transformation.transformation import ( LiteLLMResponsesTransformationHandler, ) @@ -1576,9 +1556,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): "summary" not in result ), f"Summary should NOT be present by default for effort={effort}" - print( - f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}' (no summary by default)" - ) # Test 2: With flag enabled - summary IS added litellm.reasoning_auto_summary = True @@ -1592,9 +1569,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): result["summary"] == "detailed" ), f"Summary should be 'detailed' when flag is enabled for effort={effort}" - print( - f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}', summary='detailed' (flag enabled)" - ) # Test 3: With env var enabled (flag disabled) - summary IS added litellm.reasoning_auto_summary = False @@ -1604,7 +1578,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): assert ( result["summary"] == "detailed" ), "Summary should be 'detailed' when env var is enabled" - print("✓ LITELLM_REASONING_AUTO_SUMMARY env var works correctly") # Test 4: Dict input is passed through as-is (no modification) litellm.reasoning_auto_summary = False @@ -1615,7 +1588,16 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): result_dict = handler._map_reasoning_effort(dict_input) assert result_dict["effort"] == "high" assert result_dict["summary"] == "custom_summary" - print("✓ Dict input is passed through without modification") + + # Test 5: every REASONING_EFFORT level reaches the provider, and anything else (a typo, an + # unshipped level, "default") is dropped so the request still succeeds at the provider default + from litellm.types.llms.openai import Reasoning + + for effort in ("max", "xhigh", "none"): + result_passthrough = handler._map_reasoning_effort(effort) + assert result_passthrough == Reasoning(effort=effort) + for dropped in ("ultra", "hgih", "unknown_value", "", "default"): + assert handler._map_reasoning_effort(dropped) is None print( "✓ All reasoning_effort behaviors work correctly with flag/env var control" @@ -1812,10 +1794,6 @@ def test_transform_response_preserves_annotations(): assert result.usage.completion_tokens == 20 assert result.usage.total_tokens == 30 - print( - "✓ Annotations from Responses API are correctly preserved in Chat Completions format" - ) - def test_apply_patch_tool_call_converted_to_chat_completion_tool_call(): """ @@ -2109,10 +2087,6 @@ def test_multi_tool_call_stream_no_premature_finish(): f"— only response.completed should terminate the stream" ) - print( - "✓ Multi-tool-call stream completes without premature finish_reason termination" - ) - # ============================================================================= # Tests for issue #21331: Parallel tool call indices in streaming @@ -2425,10 +2399,6 @@ def test_parallel_tool_calls_comprehensive_streaming_integration(): 1, }, f"Parallel tool calls must have distinct indices {{0, 1}}, got: {set(added_tool_call_indices)}" - print( - "✓ Parallel tool calls with split argument deltas stream correctly end-to-end" - ) - def test_map_optional_params_preserves_reasoning_summary(): """Test that reasoning_effort dict with summary field is preserved. @@ -2911,6 +2881,165 @@ def test_reasoning_items_streaming_emitted_on_response_completed(): assert ri["summary"][0]["text"] == summary_text +def test_reasoning_items_preserved_when_merged_with_tool_calls(): + """ + Regression: when a Responses turn contains [message, reasoning, function_call] + (reasoning item arriving AFTER the assistant message), the merge path that + attaches accumulated tool_calls onto the last message choice must also backfill + the structured ``reasoning_items`` (with ``encrypted_content``) onto that + message; otherwise the encrypted reasoning payload needed to round-trip on the + next turn is silently dropped. The flattened ``reasoning_content`` string is + already backfilled by the existing code; ``reasoning_items`` currently is not. + """ + from unittest.mock import Mock + + from openai.types.responses import ( + ResponseFunctionToolCall, + ResponseOutputMessage, + ResponseOutputText, + ) + from openai.types.responses.response_reasoning_item import ( + ResponseReasoningItem, + Summary, + ) + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + from litellm.types.llms.openai import ( + InputTokensDetails, + OutputTokensDetails, + ResponseAPIUsage, + ResponsesAPIResponse, + ) + from litellm.types.utils import ModelResponse, Usage + + handler = LiteLLMResponsesTransformationHandler() + + encrypted = "gAAAAABpw5xyz789FAKE==" + summary_text = "deciding which city" + preamble = "Let me check the weather." + + output_message = ResponseOutputMessage( + id="msg_merge001", + content=[ + ResponseOutputText( + annotations=[], + text=preamble, + type="output_text", + logprobs=[], + ) + ], + role="assistant", + status="completed", + type="message", + ) + reasoning_item = ResponseReasoningItem( + id="rs_merge001", + summary=[Summary(text=summary_text, type="summary_text")], + type="reasoning", + content=None, + encrypted_content=encrypted, + status=None, + ) + function_call_item = ResponseFunctionToolCall( + id="fc_merge001", + type="function_call", + status="completed", + arguments='{"city": "Paris"}', + call_id="call_paris_merge", + name="get_weather", + ) + + usage = ResponseAPIUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + audio_tokens=None, cached_tokens=0, text_tokens=None + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None), + total_tokens=30, + cost=None, + ) + raw_response = ResponsesAPIResponse( + id="resp_merge001", + created_at=1234567890, + error=None, + incomplete_details=None, + instructions=None, + metadata={}, + model="gpt-5-mini", + object="response", + output=[output_message, reasoning_item, function_call_item], + parallel_tool_calls=True, + temperature=1.0, + tool_choice="auto", + tools=[], + top_p=1.0, + max_output_tokens=None, + previous_response_id=None, + reasoning={"effort": "low", "summary": "detailed"}, + status="completed", + text={"format": {"type": "text"}, "verbosity": "medium"}, + truncation="disabled", + usage=usage, + user=None, + store=True, + background=False, + billing={"payer": "developer"}, + max_tool_calls=None, + prompt_cache_key=None, + safety_identifier=None, + service_tier="default", + top_logprobs=0, + ) + model_response = ModelResponse( + id="chatcmpl-merge001", + created=1234567890, + model=None, + object="chat.completion", + system_fingerprint=None, + choices=[], + usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0), + ) + + result = handler.transform_response( + model="gpt-5-mini", + raw_response=raw_response, + model_response=model_response, + logging_obj=Mock(), + request_data={"model": "gpt-5-mini"}, + messages=[{"role": "user", "content": "What's the weather in Paris?"}], + optional_params={}, + litellm_params={}, + encoding=Mock(), + ) + + assert len(result.choices) == 1, ( + f"merge must collapse into a single choice, got {len(result.choices)}" + ) + choice = result.choices[0] + msg = choice.message + + assert msg.tool_calls is not None and len(msg.tool_calls) == 1 + assert msg.tool_calls[0]["function"]["name"] == "get_weather" + assert msg.tool_calls[0]["function"]["arguments"] == '{"city": "Paris"}' + assert choice.finish_reason == "tool_calls" + + assert msg.content == preamble, "assistant preamble text must be preserved" + + assert msg.reasoning_content == summary_text, ( + "reasoning_content must be backfilled onto the merged choice" + ) + + assert msg.reasoning_items is not None, ( + "reasoning_items must be backfilled onto the merged choice so encrypted_content " + "can round-trip to the provider on the next turn" + ) + assert len(msg.reasoning_items) == 1 + assert msg.reasoning_items[0]["encrypted_content"] == encrypted + + def test_streaming_function_call_tool_id_for_degenerate_call_id(): """In streaming, Bedrock Mantle's function_call event carries a unique ``id`` (``fc_...``) and a non-unique, index-based ``call_id`` (``call_0``). For that From d7b2e0ca199a28cfa35d832dc55362d799801f80 Mon Sep 17 00:00:00 2001 From: Stephen Chin Date: Thu, 6 Aug 2026 19:30:49 +0000 Subject: [PATCH 3/5] test(responses): restore patch import lost in cleanup I found test_acompletion_bridge_normalizes_stream_options_on_the_wire failing with NameError: name 'patch' is not defined. An earlier unused-imports cleanup on this branch dropped patch from the unittest.mock import line even though two tests still call patch.object(...). I added patch back, plus MagicMock and the httpx import that the same tests need, and left ANY out since nothing in the file uses it anymore. All 71 tests in this file pass again. --- ...ion_extras_litellm_responses_transformation_transformation.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index 682eab957be..ed6bc06183a 100644 --- a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -4,6 +4,7 @@ import unittest from typing import TYPE_CHECKING, Final, List, Literal, Optional, Tuple, get_args from unittest.mock import ANY, MagicMock, Mock, patch +import httpx import pytest import litellm From 834ab664cd0efa86c86e192ebef346eea212963f Mon Sep 17 00:00:00 2001 From: Stephen Chin Date: Thu, 6 Aug 2026 19:31:46 +0000 Subject: [PATCH 4/5] fix(responses): extract choice helpers to cut complexity I split _convert_response_output_to_choices into two static helpers. One builds Choices from a ResponseOutputMessage's content blocks, the other merges accumulated tool_calls into the choice list. This drops the function's cyclomatic complexity from 17 to 12, back under the ruff-strict C901 ceiling of 15 that scripts/ruff_strict_gate.py checks against upstream/litellm_internal_staging. Behavior is unchanged. All 71 existing tests in test_completion_extras_litellm_responses_transformation_transformation.py still pass. --- .../transformation.py | 193 +++++++++++------- 1 file changed, 119 insertions(+), 74 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 620dbde2312..3ec29b6387a 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -51,7 +51,11 @@ from litellm.types.llms.openai import ( from litellm.types.utils import GenericStreamingChunk, ModelResponseStream if TYPE_CHECKING: - from openai.types.responses import ResponseInputImageParam, ResponseOutputItem + from openai.types.responses import ( + ResponseInputImageParam, + ResponseOutputItem, + ResponseOutputMessage, + ) from openai.types.responses.response_text_config_param import ( ResponseTextConfigParam as ResponseText, ) @@ -658,6 +662,104 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): return request_data + @staticmethod + def _choices_from_response_output_message( + item: "ResponseOutputMessage", + starting_index: int, + reasoning_content: str | None, + pending_reasoning_item: dict[str, Any] | None, + ) -> tuple[list[Any], int]: + """Build one Choices per content block on a ResponseOutputMessage. + + The first emitted choice carries the pending reasoning (content + items); + any subsequent content blocks emit choices with reasoning fields cleared, + matching the flush semantics of the original inline loop. + """ + from litellm.types.utils import Choices, Message + + new_choices: Final[list[Any]] = [] + current_index = starting_index + carry_reasoning_content = reasoning_content + carry_reasoning_item = pending_reasoning_item + for content in item.content: + response_text = getattr(content, "text", "") + raw_annotations = getattr(content, "annotations", None) + annotations = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format(raw_annotations) + reasoning_items_for_msg = cast( + list[ChatCompletionReasoningItem] | None, + ([carry_reasoning_item] if carry_reasoning_item is not None else None), + ) + msg = Message( + role=item.role, + content=response_text or "", + reasoning_content=carry_reasoning_content, + annotations=annotations, + reasoning_items=reasoning_items_for_msg, + ) + new_choices.append(Choices(message=msg, finish_reason="stop", index=current_index)) + carry_reasoning_content = None + carry_reasoning_item = None + current_index += 1 + return new_choices, current_index + + @staticmethod + def _merge_accumulated_tool_calls_into_choices( + choices: list[Any], + accumulated_tool_calls: list[dict[str, Any]], + reasoning_content: str | None, + pending_reasoning_item: dict[str, Any] | None, + fallback_index: int, + ) -> None: + """Attach accumulated tool_calls to the last text-message choice, or + append a new tool-only choice if no text message was produced. + + Backfills reasoning_content and reasoning_items onto the merged choice + so encrypted reasoning survives the assistant+tool_calls merge path. + """ + from litellm.types.utils import Choices, Message + + last_msg_choice = next( + ( + c + for c in reversed(choices) + if getattr(c, "message", None) is not None and not getattr(c.message, "tool_calls", None) + ), + None, + ) + if last_msg_choice is None: + reasoning_items_for_msg = cast( + list[ChatCompletionReasoningItem] | None, + ([pending_reasoning_item] if pending_reasoning_item is not None else None), + ) + msg = Message( + content=None, + tool_calls=accumulated_tool_calls, + reasoning_content=reasoning_content, + reasoning_items=reasoning_items_for_msg, + ) + choices.append(Choices(message=msg, finish_reason="tool_calls", index=fallback_index)) + return + + last_msg_choice.message.tool_calls = accumulated_tool_calls + if getattr(last_msg_choice.message, "content", None) is None: + last_msg_choice.message.content = "" + last_msg_choice.finish_reason = "tool_calls" + needs_reasoning_content_backfill = ( + reasoning_content is not None + and getattr(last_msg_choice.message, "reasoning_content", None) is None + ) + if needs_reasoning_content_backfill: + last_msg_choice.message.reasoning_content = reasoning_content + needs_reasoning_items_backfill = ( + pending_reasoning_item is not None + and getattr(last_msg_choice.message, "reasoning_items", None) is None + ) + if needs_reasoning_items_backfill: + last_msg_choice.message.reasoning_items = cast( + list[ChatCompletionReasoningItem] | None, + [pending_reasoning_item], + ) + @staticmethod def _convert_response_output_to_choices( output_items: Sequence[object], @@ -686,9 +788,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): except ImportError: ResponseApplyPatchToolCall = None - from litellm.types.utils import Choices, Message - - choices: Final[list[Choices]] = [] + choices: Final[list[Any]] = [] index = 0 reasoning_content: str | None = None pending_reasoning_item: _BuiltReasoningItem | None = None @@ -708,35 +808,15 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): reasoning_content = " ".join(s["text"] for s in pending_reasoning_item["summary"] if s.get("text")) elif isinstance(item, ResponseOutputMessage): - for content in item.content: - response_text = getattr(content, "text", "") - # Extract annotations from content if present - raw_annotations = getattr(content, "annotations", None) - annotations = LiteLLMResponsesTransformationHandler._convert_annotations_to_chat_format( - raw_annotations - ) - msg = Message( - role=item.role, - content=response_text if response_text else "", - reasoning_content=reasoning_content, - annotations=annotations, - reasoning_items=cast( - list[ChatCompletionReasoningItem] | None, - ([pending_reasoning_item] if pending_reasoning_item is not None else None), - ), - ) - - choices.append( - Choices( - message=msg, - finish_reason="stop", - index=index, - ) - ) - - reasoning_content = None # flush - pending_reasoning_item = None # flush - index += 1 + new_choices, index = LiteLLMResponsesTransformationHandler._choices_from_response_output_message( + item=item, + starting_index=index, + reasoning_content=reasoning_content, + pending_reasoning_item=pending_reasoning_item, + ) + choices.extend(new_choices) + reasoning_content = None + pending_reasoning_item = None elif isinstance(item, ResponseFunctionToolCall): from litellm.responses.litellm_completion_transformation.transformation import ( @@ -785,48 +865,13 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): pass # don't fail request if item in list is not supported if accumulated_tool_calls: - last_msg_choice = next( - ( - c - for c in reversed(choices) - if getattr(c, "message", None) is not None and not getattr(c.message, "tool_calls", None) - ), - None, + LiteLLMResponsesTransformationHandler._merge_accumulated_tool_calls_into_choices( + choices=choices, + accumulated_tool_calls=accumulated_tool_calls, + reasoning_content=reasoning_content, + pending_reasoning_item=pending_reasoning_item, + fallback_index=index, ) - if last_msg_choice is not None: - last_msg_choice.message.tool_calls = accumulated_tool_calls - if getattr(last_msg_choice.message, "content", None) is None: - last_msg_choice.message.content = "" - last_msg_choice.finish_reason = "tool_calls" - if ( - reasoning_content is not None - and getattr(last_msg_choice.message, "reasoning_content", None) is None - ): - last_msg_choice.message.reasoning_content = reasoning_content - # Mirror the reasoning_content backfill above: reasoning_items carries - # encrypted_content, needed to round-trip reasoning to the provider on - # the next turn, and was being dropped on this merge path. - if ( - pending_reasoning_item is not None - and getattr(last_msg_choice.message, "reasoning_items", None) is None - ): - last_msg_choice.message.reasoning_items = cast( - list[ChatCompletionReasoningItem] | None, - [pending_reasoning_item], - ) - else: - msg = Message( - content=None, - tool_calls=accumulated_tool_calls, - reasoning_content=reasoning_content, - reasoning_items=cast( - list[ChatCompletionReasoningItem] | None, - ([pending_reasoning_item] if pending_reasoning_item is not None else None), - ), - ) - choices.append(Choices(message=msg, finish_reason="tool_calls", index=index)) - reasoning_content = None - pending_reasoning_item = None return choices From 8c1e8510e06b0b88f6fe7bc4777cef142eb4d3a9 Mon Sep 17 00:00:00 2001 From: Stephen Chin <1290231+steveonjava@users.noreply.github.com> Date: Thu, 24 Sep 2026 01:11:39 -0700 Subject: [PATCH 5/5] test(responses): align merged choice coverage with main --- .../transformation.py | 6 ++--- ...responses_transformation_transformation.py | 23 ++++--------------- 2 files changed, 6 insertions(+), 23 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 3ec29b6387a..4a77a638709 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -745,14 +745,12 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): last_msg_choice.message.content = "" last_msg_choice.finish_reason = "tool_calls" needs_reasoning_content_backfill = ( - reasoning_content is not None - and getattr(last_msg_choice.message, "reasoning_content", None) is None + reasoning_content is not None and getattr(last_msg_choice.message, "reasoning_content", None) is None ) if needs_reasoning_content_backfill: last_msg_choice.message.reasoning_content = reasoning_content needs_reasoning_items_backfill = ( - pending_reasoning_item is not None - and getattr(last_msg_choice.message, "reasoning_items", None) is None + pending_reasoning_item is not None and getattr(last_msg_choice.message, "reasoning_items", None) is None ) if needs_reasoning_items_backfill: last_msg_choice.message.reasoning_items = cast( diff --git a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index ed6bc06183a..f688c014ff7 100644 --- a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -1,8 +1,7 @@ import json import os -import unittest -from typing import TYPE_CHECKING, Final, List, Literal, Optional, Tuple, get_args -from unittest.mock import ANY, MagicMock, Mock, patch +from typing import TYPE_CHECKING, Final, Literal, get_args +from unittest.mock import MagicMock, Mock, patch import httpx import pytest @@ -1590,20 +1589,6 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): assert result_dict["effort"] == "high" assert result_dict["summary"] == "custom_summary" - # Test 5: every REASONING_EFFORT level reaches the provider, and anything else (a typo, an - # unshipped level, "default") is dropped so the request still succeeds at the provider default - from litellm.types.llms.openai import Reasoning - - for effort in ("max", "xhigh", "none"): - result_passthrough = handler._map_reasoning_effort(effort) - assert result_passthrough == Reasoning(effort=effort) - for dropped in ("ultra", "hgih", "unknown_value", "", "default"): - assert handler._map_reasoning_effort(dropped) is None - - print( - "✓ All reasoning_effort behaviors work correctly with flag/env var control" - ) - finally: # Restore original values litellm.reasoning_auto_summary = original_flag @@ -4011,8 +3996,8 @@ async def test_acompletion_bridge_normalizes_tool_choice_on_the_wire( def _make_incomplete_responses_api_response( - incomplete_reason: Optional[str], - output: "List[ResponseOutputItem]", + incomplete_reason: str | None, + output: "list[ResponseOutputItem]", status: Literal["completed", "incomplete"] = "incomplete", empty_incomplete_details: bool = False, ) -> "ResponsesAPIResponse":