From 138b415e8133e9dc98e65f2f51ce7e9e1e7a47fd Mon Sep 17 00:00:00 2001 From: Cesar Garcia <128240629+Chesars@users.noreply.github.com> Date: Sat, 20 Dec 2025 05:16:14 -0300 Subject: [PATCH] fix(responses-api): use list format with input_text for tool results (#18257) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Responses API expects tool results to use input_text/input_image types, not output_text. This fix ensures consistent list format for all tool results: - String content → [{"type": "input_text", "text": "..."}] - Image content → [{"type": "input_image", "image_url": "..."}] This resolves the conflict between tests that expected different formats and aligns with OpenAI's Responses API requirements. Fixes the regression introduced in #18226. --- .../transformation.py | 27 +++++++++---------- ...responses_transformation_transformation.py | 18 +++++++------ 2 files changed, 23 insertions(+), 22 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 6f9aa192f9b..55a8e665bbd 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -167,28 +167,27 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): ) elif role == "tool": # Convert tool message to function call output format - # The Responses API expects 'output' to be a string, not a list + # The Responses API expects 'output' to be a list with input_text/input_image types + # Using list format for consistency across text and multimodal content + tool_output: List[Dict[str, Any]] if content is None: - output_str = "" + tool_output = [] elif isinstance(content, str): - output_str = content + # Convert string to list with input_text + tool_output = [{"type": "input_text", "text": content}] elif isinstance(content, list): - # If content is a list, extract text parts and join them - text_parts = [] - for item in content: - if isinstance(item, str): - text_parts.append(item) - elif isinstance(item, dict) and item.get("type") == "text": - text_parts.append(item.get("text", "")) - output_str = " ".join(text_parts) if text_parts else str(content) + # Transform list content to Responses API format + tool_output = self._convert_content_to_responses_format( + content, "user" # Use "user" role to get input_* types + ) else: - # Fallback: convert unexpected types to string - output_str = str(content) + # Fallback: convert unexpected types to input_text + tool_output = [{"type": "input_text", "text": str(content)}] input_items.append( { "type": "function_call_output", "call_id": tool_call_id, - "output": output_str, + "output": tool_output, } ) elif role == "assistant" and tool_calls and isinstance(tool_calls, list): diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index 2e7a64df8be..4320c932f41 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -802,15 +802,15 @@ def test_text_plus_tool_calls_sequence(): # ============================================================================= -def test_tool_message_output_is_string_not_list(): +def test_tool_message_output_uses_input_text_not_output_text(): """ - Test that tool message content is converted to a string, not a list. + Test that tool message content uses input_text type, not output_text. This is a regression test for a bug where tool results were transformed to: {"type": "function_call_output", "output": [{"type": "output_text", "text": "..."}]} - But the Responses API expects: - {"type": "function_call_output", "output": "..."} + But the Responses API expects input_text for tool results: + {"type": "function_call_output", "output": [{"type": "input_text", "text": "..."}]} The incorrect format caused OpenAI to reject with: "Invalid value: 'output_text'. Supported values are: 'input_text', 'input_image', and 'input_file'." @@ -856,12 +856,14 @@ def test_tool_message_output_is_string_not_list(): assert function_call_output is not None, "function_call_output not found" assert function_call_output["call_id"] == "call_abc123" - # The output should be a string, NOT a list + # The output should be a list with input_text type output = function_call_output["output"] - assert isinstance(output, str), f"output should be a string, got {type(output)}" - assert output == '{"temperature": 15, "condition": "sunny"}' + assert isinstance(output, list), f"output should be a list, got {type(output)}" + assert len(output) == 1 + assert output[0]["type"] == "input_text", f"Expected input_text, got {output[0].get('type')}" + assert output[0]["text"] == '{"temperature": 15, "condition": "sunny"}' - print("✓ Tool message output is correctly a string, not a list") + print("✓ Tool message output correctly uses input_text type") def test_multiple_tool_calls_in_single_choice():