diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index fdd219365be..06647b2b66b 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -1,12 +1,8 @@ -import datetime import json import os import sys -import unittest -from typing import List, Optional, Tuple -from unittest.mock import ANY, MagicMock, Mock, patch +from unittest.mock import Mock -import httpx import pytest sys.path.insert( @@ -51,7 +47,6 @@ def test_convert_chat_completion_messages_to_responses_api_image_input(): assert user_content in response_str assert user_image in response_str - print("response: ", response) assert response[0]["content"][1]["image_url"] == user_image @@ -142,8 +137,6 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_imag ), "image_url should be a flat string, not a nested object" assert "detail" in image_item, "detail field should be present" - print("✓ Tool result with image correctly transformed to Responses API format") - def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text(): """ @@ -220,10 +213,6 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text text_item["text"] == "15 degrees" ), f"Expected text '15 degrees', got '{text_item.get('text')}'" - print( - "✓ Tool result with text correctly transformed to use input_text for Responses API format" - ) - def test_openai_responses_chunk_parser_reasoning_summary(): from litellm.completion_extras.litellm_responses_transformation.transformation import ( @@ -508,8 +497,6 @@ and I learn to carry this small calm home.""" # Check reasoning content assert choice.message.reasoning_content == reasoning_summary.text - print("✓ transform_response correctly handled reasoning items and output messages") - def _make_empty_responses_api_response(model: str = "gpt-5.4"): from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse @@ -961,10 +948,6 @@ def test_transform_request_single_char_keys_not_matched(): assert result_correct.get("metadata") == {"user_id": "123"} assert result_correct.get("previous_response_id") == "resp_abc" - print( - "✓ Single-character keys are not incorrectly matched to metadata/previous_response_id" - ) - # ============================================================================= # Tests for issue #17246: Streaming tool_calls dropped when text + tool_calls @@ -1355,8 +1338,6 @@ def test_tool_message_output_uses_input_text_not_output_text(): ), f"Expected input_text, got {output[0].get('type')}" assert output[0]["text"] == '{"temperature": 15, "condition": "sunny"}' - print("✓ Tool message output correctly uses input_text type") - def test_multiple_tool_calls_in_single_choice(): """ @@ -1498,8 +1479,6 @@ def test_multiple_tool_calls_in_single_choice(): assert tool_calls[2]["id"] == "call_horoscope" assert tool_calls[2]["function"]["name"] == "get_horoscope" - print("✓ Multiple tool calls are correctly grouped in a single choice") - def test_map_reasoning_effort_adds_summary_detailed(): """ @@ -1513,7 +1492,6 @@ def test_map_reasoning_effort_adds_summary_detailed(): """ import os - import litellm from litellm.completion_extras.litellm_responses_transformation.transformation import ( LiteLLMResponsesTransformationHandler, ) @@ -1542,9 +1520,6 @@ def test_map_reasoning_effort_adds_summary_detailed(): "summary" not in result ), f"Summary should NOT be present by default for effort={effort}" - print( - f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}' (no summary by default)" - ) # Test 2: With flag enabled - summary IS added litellm.reasoning_auto_summary = True @@ -1558,9 +1533,6 @@ def test_map_reasoning_effort_adds_summary_detailed(): result["summary"] == "detailed" ), f"Summary should be 'detailed' when flag is enabled for effort={effort}" - print( - f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}', summary='detailed' (flag enabled)" - ) # Test 3: With env var enabled (flag disabled) - summary IS added litellm.reasoning_auto_summary = False @@ -1570,7 +1542,6 @@ def test_map_reasoning_effort_adds_summary_detailed(): assert ( result["summary"] == "detailed" ), "Summary should be 'detailed' when env var is enabled" - print("✓ LITELLM_REASONING_AUTO_SUMMARY env var works correctly") # Test 4: Dict input is passed through as-is (no modification) litellm.reasoning_auto_summary = False @@ -1581,16 +1552,11 @@ def test_map_reasoning_effort_adds_summary_detailed(): result_dict = handler._map_reasoning_effort(dict_input) assert result_dict["effort"] == "high" assert result_dict["summary"] == "custom_summary" - print("✓ Dict input is passed through without modification") # Test 5: None/unknown values return None result_unknown = handler._map_reasoning_effort("unknown_value") assert result_unknown is None - print("✓ Unknown reasoning_effort values return None") - print( - "✓ All reasoning_effort behaviors work correctly with flag/env var control" - ) finally: # Restore original values @@ -1783,10 +1749,6 @@ def test_transform_response_preserves_annotations(): assert result.usage.completion_tokens == 20 assert result.usage.total_tokens == 30 - print( - "✓ Annotations from Responses API are correctly preserved in Chat Completions format" - ) - def test_apply_patch_tool_call_converted_to_chat_completion_tool_call(): """ @@ -2080,10 +2042,6 @@ def test_multi_tool_call_stream_no_premature_finish(): f"— only response.completed should terminate the stream" ) - print( - "✓ Multi-tool-call stream completes without premature finish_reason termination" - ) - # ============================================================================= # Tests for issue #21331: Parallel tool call indices in streaming @@ -2396,10 +2354,6 @@ def test_parallel_tool_calls_comprehensive_streaming_integration(): 1, }, f"Parallel tool calls must have distinct indices {{0, 1}}, got: {set(added_tool_call_indices)}" - print( - "✓ Parallel tool calls with split argument deltas stream correctly end-to-end" - ) - def test_map_optional_params_preserves_reasoning_summary(): """Test that reasoning_effort dict with summary field is preserved. @@ -2830,6 +2784,165 @@ def test_reasoning_items_streaming_emitted_on_response_completed(): assert ri["summary"][0]["text"] == summary_text +def test_reasoning_items_preserved_when_merged_with_tool_calls(): + """ + Regression: when a Responses turn contains [message, reasoning, function_call] + (reasoning item arriving AFTER the assistant message), the merge path that + attaches accumulated tool_calls onto the last message choice must also backfill + the structured ``reasoning_items`` (with ``encrypted_content``) onto that + message; otherwise the encrypted reasoning payload needed to round-trip on the + next turn is silently dropped. The flattened ``reasoning_content`` string is + already backfilled by the existing code; ``reasoning_items`` currently is not. + """ + from unittest.mock import Mock + + from openai.types.responses import ( + ResponseFunctionToolCall, + ResponseOutputMessage, + ResponseOutputText, + ) + from openai.types.responses.response_reasoning_item import ( + ResponseReasoningItem, + Summary, + ) + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + from litellm.types.llms.openai import ( + InputTokensDetails, + OutputTokensDetails, + ResponseAPIUsage, + ResponsesAPIResponse, + ) + from litellm.types.utils import ModelResponse, Usage + + handler = LiteLLMResponsesTransformationHandler() + + encrypted = "gAAAAABpw5xyz789FAKE==" + summary_text = "deciding which city" + preamble = "Let me check the weather." + + output_message = ResponseOutputMessage( + id="msg_merge001", + content=[ + ResponseOutputText( + annotations=[], + text=preamble, + type="output_text", + logprobs=[], + ) + ], + role="assistant", + status="completed", + type="message", + ) + reasoning_item = ResponseReasoningItem( + id="rs_merge001", + summary=[Summary(text=summary_text, type="summary_text")], + type="reasoning", + content=None, + encrypted_content=encrypted, + status=None, + ) + function_call_item = ResponseFunctionToolCall( + id="fc_merge001", + type="function_call", + status="completed", + arguments='{"city": "Paris"}', + call_id="call_paris_merge", + name="get_weather", + ) + + usage = ResponseAPIUsage( + input_tokens=10, + input_tokens_details=InputTokensDetails( + audio_tokens=None, cached_tokens=0, text_tokens=None + ), + output_tokens=20, + output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None), + total_tokens=30, + cost=None, + ) + raw_response = ResponsesAPIResponse( + id="resp_merge001", + created_at=1234567890, + error=None, + incomplete_details=None, + instructions=None, + metadata={}, + model="gpt-5-mini", + object="response", + output=[output_message, reasoning_item, function_call_item], + parallel_tool_calls=True, + temperature=1.0, + tool_choice="auto", + tools=[], + top_p=1.0, + max_output_tokens=None, + previous_response_id=None, + reasoning={"effort": "low", "summary": "detailed"}, + status="completed", + text={"format": {"type": "text"}, "verbosity": "medium"}, + truncation="disabled", + usage=usage, + user=None, + store=True, + background=False, + billing={"payer": "developer"}, + max_tool_calls=None, + prompt_cache_key=None, + safety_identifier=None, + service_tier="default", + top_logprobs=0, + ) + model_response = ModelResponse( + id="chatcmpl-merge001", + created=1234567890, + model=None, + object="chat.completion", + system_fingerprint=None, + choices=[], + usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0), + ) + + result = handler.transform_response( + model="gpt-5-mini", + raw_response=raw_response, + model_response=model_response, + logging_obj=Mock(), + request_data={"model": "gpt-5-mini"}, + messages=[{"role": "user", "content": "What's the weather in Paris?"}], + optional_params={}, + litellm_params={}, + encoding=Mock(), + ) + + assert len(result.choices) == 1, ( + f"merge must collapse into a single choice, got {len(result.choices)}" + ) + choice = result.choices[0] + msg = choice.message + + assert msg.tool_calls is not None and len(msg.tool_calls) == 1 + assert msg.tool_calls[0]["function"]["name"] == "get_weather" + assert msg.tool_calls[0]["function"]["arguments"] == '{"city": "Paris"}' + assert choice.finish_reason == "tool_calls" + + assert msg.content == preamble, "assistant preamble text must be preserved" + + assert msg.reasoning_content == summary_text, ( + "reasoning_content must be backfilled onto the merged choice" + ) + + assert msg.reasoning_items is not None, ( + "reasoning_items must be backfilled onto the merged choice so encrypted_content " + "can round-trip to the provider on the next turn" + ) + assert len(msg.reasoning_items) == 1 + assert msg.reasoning_items[0]["encrypted_content"] == encrypted + + def test_streaming_function_call_tool_id_for_degenerate_call_id(): """In streaming, Bedrock Mantle's function_call event carries a unique ``id`` (``fc_...``) and a non-unique, index-based ``call_id`` (``call_0``). For that