diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 93d79bb3ac8..3fc8678d35f 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -695,7 +695,10 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): choices: Final[list[Choices]] = [] index = 0 reasoning_content: str | None = None - pending_reasoning_item: _BuiltReasoningItem | None = None + # Tuple accumulator: the Responses API can emit several reasoning items + # before the next message / tool call, and each flush below consumes + # every item seen since the previous flush. + pending_reasoning_items: tuple[_BuiltReasoningItem, ...] = () # Collect all tool calls to put them in a single choice # (Chat Completions API expects all tool calls in one message) @@ -704,12 +707,13 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): for item in output_items: if isinstance(item, ResponseReasoningItem): - pending_reasoning_item = _build_reasoning_item( + built_reasoning_item: Final = _build_reasoning_item( item_id=item.id, encrypted_content=getattr(item, "encrypted_content", None), summary_raw=item.summary, ) - reasoning_content = " ".join(s["text"] for s in pending_reasoning_item["summary"] if s.get("text")) + pending_reasoning_items = (*pending_reasoning_items, built_reasoning_item) + reasoning_content = " ".join(s["text"] for s in built_reasoning_item["summary"] if s.get("text")) elif isinstance(item, ResponseOutputMessage): for content in item.content: @@ -724,10 +728,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): content=response_text if response_text else "", reasoning_content=reasoning_content, annotations=annotations, - reasoning_items=cast( - list[ChatCompletionReasoningItem] | None, - ([pending_reasoning_item] if pending_reasoning_item is not None else None), - ), + reasoning_items=_as_chat_reasoning_items(pending_reasoning_items), ) choices.append( @@ -739,7 +740,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): ) reasoning_content = None # flush - pending_reasoning_item = None # flush + pending_reasoning_items = () # flush index += 1 elif isinstance(item, ResponseFunctionToolCall): @@ -796,10 +797,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): " ".join(value for value in (last_reasoning_content, reasoning_content) if value) or None ) merged_reasoning_items: Final = _as_chat_reasoning_items( - ( - *(last_reasoning_items or ()), - *(() if pending_reasoning_item is None else (pending_reasoning_item,)), - ) + (*(last_reasoning_items or ()), *pending_reasoning_items) ) merged_message: Final = Message( role=last_choice.message.role, @@ -819,14 +817,11 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): content=None, tool_calls=accumulated_tool_calls, reasoning_content=reasoning_content, - reasoning_items=cast( - list[ChatCompletionReasoningItem] | None, - ([pending_reasoning_item] if pending_reasoning_item is not None else None), - ), + reasoning_items=_as_chat_reasoning_items(pending_reasoning_items), ) choices.append(Choices(message=msg, finish_reason="tool_calls", index=index)) reasoning_content = None - pending_reasoning_item = None + pending_reasoning_items = () return choices diff --git a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index 25a3220792f..48833fc1909 100644 --- a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -135,9 +135,7 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_imag function_call_output = item break - assert ( - function_call_output is not None - ), "function_call_output not found in response" + assert function_call_output is not None, "function_call_output not found in response" assert function_call_output["call_id"] == "call_abc123" # Check that the output is correctly transformed @@ -147,12 +145,8 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_imag image_item = output[0] # Should be transformed to Responses API format - assert ( - image_item["type"] == "input_image" - ), f"Expected type 'input_image', got '{image_item.get('type')}'" - assert ( - image_item["image_url"] == test_image_base64 - ), "image_url should be a flat string, not a nested object" + assert image_item["type"] == "input_image", f"Expected type 'input_image', got '{image_item.get('type')}'" + assert image_item["image_url"] == test_image_base64, "image_url should be a flat string, not a nested object" assert "detail" in image_item, "detail field should be present" print("✓ Tool result with image correctly transformed to Responses API format") @@ -214,9 +208,7 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text function_call_output = item break - assert ( - function_call_output is not None - ), "function_call_output not found in response" + assert function_call_output is not None, "function_call_output not found in response" assert function_call_output["call_id"] == "call_abc123" # Check that the output is correctly transformed to use input_text, not output_text @@ -226,16 +218,12 @@ def test_convert_chat_completion_messages_to_responses_api_tool_result_with_text text_item = output[0] # Should be transformed to use input_text for tool results in Responses API format - assert ( - text_item["type"] == "input_text" - ), f"Expected type 'input_text' for tool result, got '{text_item.get('type')}'" - assert ( - text_item["text"] == "15 degrees" - ), f"Expected text '15 degrees', got '{text_item.get('text')}'" - - print( - "✓ Tool result with text correctly transformed to use input_text for Responses API format" + assert text_item["type"] == "input_text", ( + f"Expected type 'input_text' for tool result, got '{text_item.get('type')}'" ) + assert text_item["text"] == "15 degrees", f"Expected text '15 degrees', got '{text_item.get('text')}'" + + print("✓ Tool result with text correctly transformed to use input_text for Responses API format") def test_openai_responses_chunk_parser_reasoning_summary(): @@ -244,9 +232,7 @@ def test_openai_responses_chunk_parser_reasoning_summary(): ) from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = { "delta": "**Compar", @@ -278,9 +264,7 @@ def test_chunk_parser_string_output_text_delta_produces_text(): ) from litellm.types.utils import ModelResponseStream - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = {"type": "response.output_text.delta", "delta": "literal text"} @@ -301,9 +285,7 @@ def test_chunk_parser_enum_output_text_delta_produces_text(): from litellm.types.llms.openai import ResponsesAPIStreamEvents from litellm.types.utils import ModelResponseStream - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = {"type": ResponsesAPIStreamEvents.OUTPUT_TEXT_DELTA, "delta": "enum text"} @@ -324,9 +306,7 @@ def test_chunk_parser_function_call_added_produces_tool_use(): from litellm.types.llms.openai import ResponsesAPIStreamEvents from litellm.types.utils import ModelResponseStream - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = { "type": ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED, @@ -411,9 +391,7 @@ Tomorrow will bring its petitions and promises, but for now the city breathes slow and wide, and I learn to carry this small calm home.""" - output_text = ResponseOutputText( - annotations=[], text=poem_text, type="output_text", logprobs=[] - ) + output_text = ResponseOutputText(annotations=[], text=poem_text, type="output_text", logprobs=[]) output_message = ResponseOutputMessage( id="msg_04c8021b8b3188a00068e9ae0b92f4819dac64d85b4abb67ec", content=[output_text], @@ -425,9 +403,7 @@ and I learn to carry this small calm home.""" # Create usage information usage = ResponseAPIUsage( input_tokens=16, - input_tokens_details=InputTokensDetails( - audio_tokens=None, cached_tokens=0, text_tokens=None - ), + input_tokens_details=InputTokensDetails(audio_tokens=None, cached_tokens=0, text_tokens=None), output_tokens=195, output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None), total_tokens=211, @@ -776,11 +752,7 @@ def test_recover_output_items_merges_text_only_items_at_distinct_indices(): ] ) - recovered = ( - LiteLLMResponsesTransformationHandler._recover_output_items_from_raw_sse( - raw_sse - ) - ) + recovered = LiteLLMResponsesTransformationHandler._recover_output_items_from_raw_sse(raw_sse) assert len(recovered) == 2 assert recovered[0]["id"] == "msg_item_0" @@ -918,9 +890,7 @@ def test_transform_request_system_only_message_maps_to_system_input_item(): { "type": "message", "role": "system", - "content": [ - {"type": "input_text", "text": "You are a helpful assistant."} - ], + "content": [{"type": "input_text", "text": "You are a helpful assistant."}], } ] # System content lives in input only; not duplicated into instructions. @@ -992,9 +962,7 @@ def test_transform_request_single_char_keys_not_matched(): assert result_correct.get("metadata") == {"user_id": "123"} assert result_correct.get("previous_response_id") == "resp_abc" - print( - "✓ Single-character keys are not incorrectly matched to metadata/previous_response_id" - ) + print("✓ Single-character keys are not incorrectly matched to metadata/previous_response_id") # ============================================================================= @@ -1014,9 +982,7 @@ def test_message_done_does_not_emit_is_finished(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = { "type": "response.output_item.done", @@ -1028,9 +994,9 @@ def test_message_done_does_not_emit_is_finished(): # After the fix, message completion should NOT set finish_reason # ModelResponseStream doesn't have is_finished - check finish_reason instead assert len(result.choices) > 0, "result should have choices" - assert ( - result.choices[0].finish_reason is None or result.choices[0].finish_reason == "" - ), "message completion should not emit finish_reason" + assert result.choices[0].finish_reason is None or result.choices[0].finish_reason == "", ( + "message completion should not emit finish_reason" + ) def test_response_completed_emits_is_finished(): @@ -1042,9 +1008,7 @@ def test_response_completed_emits_is_finished(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = {"type": "response.completed"} @@ -1052,9 +1016,7 @@ def test_response_completed_emits_is_finished(): # response.completed should emit finish_reason='stop' assert len(result.choices) > 0, "result should have choices" - assert ( - result.choices[0].finish_reason == "stop" - ), "response.completed should emit finish_reason='stop'" + assert result.choices[0].finish_reason == "stop", "response.completed should emit finish_reason='stop'" def test_response_completed_with_function_calls_emits_tool_calls_finish_reason(): @@ -1073,9 +1035,7 @@ def test_response_completed_with_function_calls_emits_tool_calls_finish_reason() OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) # Simulate a response.completed event with function_call in output # This matches what Azure/OpenAI sends for gpt-5.1-codex-mini and similar models @@ -1101,9 +1061,9 @@ def test_response_completed_with_function_calls_emits_tool_calls_finish_reason() # response.completed with function_call should emit finish_reason='tool_calls' assert len(result.choices) > 0, "result should have choices" - assert ( - result.choices[0].finish_reason == "tool_calls" - ), "response.completed with function_call output should emit finish_reason='tool_calls'" + assert result.choices[0].finish_reason == "tool_calls", ( + "response.completed with function_call output should emit finish_reason='tool_calls'" + ) def test_response_completed_with_message_only_emits_stop_finish_reason(): @@ -1114,9 +1074,7 @@ def test_response_completed_with_message_only_emits_stop_finish_reason(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) # Simulate a response.completed event with only message output chunk = { @@ -1140,9 +1098,9 @@ def test_response_completed_with_message_only_emits_stop_finish_reason(): # response.completed with only message should emit finish_reason='stop' assert len(result.choices) > 0, "result should have choices" - assert ( - result.choices[0].finish_reason == "stop" - ), "response.completed with only message output should emit finish_reason='stop'" + assert result.choices[0].finish_reason == "stop", ( + "response.completed with only message output should emit finish_reason='stop'" + ) def test_response_completed_preserves_usage_with_cached_tokens(): @@ -1158,9 +1116,7 @@ def test_response_completed_preserves_usage_with_cached_tokens(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = { "type": "response.completed", @@ -1189,18 +1145,12 @@ def test_response_completed_preserves_usage_with_cached_tokens(): result = iterator.chunk_parser(chunk) assert result.usage is not None, "usage should be set on response.completed chunk" - assert ( - result.usage.prompt_tokens == 1226 - ), "prompt_tokens should map from input_tokens" - assert ( - result.usage.completion_tokens == 5 - ), "completion_tokens should map from output_tokens" - assert ( - result.usage.prompt_tokens_details is not None - ), "prompt_tokens_details should be set" - assert ( - result.usage.prompt_tokens_details.cached_tokens == 1024 - ), "cached_tokens should be preserved from input_tokens_details" + assert result.usage.prompt_tokens == 1226, "prompt_tokens should map from input_tokens" + assert result.usage.completion_tokens == 5, "completion_tokens should map from output_tokens" + assert result.usage.prompt_tokens_details is not None, "prompt_tokens_details should be set" + assert result.usage.prompt_tokens_details.cached_tokens == 1024, ( + "cached_tokens should be preserved from input_tokens_details" + ) def test_function_call_done_emits_is_finished(): @@ -1214,9 +1164,7 @@ def test_function_call_done_emits_is_finished(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = { "type": "response.output_item.done", @@ -1236,9 +1184,9 @@ def test_function_call_done_emits_is_finished(): "output_item.done for function_call must not emit finish_reason; " "response.completed is responsible for the terminal finish_reason" ) - assert not result.choices[ - 0 - ].delta.tool_calls, "output_item.done for function_call must not include a duplicate tool_calls delta" + assert not result.choices[0].delta.tool_calls, ( + "output_item.done for function_call must not include a duplicate tool_calls delta" + ) def test_text_plus_tool_calls_sequence(): @@ -1253,9 +1201,7 @@ def test_text_plus_tool_calls_sequence(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) # Simulate the sequence from OpenAI Responses API chunks = [ @@ -1294,28 +1240,23 @@ def test_text_plus_tool_calls_sequence(): # Check message done (index 2) does NOT have finish_reason set message_done_result = results[2] assert len(message_done_result.choices) > 0, "message done should have choices" - assert ( - message_done_result.choices[0].finish_reason is None - or message_done_result.choices[0].finish_reason == "" - ), "message done should not have finish_reason" + assert message_done_result.choices[0].finish_reason is None or message_done_result.choices[0].finish_reason == "", ( + "message done should not have finish_reason" + ) # Check function_call done (index 5) does NOT have finish_reason set # (response.completed is responsible for the terminal finish_reason) function_done_result = results[5] - assert ( - len(function_done_result.choices) > 0 - ), "function_call done should have choices" - assert ( - function_done_result.choices[0].finish_reason is None - ), "output_item.done for function_call must not emit finish_reason" + assert len(function_done_result.choices) > 0, "function_call done should have choices" + assert function_done_result.choices[0].finish_reason is None, ( + "output_item.done for function_call must not emit finish_reason" + ) # Check response.completed (index 6) has finish_reason='stop' # (the mock chunk has no nested 'response' data, so has_function_calls is False → 'stop') completed_result = results[6] assert len(completed_result.choices) > 0, "response.completed should have choices" - assert ( - completed_result.choices[0].finish_reason == "stop" - ), "response.completed should have finish_reason='stop'" + assert completed_result.choices[0].finish_reason == "stop", "response.completed should have finish_reason='stop'" # ============================================================================= @@ -1332,7 +1273,11 @@ def test_developer_message_content_uses_input_text(): assert instructions is None assert input_items == [ - {"type": "message", "role": "developer", "content": [{"type": "input_text", "text": "Always answer in French."}]} + { + "type": "message", + "role": "developer", + "content": [{"type": "input_text", "text": "Always answer in French."}], + } ] @@ -1394,9 +1339,7 @@ def test_tool_message_output_uses_input_text_not_output_text(): output = function_call_output["output"] assert isinstance(output, list), f"output should be a list, got {type(output)}" assert len(output) == 1 - assert ( - output[0]["type"] == "input_text" - ), f"Expected input_text, got {output[0].get('type')}" + assert output[0]["type"] == "input_text", f"Expected input_text, got {output[0].get('type')}" assert output[0]["text"] == '{"temperature": 15, "condition": "sunny"}' print("✓ Tool message output correctly uses input_text type") @@ -1581,13 +1524,9 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): assert result is not None, f"Result should not be None for effort={effort}" assert result["effort"] == effort, f"Effort should be {effort}" - assert ( - "summary" not in result - ), f"Summary should NOT be present by default for effort={effort}" + assert "summary" not in result, f"Summary should NOT be present by default for effort={effort}" - print( - f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}' (no summary by default)" - ) + print(f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}' (no summary by default)") # Test 2: With flag enabled - summary IS added litellm.reasoning_auto_summary = True @@ -1597,9 +1536,9 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): assert result is not None, f"Result should not be None for effort={effort}" assert result["effort"] == effort, f"Effort should be {effort}" - assert ( - result["summary"] == "detailed" - ), f"Summary should be 'detailed' when flag is enabled for effort={effort}" + assert result["summary"] == "detailed", ( + f"Summary should be 'detailed' when flag is enabled for effort={effort}" + ) print( f"✓ reasoning_effort='{effort}' correctly maps to effort='{effort}', summary='detailed' (flag enabled)" @@ -1610,9 +1549,7 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): monkeypatch.setenv("LITELLM_REASONING_AUTO_SUMMARY", "true") result = handler._map_reasoning_effort("high") - assert ( - result["summary"] == "detailed" - ), "Summary should be 'detailed' when env var is enabled" + assert result["summary"] == "detailed", "Summary should be 'detailed' when env var is enabled" print("✓ LITELLM_REASONING_AUTO_SUMMARY env var works correctly") # Test 4: Dict input is passed through as-is (no modification) @@ -1626,9 +1563,7 @@ def test_map_reasoning_effort_adds_summary_detailed(monkeypatch): assert result_dict["summary"] == "custom_summary" print("✓ Dict input is passed through without modification") - print( - "✓ All reasoning_effort behaviors work correctly with flag/env var control" - ) + print("✓ All reasoning_effort behaviors work correctly with flag/env var control") finally: # Restore original values @@ -1704,9 +1639,7 @@ def test_transform_response_preserves_annotations(): # Create usage information usage = ResponseAPIUsage( input_tokens=10, - input_tokens_details=InputTokensDetails( - audio_tokens=None, cached_tokens=0, text_tokens=None - ), + input_tokens_details=InputTokensDetails(audio_tokens=None, cached_tokens=0, text_tokens=None), output_tokens=20, output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None), total_tokens=30, @@ -1793,13 +1726,9 @@ def test_transform_response_preserves_annotations(): assert choice.message.content == "Here is some information with citations." # Check that annotations are preserved - assert hasattr( - choice.message, "annotations" - ), "Message should have annotations attribute" + assert hasattr(choice.message, "annotations"), "Message should have annotations attribute" assert choice.message.annotations is not None, "Annotations should not be None" - assert ( - len(choice.message.annotations) == 2 - ), f"Expected 2 annotations, got {len(choice.message.annotations)}" + assert len(choice.message.annotations) == 2, f"Expected 2 annotations, got {len(choice.message.annotations)}" # Verify annotation content annotation1 = choice.message.annotations[0] @@ -1821,9 +1750,7 @@ def test_transform_response_preserves_annotations(): assert result.usage.completion_tokens == 20 assert result.usage.total_tokens == 30 - print( - "✓ Annotations from Responses API are correctly preserved in Chat Completions format" - ) + print("✓ Annotations from Responses API are correctly preserved in Chat Completions format") def test_apply_patch_tool_call_converted_to_chat_completion_tool_call(): @@ -1988,9 +1915,7 @@ def test_multi_tool_call_stream_no_premature_finish(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunks = [ # 0: response created @@ -2066,12 +1991,10 @@ def test_multi_tool_call_stream_no_premature_finish(): r = results[done_idx] assert r is not None, f"{label}: chunk_parser must return a result" assert len(r.choices) > 0, f"{label}: result must have choices" - assert ( - r.choices[0].finish_reason is None - ), f"{label}: output_item.done must not emit finish_reason (stream would terminate prematurely)" - assert not r.choices[ - 0 - ].delta.tool_calls, ( + assert r.choices[0].finish_reason is None, ( + f"{label}: output_item.done must not emit finish_reason (stream would terminate prematurely)" + ) + assert not r.choices[0].delta.tool_calls, ( f"{label}: output_item.done must not include a duplicate tool_calls delta" ) @@ -2083,12 +2006,8 @@ def test_multi_tool_call_stream_no_premature_finish(): r = results[added_idx] if r is not None and r.choices and r.choices[0].delta.tool_calls: tc = r.choices[0].delta.tool_calls[0] - assert ( - tc.function.name == expected_name - ), f"output_item.added for {expected_name}: tool_call name mismatch" - assert ( - tc.id == expected_call_id - ), f"output_item.added for {expected_name}: call_id mismatch" + assert tc.function.name == expected_name, f"output_item.added for {expected_name}: tool_call name mismatch" + assert tc.id == expected_call_id, f"output_item.added for {expected_name}: call_id mismatch" # 3. argument delta events (indices 2 and 5) should carry arguments for delta_idx, expected_args, label in [ @@ -2098,17 +2017,15 @@ def test_multi_tool_call_stream_no_premature_finish(): r = results[delta_idx] if r is not None and r.choices and r.choices[0].delta.tool_calls: tc = r.choices[0].delta.tool_calls[0] - assert ( - tc.function.arguments == expected_args - ), f"{label}: argument delta mismatch" + assert tc.function.arguments == expected_args, f"{label}: argument delta mismatch" # 4. Only response.completed (index 7) emits the terminal finish_reason completed_result = results[7] assert completed_result is not None, "response.completed must return a result" assert len(completed_result.choices) > 0, "response.completed must have choices" - assert ( - completed_result.choices[0].finish_reason == "tool_calls" - ), "response.completed with function_call outputs must emit finish_reason='tool_calls'" + assert completed_result.choices[0].finish_reason == "tool_calls", ( + "response.completed with function_call outputs must emit finish_reason='tool_calls'" + ) # 5. No chunk before the last one should have finish_reason set for idx, r in enumerate(results[:-1]): @@ -2118,9 +2035,7 @@ def test_multi_tool_call_stream_no_premature_finish(): f"— only response.completed should terminate the stream" ) - print( - "✓ Multi-tool-call stream completes without premature finish_reason termination" - ) + print("✓ Multi-tool-call stream completes without premature finish_reason termination") # ============================================================================= @@ -2201,16 +2116,13 @@ def test_streaming_parallel_tool_calls_have_distinct_indices(): ] for chunk in chunks: - result = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream( - chunk - ) + result = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(chunk) expected_index = chunk["output_index"] for choice in result.choices: if choice.delta.tool_calls: for tc in choice.delta.tool_calls: assert tc.index == expected_index, ( - f"Event {chunk['type']}: expected tool_call.index={expected_index}, " - f"got {tc.index}" + f"Event {chunk['type']}: expected tool_call.index={expected_index}, got {tc.index}" ) @@ -2338,9 +2250,7 @@ def test_parallel_tool_calls_comprehensive_streaming_integration(): }, ] - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) results = [iterator.chunk_parser(chunk) for chunk in chunks] # 1. output_item.done events (indices 4 and 8) must NOT emit finish_reason @@ -2352,9 +2262,7 @@ def test_parallel_tool_calls_comprehensive_streaming_integration(): f"{label}: output_item.done must not emit finish_reason " f"(would prematurely terminate stream before subsequent tool calls arrive)" ) - assert not r.choices[ - 0 - ].delta.tool_calls, ( + assert not r.choices[0].delta.tool_calls, ( f"{label}: output_item.done must not emit a duplicate tool_calls delta" ) @@ -2388,19 +2296,15 @@ def test_parallel_tool_calls_comprehensive_streaming_integration(): for tc in tool_calls: if tc.function and tc.function.arguments: idx = tc.index - assembled_args[idx] = ( - assembled_args.get(idx, "") + tc.function.arguments - ) + assembled_args[idx] = assembled_args.get(idx, "") + tc.function.arguments # delta 1 = '{"path":' + delta 2 = '"/etc/foo"}' → '{"path":"/etc/foo"}' assert assembled_args.get(0) == '{"path":"/etc/foo"}', ( - f"Assembled args for index 0 (read_file): " - f"expected '{{\"path\":\"/etc/foo\"}}', got '{assembled_args.get(0)}'" + f"Assembled args for index 0 (read_file): expected '{{\"path\":\"/etc/foo\"}}', got '{assembled_args.get(0)}'" ) # delta 1 = '{"path":' + delta 2 = '"/tmp"}' → '{"path":"/tmp"}' assert assembled_args.get(1) == '{"path":"/tmp"}', ( - f"Assembled args for index 1 (list_dir): " - f"expected '{{\"path\":\"/tmp\"}}', got '{assembled_args.get(1)}'" + f"Assembled args for index 1 (list_dir): expected '{{\"path\":\"/tmp\"}}', got '{assembled_args.get(1)}'" ) # 4. Stream terminates with exactly one finish event, at the final response.completed chunk @@ -2409,16 +2313,13 @@ def test_parallel_tool_calls_comprehensive_streaming_integration(): for i, r in enumerate(results) if r is not None and r.choices and r.choices[0].finish_reason ] - assert ( - len(finish_events) == 1 - ), f"Expected exactly 1 finish event, got {len(finish_events)}: {finish_events}" + assert len(finish_events) == 1, f"Expected exactly 1 finish event, got {len(finish_events)}: {finish_events}" assert finish_events[0][0] == len(chunks) - 1, ( - f"Finish event must be at the last chunk (index {len(chunks) - 1}), " - f"but was at index {finish_events[0][0]}" + f"Finish event must be at the last chunk (index {len(chunks) - 1}), but was at index {finish_events[0][0]}" + ) + assert finish_events[0][1] == "tool_calls", ( + f"Terminal finish_reason must be 'tool_calls', got '{finish_events[0][1]}'" ) - assert ( - finish_events[0][1] == "tool_calls" - ), f"Terminal finish_reason must be 'tool_calls', got '{finish_events[0][1]}'" # 5. Parallel tool calls have distinct indices matching output_index (0 and 1) # Collect indices from output_item.added chunks only (they carry the call id) @@ -2434,9 +2335,7 @@ def test_parallel_tool_calls_comprehensive_streaming_integration(): 1, }, f"Parallel tool calls must have distinct indices {{0, 1}}, got: {set(added_tool_call_indices)}" - print( - "✓ Parallel tool calls with split argument deltas stream correctly end-to-end" - ) + print("✓ Parallel tool calls with split argument deltas stream correctly end-to-end") def test_map_optional_params_preserves_reasoning_summary(): @@ -2460,9 +2359,7 @@ def test_map_optional_params_preserves_reasoning_summary(): } responses_api_request = ResponsesAPIOptionalRequestParams() - handler._map_optional_params_to_responses_api_request( - optional_params, responses_api_request - ) + handler._map_optional_params_to_responses_api_request(optional_params, responses_api_request) # Verify reasoning_effort dict with summary was fully preserved assert "reasoning" in responses_api_request @@ -2735,9 +2632,7 @@ def test_reasoning_items_non_streaming_round_trip(): ) usage = ResponseAPIUsage( input_tokens=10, - input_tokens_details=InputTokensDetails( - audio_tokens=None, cached_tokens=0, text_tokens=None - ), + input_tokens_details=InputTokensDetails(audio_tokens=None, cached_tokens=0, text_tokens=None), output_tokens=20, output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None), total_tokens=30, @@ -2801,9 +2696,7 @@ def test_reasoning_items_non_streaming_round_trip(): assert len(result.choices) == 1 msg = result.choices[0].message - assert ( - msg.reasoning_content == summary_text - ), "reasoning_content should equal summary text" + assert msg.reasoning_content == summary_text, "reasoning_content should equal summary text" assert msg.reasoning_items is not None, "reasoning_items should be set" assert len(msg.reasoning_items) == 1 @@ -2828,13 +2721,9 @@ def test_reasoning_items_non_streaming_round_trip(): # The reasoning input item must appear before the assistant message item types = [item.get("type") for item in input_items] - assert ( - "reasoning" in types - ), "reasoning input item must be emitted for the assistant turn" + assert "reasoning" in types, "reasoning input item must be emitted for the assistant turn" - reasoning_input = next( - item for item in input_items if item.get("type") == "reasoning" - ) + reasoning_input = next(item for item in input_items if item.get("type") == "reasoning") assert reasoning_input["id"] == "rs_test001" assert reasoning_input["encrypted_content"] == encrypted assert reasoning_input["summary"][0]["text"] == summary_text @@ -2842,13 +2731,9 @@ def test_reasoning_items_non_streaming_round_trip(): # reasoning item must come before the assistant message item reasoning_idx = types.index("reasoning") assistant_msg_idx = next( - i - for i, item in enumerate(input_items) - if item.get("type") == "message" and item.get("role") == "assistant" + i for i, item in enumerate(input_items) if item.get("type") == "message" and item.get("role") == "assistant" ) - assert ( - reasoning_idx < assistant_msg_idx - ), "reasoning input item must precede the assistant message item" + assert reasoning_idx < assistant_msg_idx, "reasoning input item must precede the assistant message item" def test_reasoning_items_streaming_emitted_on_response_completed(): @@ -2861,9 +2746,7 @@ def test_reasoning_items_streaming_emitted_on_response_completed(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) encrypted = "gAAAAABpw5xyz987FAKE==" summary_text = "**Reasoning summary**\n\nModel thought about this carefully." @@ -2907,16 +2790,14 @@ def test_reasoning_items_streaming_emitted_on_response_completed(): assert result.choices[0].finish_reason == "stop" # reasoning_items must be on the delta - assert ( - getattr(delta, "reasoning_items", None) is not None - ), "reasoning_items must be present on the response.completed delta" + assert getattr(delta, "reasoning_items", None) is not None, ( + "reasoning_items must be present on the response.completed delta" + ) assert len(delta.reasoning_items) == 1 ri = delta.reasoning_items[0] assert ri["type"] == "reasoning" assert ri["id"] == "rs_stream001" - assert ( - ri["encrypted_content"] == encrypted - ), "encrypted_content must be preserved in streaming" + assert ri["encrypted_content"] == encrypted, "encrypted_content must be preserved in streaming" assert ri["summary"][0]["text"] == summary_text @@ -2943,9 +2824,7 @@ def test_streaming_function_call_tool_id_for_degenerate_call_id(): "arguments": "", }, } - out = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream( - chunk - ) + out = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(chunk) tool_calls = out.model_dump()["choices"][0]["delta"]["tool_calls"] assert tool_calls, "expected a tool_call chunk in the streaming delta" return tool_calls[0]["id"] @@ -2964,9 +2843,7 @@ def test_streaming_chunks_share_one_chat_completion_id(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) events = [ {"type": "response.created", "response": {"id": "resp_abc", "output": []}}, {"type": "response.output_text.delta", "delta": "Hel"}, @@ -2982,12 +2859,10 @@ def test_streaming_chunks_share_one_chat_completion_id(): assert len(set(ids)) == 1, f"streamed chunks carried different ids: {ids}" assert ids[0], "streamed chunks carried an empty id" - other_stream = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True + other_stream = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) + assert other_stream.chunk_parser(events[1]).id != ids[0], ( + "a separate stream must get its own id, not a process-wide one" ) - assert ( - other_stream.chunk_parser(events[1]).id != ids[0] - ), "a separate stream must get its own id, not a process-wide one" @pytest.mark.asyncio @@ -2998,9 +2873,7 @@ def test_streaming_chunks_share_one_chat_completion_id(): ({"include_usage": True}, None), ], ) -async def test_acompletion_bridge_normalizes_stream_options_on_the_wire( - stream_options, expected_wire_stream_options -): +async def test_acompletion_bridge_normalizes_stream_options_on_the_wire(stream_options, expected_wire_stream_options): """include_usage must be stripped from the /v1/responses body; include_obfuscation must survive as a dict.""" from unittest.mock import AsyncMock @@ -3076,9 +2949,7 @@ def test_chunk_parser_custom_tool_call_stream_sequence(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) added = iterator.chunk_parser( { @@ -3156,9 +3027,7 @@ def test_chunk_parser_remaps_tool_call_indices_sequentially(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) first = iterator.chunk_parser( { @@ -3735,9 +3604,7 @@ def _make_incomplete_responses_api_response( created_at=1760144904, error=None, incomplete_details=( - {"reason": incomplete_reason} - if incomplete_reason is not None or empty_incomplete_details - else None + {"reason": incomplete_reason} if incomplete_reason is not None or empty_incomplete_details else None ), instructions=None, metadata={}, @@ -3757,13 +3624,9 @@ def _make_incomplete_responses_api_response( truncation="disabled", usage=ResponseAPIUsage( input_tokens=37, - input_tokens_details=InputTokensDetails( - audio_tokens=None, cached_tokens=0, text_tokens=None - ), + input_tokens_details=InputTokensDetails(audio_tokens=None, cached_tokens=0, text_tokens=None), output_tokens=16, - output_tokens_details=OutputTokensDetails( - reasoning_tokens=16, text_tokens=None - ), + output_tokens_details=OutputTokensDetails(reasoning_tokens=16, text_tokens=None), total_tokens=53, cost=None, ), @@ -3813,9 +3676,7 @@ def _call_transform_response( def test_transform_response_incomplete_reasoning_only_returns_empty_length_choice(): handler = LiteLLMResponsesTransformationHandler() - raw_response = _make_incomplete_responses_api_response( - "max_output_tokens", [_make_reasoning_only_output_item()] - ) + raw_response = _make_incomplete_responses_api_response("max_output_tokens", [_make_reasoning_only_output_item()]) result = _call_transform_response(handler, raw_response) @@ -3834,9 +3695,7 @@ def test_transform_response_incomplete_reasoning_only_returns_empty_length_choic def test_transform_response_incomplete_content_filter_maps_finish_reason(): handler = LiteLLMResponsesTransformationHandler() - raw_response = _make_incomplete_responses_api_response( - "content_filter", [_make_reasoning_only_output_item()] - ) + raw_response = _make_incomplete_responses_api_response("content_filter", [_make_reasoning_only_output_item()]) result = _call_transform_response(handler, raw_response) @@ -3859,11 +3718,7 @@ def test_transform_response_completed_with_reasonless_incomplete_details_keeps_s handler = LiteLLMResponsesTransformationHandler() output_message = ResponseOutputMessage( id="msg_complete", - content=[ - ResponseOutputText( - annotations=[], text="full answer", type="output_text", logprobs=[] - ) - ], + content=[ResponseOutputText(annotations=[], text="full answer", type="output_text", logprobs=[])], role="assistant", status="completed", type="message", @@ -3885,11 +3740,7 @@ def test_transform_response_incomplete_partial_text_overrides_finish_reason_to_l handler = LiteLLMResponsesTransformationHandler() output_message = ResponseOutputMessage( id="msg_partial", - content=[ - ResponseOutputText( - annotations=[], text="partial answer", type="output_text", logprobs=[] - ) - ], + content=[ResponseOutputText(annotations=[], text="partial answer", type="output_text", logprobs=[])], role="assistant", status="incomplete", type="message", @@ -3911,9 +3762,7 @@ def test_response_incomplete_stream_event_emits_length_and_usage(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = { "type": "response.incomplete", @@ -3954,9 +3803,7 @@ def test_response_incomplete_stream_event_content_filter_maps_finish_reason(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = { "type": "response.incomplete", @@ -3978,9 +3825,7 @@ def test_response_incomplete_stream_event_without_details_defaults_to_length(): OpenAiResponsesToChatCompletionStreamIterator, ) - iterator = OpenAiResponsesToChatCompletionStreamIterator( - streaming_response=None, sync_stream=True - ) + iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True) chunk = { "type": "response.incomplete", @@ -4060,9 +3905,7 @@ def test_thinking_only_assistant_turn_still_sends_its_reasoning(): { "role": "assistant", "content": None, - "thinking_blocks": [ - {"type": "thinking", "thinking": "August in Denver is dry.", "signature": "sig1"} - ], + "thinking_blocks": [{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "sig1"}], }, {"role": "user", "content": "Why?"}, ] @@ -4088,9 +3931,7 @@ def test_stored_reasoning_items_win_over_thinking_blocks(): "summary": [{"type": "summary_text", "text": "August in Denver is dry."}], } ], - "thinking_blocks": [ - {"type": "thinking", "thinking": "August in Denver is dry.", "signature": "rs_real"} - ], + "thinking_blocks": [{"type": "thinking", "thinking": "August in Denver is dry.", "signature": "rs_real"}], }, ] @@ -4560,3 +4401,64 @@ def test_every_bridged_chunk_after_response_created_carries_the_served_service_t relayed = [iterator.chunk_parser(event).model_dump().get("service_tier") for event in events] assert relayed == ["default"] * len(events), relayed + + +def test_convert_response_output_keeps_every_reasoning_item_before_a_message(): + """Regression for issue #43620 (stream=False): the Responses API can emit several + reasoning items before the assistant message, and the bridge used to overwrite + its pending reasoning item per item, so only the last one reached + message.reasoning_items. Every item must survive the flush.""" + from openai.types.responses import ResponseOutputMessage, ResponseOutputText, ResponseReasoningItem + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + rs_1 = ResponseReasoningItem(type="reasoning", id="rs_1", summary=[], encrypted_content="enc1") + rs_2 = ResponseReasoningItem(type="reasoning", id="rs_2", summary=[], encrypted_content="enc2") + message = ResponseOutputMessage( + type="message", + id="msg_1", + role="assistant", + status="completed", + content=[ResponseOutputText(type="output_text", text="hello", annotations=[])], + ) + + choices = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices([rs_1, rs_2, message]) + + assert len(choices) == 1 + reasoning_items = choices[0].message.reasoning_items + assert reasoning_items is not None + assert [item["id"] for item in reasoning_items] == ["rs_1", "rs_2"] + + +def test_convert_response_output_keeps_every_reasoning_item_before_a_tool_call(): + """Regression for issue #43620 (stream=False): [reasoning rs_1, reasoning rs_2, + function_call] must merge both reasoning items into the trailing tool_calls + choice instead of keeping only rs_2.""" + from openai.types.responses import ResponseFunctionToolCall, ResponseReasoningItem + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + rs_1 = ResponseReasoningItem(type="reasoning", id="rs_1", summary=[], encrypted_content="enc1") + rs_2 = ResponseReasoningItem(type="reasoning", id="rs_2", summary=[], encrypted_content="enc2") + function_call = ResponseFunctionToolCall( + type="function_call", + id="fc_1", + call_id="call_1", + name="get_weather", + arguments="{}", + status="completed", + ) + + choices = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices([rs_1, rs_2, function_call]) + + assert len(choices) == 1 + choice = choices[0] + assert choice.finish_reason == "tool_calls" + reasoning_items = choice.message.reasoning_items + assert reasoning_items is not None + assert [item["id"] for item in reasoning_items] == ["rs_1", "rs_2"] + assert choice.message.tool_calls[0].id == "fc_1"