From f49f540b6d3e5272682d04b4e6edb34502978400 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Fri, 10 Oct 2025 18:29:37 -0700 Subject: [PATCH] feat(litellm_responses_transformation/transformation.py): parse thinking content in response<-> chat completion bridge allows gpt-5 to return thinking content when called via responses api --- .../transformation.py | 25 ++- ...responses_transformation_transformation.py | 175 ++++++++++++++++++ 2 files changed, 196 insertions(+), 4 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 5f732fc5219..612a941c231 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -221,7 +221,6 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): json_mode: Optional[bool] = None, ) -> "ModelResponse": """Transform Responses API response to chat completion response""" - from openai.types.responses import ( ResponseFunctionToolCall, ResponseOutputMessage, @@ -240,19 +239,35 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): choices: List[Choices] = [] index = 0 + + reasoning_content: Optional[str] = None + for item in raw_response.output: + if isinstance(item, ResponseReasoningItem): - pass # ignore for now. + + for content in item.summary: + response_text = getattr(content, "text", "") + reasoning_content = response_text if response_text else "" + elif isinstance(item, ResponseOutputMessage): for content in item.content: response_text = getattr(content, "text", "") msg = Message( - role=item.role, content=response_text if response_text else "" + role=item.role, + content=response_text if response_text else "", + reasoning_content=reasoning_content, ) choices.append( - Choices(message=msg, finish_reason="stop", index=index) + Choices( + message=msg, + finish_reason="stop", + index=index, + ) ) + + reasoning_content = None # flush reasoning content index += 1 elif isinstance(item, ResponseFunctionToolCall): msg = Message( @@ -267,11 +282,13 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): "type": "function", } ], + reasoning_content=reasoning_content, ) choices.append( Choices(message=msg, finish_reason="tool_calls", index=index) ) + reasoning_content = None # flush reasoning content index += 1 else: pass # don't fail request if item in list is not supported diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index ef76cfa02d1..e3fecb8852a 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -84,3 +84,178 @@ def test_openai_responses_chunk_parser_reasoning_summary(): assert delta.reasoning_content == "**Compar" assert delta.tool_calls is None assert delta.function_call is None + + +def test_transform_response_with_reasoning_and_output(): + """Test transform_response handles ResponsesAPIResponse with reasoning items and output messages.""" + from unittest.mock import Mock + + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + from openai.types.responses.response_reasoning_item import ( + ResponseReasoningItem, + Summary, + ) + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + from litellm.types.llms.openai import ( + InputTokensDetails, + OutputTokensDetails, + ResponseAPIUsage, + ResponsesAPIResponse, + ) + from litellm.types.utils import ModelResponse, Usage + + handler = LiteLLMResponsesTransformationHandler() + + # Create the reasoning item with summary + reasoning_summary = Summary( + text="**Creating a poem**\n\nThe user wants a poem without constraints, which is great! I need to focus on keeping it original and evocative.", + type="summary_text", + ) + reasoning_item = ResponseReasoningItem( + id="rs_04c8021b8b3188a00068e9ae08c2d8819d82268b129351a979", + summary=[reasoning_summary], + type="reasoning", + content=None, + encrypted_content=None, + status=None, + ) + + # Create the output message with the poem + poem_text = """I found a pocket of evening +hidden behind the gutters of the day — +a small, folded sky of blue +that hummed like a hush. + +The streetlight rehearsed its first apology, +slowly pulling down the curtain +on the city's impatient laughter. +Windows blinked awake like tired eyes, +and the air remembered rain it once promised. + +You walked by with a map of quiet in your hands, +tracing routes that led away from all the clocks. +For a moment the coffee shop's bell +tied our minutes together — bright and accidental — +and the world refined itself to the size of that bell's sound. + +We did not name the solitude; we sipped it. +You left a warmth on the bench like a small sun, +and night stitched the rest into blue and shadow. +Tomorrow will bring its petitions and promises, +but for now the city breathes slow and wide, +and I learn to carry this small calm home.""" + + output_text = ResponseOutputText( + annotations=[], text=poem_text, type="output_text", logprobs=[] + ) + output_message = ResponseOutputMessage( + id="msg_04c8021b8b3188a00068e9ae0b92f4819dac64d85b4abb67ec", + content=[output_text], + role="assistant", + status="completed", + type="message", + ) + + # Create usage information + usage = ResponseAPIUsage( + input_tokens=16, + input_tokens_details=InputTokensDetails( + audio_tokens=None, cached_tokens=0, text_tokens=None + ), + output_tokens=195, + output_tokens_details=OutputTokensDetails(reasoning_tokens=0, text_tokens=None), + total_tokens=211, + cost=None, + ) + + # Create the full ResponsesAPIResponse + raw_response = ResponsesAPIResponse( + id="resp_bGl0ZWxsbTpjdXN0b21fbGxtX3Byb3ZpZGVyOm9wZW5haTttb2RlbF9pZDpOb25lO3Jlc3BvbnNlX2lkOnJlc3BfMDRjODAyMWI4YjMxODhhMDAwNjhlOWFlMDgyYmZjODE5ZDhmNDk0OTI5MWMzMzM4YTc=", + created_at=1760144904, + error=None, + incomplete_details=None, + instructions=None, + metadata={}, + model="gpt-5-mini-2025-08-07", + object="response", + output=[reasoning_item, output_message], + parallel_tool_calls=True, + temperature=1.0, + tool_choice="auto", + tools=[], + top_p=1.0, + max_output_tokens=None, + previous_response_id=None, + reasoning={"effort": "low", "summary": "detailed"}, + status="completed", + text={"format": {"type": "text"}, "verbosity": "medium"}, + truncation="disabled", + usage=usage, + user=None, + store=True, + background=False, + billing={"payer": "developer"}, + max_tool_calls=None, + prompt_cache_key=None, + safety_identifier=None, + service_tier="default", + top_logprobs=0, + ) + + # Create empty model_response + model_response = ModelResponse( + id="chatcmpl-42e863c4-7a31-4229-84f3-4c3a6eeb7610", + created=1760144904, + model=None, + object="chat.completion", + system_fingerprint=None, + choices=[], + usage=Usage(completion_tokens=0, prompt_tokens=0, total_tokens=0), + ) + + # Create mock objects for required parameters + logging_obj = Mock() + messages = [{"role": "user", "content": "Think of a poem, and then write it."}] + request_data = {"model": "gpt-5-mini"} + optional_params = {"reasoning_effort": "low", "extra_body": {}} + litellm_params = {"acompletion": False, "api_key": None} + encoding = Mock() + + # Call transform_response + result = handler.transform_response( + model="gpt-5-mini", + raw_response=raw_response, + model_response=model_response, + logging_obj=logging_obj, + request_data=request_data, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=encoding, + api_key=None, + json_mode=None, + ) + + # Assertions + assert result.model == "gpt-5-mini" + assert len(result.choices) == 1 + + # Check the choice + choice = result.choices[0] + assert choice.finish_reason == "stop" + assert choice.index == 0 + assert choice.message.role == "assistant" + assert choice.message.content == poem_text + + # Check usage + assert result.usage.prompt_tokens == 16 + assert result.usage.completion_tokens == 195 + assert result.usage.total_tokens == 211 + + # Check reasoning content + assert choice.message.reasoning_content == reasoning_summary.text + + print("✓ transform_response correctly handled reasoning items and output messages")