fix: make gemini and openai responses return reasoning by default

This aligns the proxy experience with other models that think
automatically (e.g. Deepseek R1 and grok3). It does so by setting
the necessary request input to return thinking, but not specifying
a budget or effort (thus defaulting to the internal automatic level).
This commit is contained in:
Adam Holmberg 2025-07-22 10:41:29 -05:00
parent 774af8085e
commit 2ce03d9735
4 changed files with 222 additions and 67 deletions

View file

@ -157,8 +157,8 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
responses_api_request["metadata"] = value
elif key in ("previous_response_id"):
responses_api_request["previous_response_id"] = value
elif key == "reasoning_effort":
responses_api_request["reasoning"] = self._map_reasoning_effort(value)
responses_api_request["reasoning"] = self._map_reasoning_effort(optional_params.get("reasoning_effort"))
# Get stream parameter from litellm_params if not in optional_params
stream = optional_params.get("stream") or litellm_params.get("stream", False)
@ -452,7 +452,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
responses_tools.append(tool)
return cast(List["ALL_RESPONSES_API_TOOL_PARAMS"], responses_tools)
def _map_reasoning_effort(self, reasoning_effort: str) -> Optional[Reasoning]:
def _map_reasoning_effort(self, reasoning_effort: Optional[str]) -> Reasoning:
if reasoning_effort == "high":
return Reasoning(effort="high", summary="detailed")
elif reasoning_effort == "medium":
@ -460,7 +460,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
return Reasoning(effort="medium", summary="auto")
elif reasoning_effort == "low":
return Reasoning(effort="low", summary="auto")
return None
return Reasoning(summary="auto")
def _map_responses_status_to_finish_reason(self, status: Optional[str]) -> str:
"""Map responses API status to chat completion finish_reason"""

View file

@ -417,8 +417,10 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
@staticmethod
def _map_reasoning_effort_to_thinking_budget(
reasoning_effort: str,
reasoning_effort: Optional[str],
) -> GeminiThinkingConfig:
if not reasoning_effort:
return { "includeThoughts": True }
if reasoning_effort == "low":
return {
"thinkingBudget": DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET,
@ -595,10 +597,6 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
optional_params["parallel_tool_calls"] = value
elif param == "seed":
optional_params["seed"] = value
elif param == "reasoning_effort" and isinstance(value, str):
optional_params["thinkingConfig"] = (
VertexGeminiConfig._map_reasoning_effort_to_thinking_budget(value)
)
elif param == "thinking":
optional_params["thinkingConfig"] = (
VertexGeminiConfig._map_thinking_param(
@ -613,6 +611,12 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
optional_params = self._add_tools_to_optional_params(
optional_params, [_tools]
)
if supports_reasoning(model):
optional_params["thinkingConfig"] = (
VertexGeminiConfig._map_reasoning_effort_to_thinking_budget(
non_default_params.get("reasoning_effort")
)
)
if litellm.vertex_ai_safety_settings is not None:
optional_params["safety_settings"] = litellm.vertex_ai_safety_settings

View file

@ -10,77 +10,152 @@ import httpx
import pytest
sys.path.insert(
0, os.path.abspath("../../..")
) # Adds the parent directory to the system-path
0, os.path.abspath("../../../../..")
) # Adds the parent directory to the system path
import litellm
from litellm.completion_extras.litellm_responses_transformation.transformation import (
LiteLLMResponsesTransformationHandler,
OpenAiResponsesToChatCompletionStreamIterator,
)
from litellm.types.llms.openai import Reasoning
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices
def test_convert_chat_completion_messages_to_responses_api_image_input():
from litellm.completion_extras.litellm_responses_transformation.transformation import (
LiteLLMResponsesTransformationHandler,
)
class TestLiteLLMResponsesTransformation:
def setup_method(self):
self.handler = LiteLLMResponsesTransformationHandler()
self.model = "responses-api-model"
self.logging_obj = MagicMock()
handler = LiteLLMResponsesTransformationHandler()
def test_transform_request_reasoning_effort(self):
"""
Test that reasoning_effort is mapped to reasoning parameter correctly.
"""
# Case 1: reasoning_effort = "high"
optional_params_high = {"reasoning_effort": "high"}
result_high = self.handler.transform_request(
model=self.model,
messages=[],
optional_params=optional_params_high,
litellm_params={},
headers={},
litellm_logging_obj=self.logging_obj,
)
assert "reasoning" in result_high
assert result_high["reasoning"] == Reasoning(effort="high", summary="detailed")
user_content = "What's in this image?"
user_image = "https://w7.pngwing.com/pngs/666/274/png-transparent-image-pictures-icon-photo-thumbnail.png"
# Case 2: reasoning_effort = "medium"
optional_params_medium = {"reasoning_effort": "medium"}
result_medium = self.handler.transform_request(
model=self.model,
messages=[],
optional_params=optional_params_medium,
litellm_params={},
headers={},
litellm_logging_obj=self.logging_obj,
)
assert "reasoning" in result_medium
assert result_medium["reasoning"] == Reasoning(effort="medium", summary="auto")
messages = [
{
"role": "user",
"content": [
{
"type": "text",
"text": user_content,
},
{
"type": "image_url",
"image_url": {"url": user_image},
},
],
},
]
# Case 3: reasoning_effort = "low"
optional_params_low = {"reasoning_effort": "low"}
result_low = self.handler.transform_request(
model=self.model,
messages=[],
optional_params=optional_params_low,
litellm_params={},
headers={},
litellm_logging_obj=self.logging_obj,
)
assert "reasoning" in result_low
assert result_low["reasoning"] == Reasoning(effort="low", summary="auto")
response, _ = handler.convert_chat_completion_messages_to_responses_api(messages)
# Case 4: no reasoning_effort
optional_params_none = {}
result_none = self.handler.transform_request(
model=self.model,
messages=[],
optional_params=optional_params_none,
litellm_params={},
headers={},
litellm_logging_obj=self.logging_obj,
)
assert "reasoning" in result_none
assert result_none["reasoning"] == Reasoning(summary="auto")
response_str = json.dumps(response)
# Case 5: reasoning_effort = None
optional_params_explicit_none = {"reasoning_effort": None}
result_explicit_none = self.handler.transform_request(
model=self.model,
messages=[],
optional_params=optional_params_explicit_none,
litellm_params={},
headers={},
litellm_logging_obj=self.logging_obj,
)
assert "reasoning" in result_explicit_none
assert result_explicit_none["reasoning"] == Reasoning(summary="auto")
assert user_content in response_str
assert user_image in response_str
def test_convert_chat_completion_messages_to_responses_api_image_input(self):
"""
Test that chat completion messages with image inputs are converted correctly.
"""
user_content = "What's in this image?"
user_image = "https://w7.pngwing.com/pngs/666/274/png-transparent-image-pictures-icon-photo-thumbnail.png"
print("response: ", response)
assert response[0]["content"][1]["image_url"] == user_image
messages = [
{
"role": "user",
"content": [
{
"type": "text",
"text": user_content,
},
{
"type": "image_url",
"image_url": {"url": user_image},
},
],
},
]
response, _ = self.handler.convert_chat_completion_messages_to_responses_api(messages)
def test_openai_responses_chunk_parser_reasoning_summary():
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices
response_str = json.dumps(response)
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
assert user_content in response_str
assert user_image in response_str
chunk = {
"delta": "**Compar",
"item_id": "rs_686d544208748198b6912e27b7c299c00e24bd875d35bade",
"output_index": 0,
"sequence_number": 4,
"summary_index": 0,
"type": "response.reasoning_summary_text.delta",
}
print("response: ", response)
assert response[0]["content"][1]["image_url"] == user_image
result = iterator.chunk_parser(chunk)
def test_openai_responses_chunk_parser_reasoning_summary(self):
"""
Test that OpenAI responses chunk parser handles reasoning summary correctly.
"""
iterator = OpenAiResponsesToChatCompletionStreamIterator(
streaming_response=None, sync_stream=True
)
assert isinstance(result, ModelResponseStream)
assert len(result.choices) == 1
choice = result.choices[0]
assert isinstance(choice, StreamingChoices)
assert choice.index == 0
delta = choice.delta
assert isinstance(delta, Delta)
assert delta.content is None
assert delta.reasoning_content == "**Compar"
assert delta.tool_calls is None
assert delta.function_call is None
chunk = {
"delta": "**Compar",
"item_id": "rs_686d544208748198b6912e27b7c299c00e24bd875d35bade",
"output_index": 0,
"sequence_number": 4,
"summary_index": 0,
"type": "response.reasoning_summary_text.delta",
}
result = iterator.chunk_parser(chunk)
assert isinstance(result, ModelResponseStream)
assert len(result.choices) == 1
choice = result.choices[0]
assert isinstance(choice, StreamingChoices)
assert choice.index == 0
delta = choice.delta
assert isinstance(delta, Delta)
assert delta.content is None
assert delta.reasoning_content == "**Compar"
assert delta.tool_calls is None
assert delta.function_call is None

View file

@ -441,6 +441,82 @@ def test_vertex_ai_map_thinking_param_with_budget_tokens_0():
}
def test_vertex_ai_reasoning_effort_mapping():
"""
Test that reasoning_effort is mapped to thinkingConfig correctly for models that support it.
- A default thinking config is applied if reasoning_effort is not specified.
- reasoning_effort correctly maps to thinkingConfig.
- No thinkingConfig is applied for models that do not support reasoning.
- reasoning_effort is prioritized over thinking param.
"""
v = VertexGeminiConfig()
optional_params = {}
# Case 1: Model supports reasoning, no reasoning_effort provided
# Should apply default thinkingConfig
with patch(
"litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini.supports_reasoning",
return_value=True,
):
result_params = v.map_openai_params(
non_default_params={},
optional_params=deepcopy(optional_params),
model="gemini-2.5-pro",
drop_params=False,
)
assert "thinkingConfig" in result_params
assert result_params["thinkingConfig"] == {"includeThoughts": True}
# Case 2: Model supports reasoning, reasoning_effort is 'low'
# Should apply thinkingConfig with budget
with patch(
"litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini.supports_reasoning",
return_value=True,
):
result_params_with_effort = v.map_openai_params(
non_default_params={"reasoning_effort": "low"},
optional_params=deepcopy(optional_params),
model="gemini-2.5-pro",
drop_params=False,
)
assert "thinkingConfig" in result_params_with_effort
assert result_params_with_effort["thinkingConfig"]["includeThoughts"] is True
assert "thinkingBudget" in result_params_with_effort["thinkingConfig"]
# Case 3: Model does not support reasoning
# Should not apply thinkingConfig
with patch(
"litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini.supports_reasoning",
return_value=False,
):
result_params_no_support = v.map_openai_params(
non_default_params={},
optional_params=deepcopy(optional_params),
model="gemini-pro",
drop_params=False,
)
assert "thinkingConfig" not in result_params_no_support
# Case 4: Model supports reasoning, but reasoning_effort is set, should be prioritized over thinking
with patch(
"litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini.supports_reasoning",
return_value=True,
):
result_params_with_effort = v.map_openai_params(
non_default_params={
"reasoning_effort": "low",
"thinking": {"type": "enabled", "budget_tokens": 1000},
},
optional_params=deepcopy(optional_params),
model="gemini-2.5-pro",
drop_params=False,
)
assert "thinkingConfig" in result_params_with_effort
assert result_params_with_effort["thinkingConfig"]["includeThoughts"] is True
assert "thinkingBudget" in result_params_with_effort["thinkingConfig"]
assert result_params_with_effort["thinkingConfig"]["thinkingBudget"] != 1000
def test_vertex_ai_map_tools():
v = VertexGeminiConfig()
tools = v._map_function(value=[{"code_execution": {}}])