Merge pull request #41201 from BerriAI/litellm_gemini_37_38_flash_no_minimal_thinking

fix(gemini): map minimal thinking to low for Gemini 3.7 and 3.8 Flash
This commit is contained in:
Mateo Wang 2026-09-16 16:52:08 -07:00 • committed by GitHub
commit 319f427c40
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 159 additions and 31 deletions

View file

@ -4,9 +4,9 @@ from typing import Final
import litellm
from litellm.utils import (
_is_explicitly_disabled_factory,
_supports_factory,
declared_value_factory,
is_explicitly_disabled_factory,
)
from .gpt_transformation import OpenAIGPTConfig
@ -192,7 +192,7 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
Use this for opt-out checks where unknown models should be allowed through.
"""
return _is_explicitly_disabled_factory(
return is_explicitly_disabled_factory(
model=cls._model_map_lookup_name(model),
custom_llm_provider=None,
key=f"supports_{level}_reasoning_effort",

View file

@ -79,6 +79,7 @@ from litellm.utils import (
CustomStreamWrapper,
ModelResponse,
is_base64_encoded,
is_explicitly_disabled_factory,
supports_reasoning,
)
@ -866,6 +867,14 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
else:
raise _unsupported_reasoning_effort(reasoning_effort)
@staticmethod
def _supports_minimal_thinking_level(model: str) -> bool:
lowered: Final = model.lower()
is_gemini3flash: Final = "gemini-3" in lowered and "flash" in lowered
return is_gemini3flash and not is_explicitly_disabled_factory(
model=model, custom_llm_provider=None, key="supports_minimal_reasoning_effort"
)
@staticmethod
def _map_reasoning_effort_to_thinking_level(
reasoning_effort: str,
@ -880,13 +889,11 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
Returns:
GeminiThinkingConfig with thinkingLevel and includeThoughts
"""
# Check if this is gemini-3-flash which supports MINIMAL thinking level
# Covers gemini-3-flash, gemini-3-flash-preview, gemini-3.1-flash, gemini-3.1-flash-lite-preview,
# gemini-3.5-flash, and any future 3.x-flash variants.
is_gemini3flash: Final = model and ("flash" in model.lower() and "gemini-3" in model.lower())
supports_minimal: Final = bool(model) and VertexGeminiConfig._supports_minimal_thinking_level(model)
is_gemini31pro: Final = model and ("gemini-3.1-pro-preview" in model.lower())
if reasoning_effort == "minimal":
if is_gemini3flash:
if supports_minimal:
return {"thinkingLevel": "minimal", "includeThoughts": True}
else:
return {"thinkingLevel": "low", "includeThoughts": True}
@ -899,18 +906,11 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
return {"thinkingLevel": "high", "includeThoughts": True}
elif reasoning_effort == "high":
return {"thinkingLevel": "high", "includeThoughts": True}
elif reasoning_effort == "disable":
# Gemini 3 cannot fully disable thinking, so we use "minimal" for gemini-3-flash-preview, "low" for others
if is_gemini3flash:
return {"thinkingLevel": "minimal", "includeThoughts": False}
else:
return {"thinkingLevel": "low", "includeThoughts": False}
elif reasoning_effort == "none":
# For gemini-3-flash-preview, use "minimal" instead of "low"
if is_gemini3flash:
return {"thinkingLevel": "minimal", "includeThoughts": False}
else:
return {"thinkingLevel": "low", "includeThoughts": False}
elif reasoning_effort in ("disable", "none"):
return {
"thinkingLevel": "minimal" if supports_minimal else "low",
"includeThoughts": False,
}
else:
raise _unsupported_reasoning_effort(reasoning_effort)
@ -977,8 +977,9 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig):
params["includeThoughts"] = True
# Follow provider defaults unless explicitly opted into legacy behavior.
if litellm.enable_gemini_default_thinking_level_low is True:
is_gemini3flash: Final = "gemini-3" in model.lower() and "flash" in model.lower()
params["thinkingLevel"] = "minimal" if is_gemini3flash else "low"
params["thinkingLevel"] = (
"minimal" if VertexGeminiConfig._supports_minimal_thinking_level(model) else "low"
)
else:
# Thinking disabled
params["includeThoughts"] = False

View file

@ -26105,6 +26105,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -26162,6 +26163,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -28111,6 +28113,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -28170,6 +28173,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -28592,6 +28596,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -28649,6 +28654,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,

View file

@ -2689,7 +2689,7 @@ def declared_value_factory(model: str, custom_llm_provider: str | None, key: str
"""Return a string value the model map declares for *key*, or ``None`` when it says nothing.
The string-valued sibling of :func:`_supports_factory` and
:func:`_is_explicitly_disabled_factory`, public where those two are not because it is read
:func:`is_explicitly_disabled_factory`, public like the latter because both are read
from the provider configs rather than from this module, sharing their
``get_llm_provider`` -> ``_get_model_info_helper`` chain and their unprefixed-twin
fallback (#20885), so a provider-prefixed entry that omits the key still answers
@ -2725,7 +2725,7 @@ def declared_value_factory(model: str, custom_llm_provider: str | None, key: str
return None
def _is_explicitly_disabled_factory(model: str, custom_llm_provider: str | None, key: str) -> bool:
def is_explicitly_disabled_factory(model: str, custom_llm_provider: str | None, key: str) -> bool:
"""Return True only when the model map explicitly sets *key* to ``False``.
This is the opt-out mirror of :func:`_supports_factory`. Where
@ -2844,7 +2844,7 @@ def is_vision_explicitly_disabled(model: str, custom_llm_provider: str | None =
The opt-out mirror of :func:`supports_vision`: a missing declaration reads as not
disabled, so unknown or newly added models stay eligible for image routing.
"""
return _is_explicitly_disabled_factory(model, custom_llm_provider, "supports_vision")
return is_explicitly_disabled_factory(model, custom_llm_provider, "supports_vision")
def supports_vision(model: str, custom_llm_provider: str | None = None) -> bool:

View file

@ -26105,6 +26105,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -26162,6 +26163,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -28111,6 +28113,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -28170,6 +28173,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -28592,6 +28596,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
@ -28649,6 +28654,7 @@
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_minimal_reasoning_effort": false,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,

View file

@ -8,7 +8,7 @@ from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config
from litellm.llms.openai.openai import OpenAIConfig
from litellm.utils import (
_is_explicitly_disabled_factory,
is_explicitly_disabled_factory,
peek_reasoning_summary_aliases,
strip_reasoning_summary_aliases_from_optional_params,
)
@ -524,19 +524,19 @@ def test_gpt5_minimal_explicitly_disabled_check(gpt5_config: OpenAIGPT5Config):
def test_is_explicitly_disabled_factory_minimal():
"""_is_explicitly_disabled_factory returns True only for explicit False entries.
"""is_explicitly_disabled_factory returns True only for explicit False entries.
Verifies the shared helper used by _is_reasoning_effort_level_explicitly_disabled
directly — so future changes to the helper are caught without going through the
method wrapper.
"""
key = "supports_minimal_reasoning_effort"
assert _is_explicitly_disabled_factory("gpt-5.4-mini", None, key)
assert _is_explicitly_disabled_factory("gpt-5.4-nano", None, key)
assert _is_explicitly_disabled_factory("openai/gpt-5.4-mini", None, key)
assert _is_explicitly_disabled_factory("gpt-5.4", None, key)
assert _is_explicitly_disabled_factory("gpt-5.4-pro", None, key)
assert not _is_explicitly_disabled_factory("gpt-5.4-turbo-preview", None, key)
assert is_explicitly_disabled_factory("gpt-5.4-mini", None, key)
assert is_explicitly_disabled_factory("gpt-5.4-nano", None, key)
assert is_explicitly_disabled_factory("openai/gpt-5.4-mini", None, key)
assert is_explicitly_disabled_factory("gpt-5.4", None, key)
assert is_explicitly_disabled_factory("gpt-5.4-pro", None, key)
assert not is_explicitly_disabled_factory("gpt-5.4-turbo-preview", None, key)
def test_gpt5_unknown_model_passes_through_minimal(config: OpenAIConfig):

View file

@ -5,11 +5,14 @@ from copy import deepcopy
from typing import Final, List, cast
from unittest.mock import MagicMock, patch
import httpx
import pytest
from pydantic import BaseModel
import litellm
from litellm import ModelResponse, completion
from litellm.llms.anthropic.experimental_pass_through.messages import handler as anthropic_messages_handler
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.llms.gemini.chat.transformation import GoogleAIStudioGeminiConfig
from litellm.llms.vertex_ai.common_utils import VertexAIError
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
@ -2678,6 +2681,118 @@ def test_reasoning_effort_maps_to_thinking_level_gemini_3():
assert result["thinkingConfig"]["includeThoughts"] is False
@pytest.mark.parametrize(
"model",
[
"gemini-3.7-flash",
"vertex_ai/gemini-3.8-flash",
"gemini/gemini-3.8-flash",
],
)
@pytest.mark.parametrize(
("reasoning_effort", "include_thoughts"),
[("minimal", True), ("none", False), ("disable", False)],
)
def test_gemini_37_38_flash_floor_minimal_thinking_level(
local_model_cost_map, model, reasoning_effort, include_thoughts
):
result = VertexGeminiConfig._map_reasoning_effort_to_thinking_level(
reasoning_effort, model
)
assert result["thinkingLevel"] == "low"
assert result["includeThoughts"] is include_thoughts
@pytest.mark.parametrize(
("model", "reasoning_effort", "expected_level", "include_thoughts"),
[
("gemini-3-flash-preview", "minimal", "minimal", True),
("gemini-3-flash-preview", "none", "minimal", False),
("gemini-3-flash-preview", "disable", "minimal", False),
("gemini-3.6-flash", "minimal", "minimal", True),
("gemini-3.6-flash", "none", "minimal", False),
("gemini-3.6-flash", "disable", "minimal", False),
("gemini-3.5-flash", "minimal", "minimal", True),
("gemini-3.5-flash", "none", "minimal", False),
("gemini-3.5-flash", "disable", "minimal", False),
("gemini-3.8-flash", "medium", "medium", True),
],
)
def test_gemini_flash_minimal_thinking_support(
local_model_cost_map, model, reasoning_effort, expected_level, include_thoughts
):
result = VertexGeminiConfig._map_reasoning_effort_to_thinking_level(
reasoning_effort, model
)
assert result["thinkingLevel"] == expected_level
assert result["includeThoughts"] is include_thoughts
def test_gemini_38_flash_feature_flag_uses_low_thinking_level(local_model_cost_map, monkeypatch):
monkeypatch.setattr(litellm, "enable_gemini_default_thinking_level_low", True)
thinking_param = {"type": "enabled", "budget_tokens": 1024}
result_38 = VertexGeminiConfig._map_thinking_param(
thinking_param, model="gemini-3.8-flash"
)
result_36 = VertexGeminiConfig._map_thinking_param(
thinking_param, model="gemini-3.6-flash"
)
assert result_38["thinkingLevel"] == "low"
assert result_36["thinkingLevel"] == "minimal"
def test_gemini_38_flash_public_reasoning_effort_none_uses_low(local_model_cost_map):
result = VertexGeminiConfig().map_openai_params(
non_default_params={"reasoning_effort": "none"},
optional_params={},
model="gemini-3.8-flash",
drop_params=False,
)
assert result["thinkingConfig"] == {
"thinkingLevel": "low",
"includeThoughts": False,
}
@pytest.mark.asyncio
async def test_gemini_38_flash_messages_bridge_thinking_disabled_sends_low_thinking_level(local_model_cost_map):
captured: dict[str, dict] = {}
def upstream(request: httpx.Request) -> httpx.Response:
captured["body"] = json.loads(request.content)
return httpx.Response(
200,
json={
"candidates": [{"content": {"parts": [{"text": "hi"}], "role": "model"}, "finishReason": "STOP"}],
"usageMetadata": {"promptTokenCount": 1, "candidatesTokenCount": 1, "totalTokenCount": 2},
},
request=request,
)
client = AsyncHTTPHandler()
client.client = httpx.AsyncClient(transport=httpx.MockTransport(upstream))
await anthropic_messages_handler.anthropic_messages(
max_tokens=16,
messages=[{"role": "user", "content": "hi"}],
model="gemini/gemini-3.8-flash",
custom_llm_provider="gemini",
thinking={"type": "disabled"},
api_key="fake-gemini-key",
client=client,
)
assert captured["body"]["generationConfig"]["thinkingConfig"] == {
"thinkingLevel": "low",
"includeThoughts": False,
}
def test_reasoning_effort_dict_format_gemini_3():
"""
Test that reasoning_effort works when passed as dict format from OpenAI Agents SDK.