diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index b5f32d57061..8ef1a15f17e 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -281,14 +281,24 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): # Fallback: delegate to parent for unknown types return super().get_json_schema_from_pydantic_object(response_format) + @staticmethod + def _strip_regional_prefix(model: str) -> str: + return re.sub( + r"(^|/)(?:[a-z0-9_-]+\.)+(?=gem(?:ini|ma)-)", + r"\1", + model, + flags=re.IGNORECASE, + ) + @staticmethod def _is_gemini_3_or_newer(model: str) -> bool: """ Check if the model is Gemini 3 or newer. """ - model_name: Final = model.split("/")[-1].lower() + normalized_model: Final = VertexGeminiConfig._strip_regional_prefix(model).lower() + model_name: Final = normalized_model.split("/")[-1] is_vertex_fine_tuned_model: Final = model_name.isdigit() or ( - model.startswith("gemini/") and not model_name.startswith("gemini-") + normalized_model.startswith("gemini/") and not model_name.startswith("gemini-") ) if not model_name or is_vertex_fine_tuned_model or model_name.startswith("gemma-"): return False @@ -297,6 +307,18 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): return False return True + @staticmethod + def _is_gemini_3_flash(model: str | None) -> bool: + if not model: + return False + return VertexGeminiConfig._is_gemini_3_or_newer(model) and "flash" in model.lower() + + @staticmethod + def _is_gemini_3_1_pro_or_newer(model: str | None) -> bool: + if not model: + return False + return bool(re.search(r"gemini-3\.\d+-pro", model.lower())) + @staticmethod def _forward_gemini_function_call_id(model: str) -> bool: """ @@ -346,7 +368,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): if self._supports_penalty_parameters(model): supported_params.extend(["frequency_penalty", "presence_penalty"]) - if supports_reasoning(model) or self._is_gemini_3_or_newer(model): + if supports_reasoning(self._strip_regional_prefix(model)) or self._is_gemini_3_or_newer(model): supported_params.append("reasoning_effort") supported_params.append("thinking") @@ -870,10 +892,9 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): @staticmethod def _supports_minimal_thinking_level(model: str) -> bool: - lowered: Final = model.lower() - is_gemini3_or_newer_flash: Final = VertexGeminiConfig._is_gemini_3_or_newer(model) and "flash" in lowered - return is_gemini3_or_newer_flash and not is_explicitly_disabled_factory( - model=model, custom_llm_provider=None, key="supports_minimal_reasoning_effort" + normalized_model: Final = VertexGeminiConfig._strip_regional_prefix(model) + return VertexGeminiConfig._is_gemini_3_flash(normalized_model) and not is_explicitly_disabled_factory( + model=normalized_model, custom_llm_provider=None, key="supports_minimal_reasoning_effort" ) @staticmethod @@ -966,19 +987,17 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): # For Gemini 3+ models, use thinkingLevel instead of thinkingBudget if model and VertexGeminiConfig._is_gemini_3_or_newer(model): - if thinking_enabled: - if thinking_budget == 0: - params["includeThoughts"] = False - else: - params["includeThoughts"] = True - # Follow provider defaults unless explicitly opted into legacy behavior. - if litellm.enable_gemini_default_thinking_level_low is True: - params["thinkingLevel"] = ( - "minimal" if VertexGeminiConfig._supports_minimal_thinking_level(model) else "low" - ) + default_low_thinking_level: Final = ( + "minimal" if VertexGeminiConfig._supports_minimal_thinking_level(model) else "low" + ) + if thinking_enabled and thinking_budget != 0: + params["includeThoughts"] = True + # Follow provider defaults unless explicitly opted into legacy behavior. + if litellm.enable_gemini_default_thinking_level_low is True: + params["thinkingLevel"] = default_low_thinking_level else: - # Thinking disabled params["includeThoughts"] = False + params["thinkingLevel"] = default_low_thinking_level else: # For older Gemini models, use thinkingBudget if thinking_enabled and not VertexGeminiConfig._is_thinking_budget_zero(thinking_budget): diff --git a/tests/llm_translation/test_gemini.py b/tests/llm_translation/test_gemini.py index 7b0b741563d..937cf383604 100644 --- a/tests/llm_translation/test_gemini.py +++ b/tests/llm_translation/test_gemini.py @@ -1579,10 +1579,7 @@ def test_anthropic_thinking_param_to_gemini_3_provider_defaults(): ) assert result_disabled.get("includeThoughts") is False - assert ( - "thinkingLevel" not in result_disabled - or result_disabled.get("thinkingLevel") is None - ) + assert result_disabled.get("thinkingLevel") == "low" # Test 3: Budget tokens = 0 for Gemini 3 thinking_param_zero: AnthropicThinkingParam = { @@ -1596,10 +1593,7 @@ def test_anthropic_thinking_param_to_gemini_3_provider_defaults(): ) assert result_zero["includeThoughts"] is False - assert ( - "thinkingLevel" not in result_zero - or result_zero.get("thinkingLevel") is None - ) + assert result_zero["thinkingLevel"] == "minimal" # Test 4: Gemini 3 flash-preview should also follow provider defaults by default result_gemini3flashpreview = VertexGeminiConfig._map_thinking_param( diff --git a/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py b/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py index fd735afb16e..bbadec16660 100644 --- a/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py +++ b/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py @@ -6356,3 +6356,184 @@ def test_gemini_multi_candidate_messages_do_not_share_state(): assert resp.choices[1].message.tool_calls is None assert getattr(resp.choices[1].message, "reasoning_content", None) is None assert resp.choices[1].provider_specific_fields["native_finish_reason"] == "STOP" + + +@pytest.mark.parametrize( + ("model", "expected_is_gemini_3", "expected_is_flash", "expected_is_3_1_pro_or_newer"), + [ + ("gemini-3.5-flash", True, True, False), + ("gemini-3.5-flash-preview", True, True, False), + ("gemini-3.1-flash", True, True, False), + ("gemini-3.1-flash-lite-preview", True, True, False), + ("gemini-3-flash", True, True, False), + ("au.gemini-3.5-flash", True, True, False), + ("eu.gemini-3.5-flash", True, True, False), + ("us.gemini-3.5-flash", True, True, False), + ("ca.gemini-3.5-flash", True, True, False), + ("jp.gemini-3.5-flash", True, True, False), + ("uk.gemini-3.5-flash", True, True, False), + ("in.gemini-3.5-flash", True, True, False), + ("sg.gemini-3.5-flash", True, True, False), + ("kr.gemini-3.5-flash", True, True, False), + ("global.gemini-3.5-flash", True, True, False), + ("apac.gemini-3.5-flash", True, True, False), + ("us-central1.gemini-3.5-flash", True, True, False), + ("europe-west4.gemini-3.5-flash", True, True, False), + ("australia-southeast1.gemini-3.5-flash", True, True, False), + ("asia-northeast1.gemini-3.5-flash", True, True, False), + ("vertex_ai/au.gemini-3.5-flash", True, True, False), + ("vertex_ai/eu.gemini-3.5-flash", True, True, False), + ("vertex_ai/us-central1.gemini-3.5-flash", True, True, False), + ("gemini/au.gemini-3.5-flash", True, True, False), + ("gemini/eu.gemini-3.5-flash", True, True, False), + ("gemini-3.1-pro", True, False, True), + ("gemini-3.1-pro-preview", True, False, True), + ("gemini-3.5-pro", True, False, True), + ("au.gemini-3.5-pro", True, False, True), + ("eu.gemini-3.1-pro", True, False, True), + ("us-central1.gemini-3.5-pro", True, False, True), + ("gemini-3-pro-preview", True, False, False), + ("au.gemini-2.5-flash", False, False, False), + ("eu.gemini-2.5-pro", False, False, False), + ("us.gemini-2.5-flash", False, False, False), + ("us-central1.gemini-2.0-flash", False, False, False), + ("vertex_ai/au.gemini-2.5-flash", False, False, False), + ("gemini/eu.gemini-2.5-flash", False, False, False), + ("eu.gemma-3-27b-it", False, False, False), + ], +) +def test_gemini_3_helpers_with_regional_prefixes( + model: str, + expected_is_gemini_3: bool, + expected_is_flash: bool, + expected_is_3_1_pro_or_newer: bool, +): + assert VertexGeminiConfig._is_gemini_3_or_newer(model) is expected_is_gemini_3 + assert VertexGeminiConfig._is_gemini_3_flash(model) is expected_is_flash + assert VertexGeminiConfig._is_gemini_3_1_pro_or_newer(model) is expected_is_3_1_pro_or_newer + + +@pytest.mark.parametrize( + "model", + [ + "gemini-3.5-flash", + "au.gemini-3.5-flash", + "eu.gemini-3.5-flash", + "us.gemini-3.5-flash", + "global.gemini-3.5-flash", + "us-central1.gemini-3.5-flash", + "europe-west4.gemini-3.5-flash", + "australia-southeast1.gemini-3.5-flash", + "vertex_ai/au.gemini-3.5-flash", + "gemini/au.gemini-3.5-flash", + "gemini-3.5-flash-preview", + "gemini-3.1-flash", + "gemini-3-flash", + ], +) +@pytest.mark.parametrize( + ("reasoning_effort", "expected"), + [ + ("none", {"thinkingLevel": "minimal", "includeThoughts": False}), + ("disable", {"thinkingLevel": "minimal", "includeThoughts": False}), + ("minimal", {"thinkingLevel": "minimal", "includeThoughts": True}), + ("low", {"thinkingLevel": "low", "includeThoughts": True}), + ("medium", {"thinkingLevel": "medium", "includeThoughts": True}), + ("high", {"thinkingLevel": "high", "includeThoughts": True}), + ], +) +def test_gemini_3_flash_reasoning_effort_mapping_all_regions( + local_model_cost_map, + model: str, + reasoning_effort: str, + expected, +): + result: Final = VertexGeminiConfig._map_reasoning_effort_to_thinking_level( + reasoning_effort=reasoning_effort, + model=model, + ) + assert result == expected + + +@pytest.mark.parametrize( + "model", + [ + "gemini-3.1-pro", + "gemini-3.1-pro-preview", + "gemini-3.5-pro", + "au.gemini-3.1-pro", + "eu.gemini-3.5-pro", + "us-central1.gemini-3.5-pro", + "vertex_ai/au.gemini-3.7-flash", + "us-central1.gemini-3.8-flash", + ], +) +@pytest.mark.parametrize( + ("reasoning_effort", "expected"), + [ + ("none", {"thinkingLevel": "low", "includeThoughts": False}), + ("disable", {"thinkingLevel": "low", "includeThoughts": False}), + ("minimal", {"thinkingLevel": "low", "includeThoughts": True}), + ("low", {"thinkingLevel": "low", "includeThoughts": True}), + ("medium", {"thinkingLevel": "medium", "includeThoughts": True}), + ("high", {"thinkingLevel": "high", "includeThoughts": True}), + ], +) +def test_gemini_3_pro_and_37_38_flash_reasoning_effort_mapping_all_regions( + local_model_cost_map, + model: str, + reasoning_effort: str, + expected, +): + result: Final = VertexGeminiConfig._map_reasoning_effort_to_thinking_level( + reasoning_effort=reasoning_effort, + model=model, + ) + assert result == expected + + +@pytest.mark.parametrize( + ("model", "expected_level"), + [ + ("gemini-3.5-flash", "minimal"), + ("au.gemini-3.5-flash", "minimal"), + ("eu.gemini-3.5-flash", "minimal"), + ("us.gemini-3.5-flash", "minimal"), + ("us-central1.gemini-3.5-flash", "minimal"), + ("europe-west4.gemini-3.5-flash", "minimal"), + ("australia-southeast1.gemini-3.5-flash", "minimal"), + ("vertex_ai/au.gemini-3.5-flash", "minimal"), + ("gemini/eu.gemini-3.5-flash", "minimal"), + ("gemini-3.5-flash-preview", "minimal"), + ("gemini-3.1-flash", "minimal"), + ("gemini-3-flash", "minimal"), + ("gemini-3.1-pro", "low"), + ("gemini-3.1-pro-preview", "low"), + ("gemini-3.5-pro", "low"), + ("au.gemini-3.5-pro", "low"), + ("gemini-3.8-flash", "low"), + ("au.gemini-3.8-flash", "low"), + ], +) +@pytest.mark.parametrize( + "thinking_param", + [ + {"type": "disabled"}, + {"type": "enabled", "budget_tokens": 0}, + ], +) +def test_map_thinking_param_disabled_or_zero_budget_gemini_3( + local_model_cost_map, + model: str, + expected_level: str, + thinking_param, +): + result: Final = VertexGeminiConfig._map_thinking_param( + thinking_param=thinking_param, + model=model, + ) + assert result == { + "thinkingLevel": expected_level, + "includeThoughts": False, + } +