From 729cb0f2445b96888ca68ffaf1bb6a56d6bb8b3c Mon Sep 17 00:00:00 2001 From: ArthurAAM <100235777+ArthurAAM@users.noreply.github.com> Date: Tue, 25 Aug 2026 18:15:04 -0300 Subject: [PATCH] fix(vertex_ai): keep Gemini 1.x off the responseJsonSchema channel Neither the global setting nor the per request override can select a channel the model has no field for, so asking for the JSON Schema channel on a Gemini 1.x model now logs a warning and keeps the natively converted responseSchema. --- litellm/llms/vertex_ai/common_utils.py | 25 ++++++++++++----- .../vertex_ai/gemini/test_transformation.py | 27 ++++++++++--------- .../vertex_ai/test_vertex_ai_common_utils.py | 9 +++++-- 3 files changed, 41 insertions(+), 20 deletions(-) diff --git a/litellm/llms/vertex_ai/common_utils.py b/litellm/llms/vertex_ai/common_utils.py index 03f28b1fb66..90ac9d2a061 100644 --- a/litellm/llms/vertex_ai/common_utils.py +++ b/litellm/llms/vertex_ai/common_utils.py @@ -271,10 +271,16 @@ def supports_response_json_schema(model: str) -> bool: return bool(gemini_2_plus_pattern.search(model_lower)) +GEMINI_1_MODEL_PATTERN: Final = re.compile(r"gemini-1(?:\.|-)") VERTEX_AI_USE_RESPONSE_JSON_SCHEMA_PARAM: Final = "vertex_ai_use_response_json_schema" VERTEX_AI_VERBATIM_RESPONSE_SCHEMA_PARAM: Final = "litellm_param_vertex_ai_verbatim_response_schema" +def _rejects_response_json_schema(model: str) -> bool: + """Gemini 1.x generateContent has no responseJsonSchema field, so no override can reach it""" + return bool(GEMINI_1_MODEL_PATTERN.search(model.lower())) + + def should_use_response_json_schema(model: str, request_override: bool | None = None) -> bool: """ Resolve which structured output channel a json_schema response_format goes to. @@ -283,13 +289,20 @@ def should_use_response_json_schema(model: str, request_override: bool | None = natively converted ``responseSchema`` (nullable unions flattened, constraints hoisted, ``propertyOrdering`` added). Precedence: per request ``vertex_ai_use_response_json_schema``, then - ``litellm.vertex_ai_use_response_json_schema``, then the model heuristic + ``litellm.vertex_ai_use_response_json_schema``, then the model heuristic. Neither + override can select a channel the model has no field for """ - if request_override is not None: - return request_override - if litellm.vertex_ai_use_response_json_schema is not None: - return litellm.vertex_ai_use_response_json_schema - return supports_response_json_schema(model) + override: Final = request_override if request_override is not None else litellm.vertex_ai_use_response_json_schema + if override is None: + return supports_response_json_schema(model) + if override and _rejects_response_json_schema(model): + verbose_logger.warning( + "vertex_ai_use_response_json_schema=True ignored for model=%s: it has no responseJsonSchema field, " + "so the schema stays on responseSchema", + model, + ) + return False + return override from typing import Literal diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_transformation.py b/tests/test_litellm/llms/vertex_ai/gemini/test_transformation.py index 5ffcb76bfff..a1a7ae16500 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_transformation.py @@ -1,5 +1,7 @@ import json +from collections.abc import Mapping +from typing import Optional import pytest @@ -354,7 +356,14 @@ RESPONSE_SCHEMA_CHANNEL_CLIENT_SCHEMA = { } -def _gemini_request_body(model: str, litellm_params: dict, **completion_kwargs) -> RequestBody: +def _gemini_request_body( + model: str, + litellm_params: Mapping[str, bool], + request_override: Optional[bool] = None, +) -> RequestBody: + override_kwargs: dict[str, bool] = ( + {} if request_override is None else {"vertex_ai_use_response_json_schema": request_override} + ) optional_params = litellm.utils.get_optional_params( model=model, custom_llm_provider="vertex_ai", @@ -365,14 +374,14 @@ def _gemini_request_body(model: str, litellm_params: dict, **completion_kwargs) "schema": json.loads(json.dumps(RESPONSE_SCHEMA_CHANNEL_CLIENT_SCHEMA)), }, }, - **completion_kwargs, + **override_kwargs, ) return transformation._transform_request_body( messages=[{"role": "user", "content": "extract it"}], model=model, optional_params=optional_params, custom_llm_provider="vertex_ai", - litellm_params=litellm_params, + litellm_params=dict(litellm_params), cached_content=None, ) @@ -382,9 +391,7 @@ def test__transform_request_body_per_request_response_json_schema_opt_out(): vertex_ai_use_response_json_schema=False on the request puts the schema on Vertex's native responseSchema channel, and the knob itself never reaches the provider body """ - body = _gemini_request_body( - "gemini-2.5-flash", {}, vertex_ai_use_response_json_schema=False - ) + body = _gemini_request_body("gemini-2.5-flash", {}, request_override=False) generation_config = body["generationConfig"] assert "response_json_schema" not in generation_config @@ -398,9 +405,7 @@ def test__transform_request_body_per_request_response_json_schema_opt_out(): def test__transform_request_body_deployment_response_json_schema_opt_out(): """A deployment's litellm_params opts every request routed to it out of responseJsonSchema""" - body = _gemini_request_body( - "gemini-2.5-flash", {"vertex_ai_use_response_json_schema": False} - ) + body = _gemini_request_body("gemini-2.5-flash", {"vertex_ai_use_response_json_schema": False}) generation_config = body["generationConfig"] assert "response_json_schema" not in generation_config @@ -414,9 +419,7 @@ def test__transform_request_body_per_request_opt_in_beats_global_opt_out(monkeyp """ monkeypatch.setattr(litellm, "vertex_ai_use_response_json_schema", False) - body = _gemini_request_body( - "gemini-2.5-flash", {}, vertex_ai_use_response_json_schema=True - ) + body = _gemini_request_body("gemini-2.5-flash", {}, request_override=True) generation_config = body["generationConfig"] assert "response_schema" not in generation_config diff --git a/tests/test_litellm/llms/vertex_ai/test_vertex_ai_common_utils.py b/tests/test_litellm/llms/vertex_ai/test_vertex_ai_common_utils.py index 897dd23cdc1..eef69722830 100644 --- a/tests/test_litellm/llms/vertex_ai/test_vertex_ai_common_utils.py +++ b/tests/test_litellm/llms/vertex_ai/test_vertex_ai_common_utils.py @@ -200,7 +200,9 @@ CLIENT_SCHEMA = { (None, None, "gemini-2.5-flash", True), (None, None, "gemini-1.5-pro", False), (False, None, "gemini-2.5-flash", False), - (True, None, "gemini-1.5-pro", True), + (True, None, "gemini-flash-latest", True), + (True, None, "gemini-1.5-pro", False), + (None, True, "gemini-1.5-pro", False), (False, True, "gemini-2.5-flash", True), (True, False, "gemini-2.5-flash", False), ], @@ -208,7 +210,10 @@ CLIENT_SCHEMA = { def test_should_use_response_json_schema_precedence( monkeypatch, global_setting, request_override, model, expected ): - """Per request override beats litellm.vertex_ai_use_response_json_schema, which beats the model heuristic""" + """ + Per request override beats litellm.vertex_ai_use_response_json_schema, which beats the model + heuristic, and neither can put a schema on a channel Gemini 1.x has no field for + """ monkeypatch.setattr(litellm, "vertex_ai_use_response_json_schema", global_setting) assert should_use_response_json_schema(model, request_override) is expected