fix(vertex_ai): keep Gemini 1.x off the responseJsonSchema channel

Neither the global setting nor the per request override can select a channel the
model has no field for, so asking for the JSON Schema channel on a Gemini 1.x
model now logs a warning and keeps the natively converted responseSchema.
This commit is contained in:
ArthurAAM 2026-08-25 18:15:04 -03:00
parent 83e3458f8a
commit 729cb0f244
3 changed files with 41 additions and 20 deletions

View file

@ -271,10 +271,16 @@ def supports_response_json_schema(model: str) -> bool:
return bool(gemini_2_plus_pattern.search(model_lower))
GEMINI_1_MODEL_PATTERN: Final = re.compile(r"gemini-1(?:\.|-)")
VERTEX_AI_USE_RESPONSE_JSON_SCHEMA_PARAM: Final = "vertex_ai_use_response_json_schema"
VERTEX_AI_VERBATIM_RESPONSE_SCHEMA_PARAM: Final = "litellm_param_vertex_ai_verbatim_response_schema"
def _rejects_response_json_schema(model: str) -> bool:
"""Gemini 1.x generateContent has no responseJsonSchema field, so no override can reach it"""
return bool(GEMINI_1_MODEL_PATTERN.search(model.lower()))
def should_use_response_json_schema(model: str, request_override: bool | None = None) -> bool:
"""
Resolve which structured output channel a json_schema response_format goes to.
@ -283,13 +289,20 @@ def should_use_response_json_schema(model: str, request_override: bool | None =
natively converted ``responseSchema`` (nullable unions flattened, constraints
hoisted, ``propertyOrdering`` added). Precedence: per request
``vertex_ai_use_response_json_schema``, then
``litellm.vertex_ai_use_response_json_schema``, then the model heuristic
``litellm.vertex_ai_use_response_json_schema``, then the model heuristic. Neither
override can select a channel the model has no field for
"""
if request_override is not None:
return request_override
if litellm.vertex_ai_use_response_json_schema is not None:
return litellm.vertex_ai_use_response_json_schema
return supports_response_json_schema(model)
override: Final = request_override if request_override is not None else litellm.vertex_ai_use_response_json_schema
if override is None:
return supports_response_json_schema(model)
if override and _rejects_response_json_schema(model):
verbose_logger.warning(
"vertex_ai_use_response_json_schema=True ignored for model=%s: it has no responseJsonSchema field, "
"so the schema stays on responseSchema",
model,
)
return False
return override
from typing import Literal

View file

@ -1,5 +1,7 @@
import json
from collections.abc import Mapping
from typing import Optional
import pytest
@ -354,7 +356,14 @@ RESPONSE_SCHEMA_CHANNEL_CLIENT_SCHEMA = {
}
def _gemini_request_body(model: str, litellm_params: dict, **completion_kwargs) -> RequestBody:
def _gemini_request_body(
model: str,
litellm_params: Mapping[str, bool],
request_override: Optional[bool] = None,
) -> RequestBody:
override_kwargs: dict[str, bool] = (
{} if request_override is None else {"vertex_ai_use_response_json_schema": request_override}
)
optional_params = litellm.utils.get_optional_params(
model=model,
custom_llm_provider="vertex_ai",
@ -365,14 +374,14 @@ def _gemini_request_body(model: str, litellm_params: dict, **completion_kwargs)
"schema": json.loads(json.dumps(RESPONSE_SCHEMA_CHANNEL_CLIENT_SCHEMA)),
},
},
**completion_kwargs,
**override_kwargs,
)
return transformation._transform_request_body(
messages=[{"role": "user", "content": "extract it"}],
model=model,
optional_params=optional_params,
custom_llm_provider="vertex_ai",
litellm_params=litellm_params,
litellm_params=dict(litellm_params),
cached_content=None,
)
@ -382,9 +391,7 @@ def test__transform_request_body_per_request_response_json_schema_opt_out():
vertex_ai_use_response_json_schema=False on the request puts the schema on Vertex's native
responseSchema channel, and the knob itself never reaches the provider body
"""
body = _gemini_request_body(
"gemini-2.5-flash", {}, vertex_ai_use_response_json_schema=False
)
body = _gemini_request_body("gemini-2.5-flash", {}, request_override=False)
generation_config = body["generationConfig"]
assert "response_json_schema" not in generation_config
@ -398,9 +405,7 @@ def test__transform_request_body_per_request_response_json_schema_opt_out():
def test__transform_request_body_deployment_response_json_schema_opt_out():
"""A deployment's litellm_params opts every request routed to it out of responseJsonSchema"""
body = _gemini_request_body(
"gemini-2.5-flash", {"vertex_ai_use_response_json_schema": False}
)
body = _gemini_request_body("gemini-2.5-flash", {"vertex_ai_use_response_json_schema": False})
generation_config = body["generationConfig"]
assert "response_json_schema" not in generation_config
@ -414,9 +419,7 @@ def test__transform_request_body_per_request_opt_in_beats_global_opt_out(monkeyp
"""
monkeypatch.setattr(litellm, "vertex_ai_use_response_json_schema", False)
body = _gemini_request_body(
"gemini-2.5-flash", {}, vertex_ai_use_response_json_schema=True
)
body = _gemini_request_body("gemini-2.5-flash", {}, request_override=True)
generation_config = body["generationConfig"]
assert "response_schema" not in generation_config

View file

@ -200,7 +200,9 @@ CLIENT_SCHEMA = {
(None, None, "gemini-2.5-flash", True),
(None, None, "gemini-1.5-pro", False),
(False, None, "gemini-2.5-flash", False),
(True, None, "gemini-1.5-pro", True),
(True, None, "gemini-flash-latest", True),
(True, None, "gemini-1.5-pro", False),
(None, True, "gemini-1.5-pro", False),
(False, True, "gemini-2.5-flash", True),
(True, False, "gemini-2.5-flash", False),
],
@ -208,7 +210,10 @@ CLIENT_SCHEMA = {
def test_should_use_response_json_schema_precedence(
monkeypatch, global_setting, request_override, model, expected
):
"""Per request override beats litellm.vertex_ai_use_response_json_schema, which beats the model heuristic"""
"""
Per request override beats litellm.vertex_ai_use_response_json_schema, which beats the model
heuristic, and neither can put a schema on a channel Gemini 1.x has no field for
"""
monkeypatch.setattr(litellm, "vertex_ai_use_response_json_schema", global_setting)
assert should_use_response_json_schema(model, request_override) is expected