mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(vertex_ai): keep Gemini 1.x off the responseJsonSchema channel
Neither the global setting nor the per request override can select a channel the model has no field for, so asking for the JSON Schema channel on a Gemini 1.x model now logs a warning and keeps the natively converted responseSchema.
This commit is contained in:
parent
83e3458f8a
commit
729cb0f244
3 changed files with 41 additions and 20 deletions
|
|
@ -271,10 +271,16 @@ def supports_response_json_schema(model: str) -> bool:
|
|||
return bool(gemini_2_plus_pattern.search(model_lower))
|
||||
|
||||
|
||||
GEMINI_1_MODEL_PATTERN: Final = re.compile(r"gemini-1(?:\.|-)")
|
||||
VERTEX_AI_USE_RESPONSE_JSON_SCHEMA_PARAM: Final = "vertex_ai_use_response_json_schema"
|
||||
VERTEX_AI_VERBATIM_RESPONSE_SCHEMA_PARAM: Final = "litellm_param_vertex_ai_verbatim_response_schema"
|
||||
|
||||
|
||||
def _rejects_response_json_schema(model: str) -> bool:
|
||||
"""Gemini 1.x generateContent has no responseJsonSchema field, so no override can reach it"""
|
||||
return bool(GEMINI_1_MODEL_PATTERN.search(model.lower()))
|
||||
|
||||
|
||||
def should_use_response_json_schema(model: str, request_override: bool | None = None) -> bool:
|
||||
"""
|
||||
Resolve which structured output channel a json_schema response_format goes to.
|
||||
|
|
@ -283,13 +289,20 @@ def should_use_response_json_schema(model: str, request_override: bool | None =
|
|||
natively converted ``responseSchema`` (nullable unions flattened, constraints
|
||||
hoisted, ``propertyOrdering`` added). Precedence: per request
|
||||
``vertex_ai_use_response_json_schema``, then
|
||||
``litellm.vertex_ai_use_response_json_schema``, then the model heuristic
|
||||
``litellm.vertex_ai_use_response_json_schema``, then the model heuristic. Neither
|
||||
override can select a channel the model has no field for
|
||||
"""
|
||||
if request_override is not None:
|
||||
return request_override
|
||||
if litellm.vertex_ai_use_response_json_schema is not None:
|
||||
return litellm.vertex_ai_use_response_json_schema
|
||||
return supports_response_json_schema(model)
|
||||
override: Final = request_override if request_override is not None else litellm.vertex_ai_use_response_json_schema
|
||||
if override is None:
|
||||
return supports_response_json_schema(model)
|
||||
if override and _rejects_response_json_schema(model):
|
||||
verbose_logger.warning(
|
||||
"vertex_ai_use_response_json_schema=True ignored for model=%s: it has no responseJsonSchema field, "
|
||||
"so the schema stays on responseSchema",
|
||||
model,
|
||||
)
|
||||
return False
|
||||
return override
|
||||
|
||||
|
||||
from typing import Literal
|
||||
|
|
|
|||
|
|
@ -1,5 +1,7 @@
|
|||
|
||||
import json
|
||||
from collections.abc import Mapping
|
||||
from typing import Optional
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -354,7 +356,14 @@ RESPONSE_SCHEMA_CHANNEL_CLIENT_SCHEMA = {
|
|||
}
|
||||
|
||||
|
||||
def _gemini_request_body(model: str, litellm_params: dict, **completion_kwargs) -> RequestBody:
|
||||
def _gemini_request_body(
|
||||
model: str,
|
||||
litellm_params: Mapping[str, bool],
|
||||
request_override: Optional[bool] = None,
|
||||
) -> RequestBody:
|
||||
override_kwargs: dict[str, bool] = (
|
||||
{} if request_override is None else {"vertex_ai_use_response_json_schema": request_override}
|
||||
)
|
||||
optional_params = litellm.utils.get_optional_params(
|
||||
model=model,
|
||||
custom_llm_provider="vertex_ai",
|
||||
|
|
@ -365,14 +374,14 @@ def _gemini_request_body(model: str, litellm_params: dict, **completion_kwargs)
|
|||
"schema": json.loads(json.dumps(RESPONSE_SCHEMA_CHANNEL_CLIENT_SCHEMA)),
|
||||
},
|
||||
},
|
||||
**completion_kwargs,
|
||||
**override_kwargs,
|
||||
)
|
||||
return transformation._transform_request_body(
|
||||
messages=[{"role": "user", "content": "extract it"}],
|
||||
model=model,
|
||||
optional_params=optional_params,
|
||||
custom_llm_provider="vertex_ai",
|
||||
litellm_params=litellm_params,
|
||||
litellm_params=dict(litellm_params),
|
||||
cached_content=None,
|
||||
)
|
||||
|
||||
|
|
@ -382,9 +391,7 @@ def test__transform_request_body_per_request_response_json_schema_opt_out():
|
|||
vertex_ai_use_response_json_schema=False on the request puts the schema on Vertex's native
|
||||
responseSchema channel, and the knob itself never reaches the provider body
|
||||
"""
|
||||
body = _gemini_request_body(
|
||||
"gemini-2.5-flash", {}, vertex_ai_use_response_json_schema=False
|
||||
)
|
||||
body = _gemini_request_body("gemini-2.5-flash", {}, request_override=False)
|
||||
|
||||
generation_config = body["generationConfig"]
|
||||
assert "response_json_schema" not in generation_config
|
||||
|
|
@ -398,9 +405,7 @@ def test__transform_request_body_per_request_response_json_schema_opt_out():
|
|||
|
||||
def test__transform_request_body_deployment_response_json_schema_opt_out():
|
||||
"""A deployment's litellm_params opts every request routed to it out of responseJsonSchema"""
|
||||
body = _gemini_request_body(
|
||||
"gemini-2.5-flash", {"vertex_ai_use_response_json_schema": False}
|
||||
)
|
||||
body = _gemini_request_body("gemini-2.5-flash", {"vertex_ai_use_response_json_schema": False})
|
||||
|
||||
generation_config = body["generationConfig"]
|
||||
assert "response_json_schema" not in generation_config
|
||||
|
|
@ -414,9 +419,7 @@ def test__transform_request_body_per_request_opt_in_beats_global_opt_out(monkeyp
|
|||
"""
|
||||
monkeypatch.setattr(litellm, "vertex_ai_use_response_json_schema", False)
|
||||
|
||||
body = _gemini_request_body(
|
||||
"gemini-2.5-flash", {}, vertex_ai_use_response_json_schema=True
|
||||
)
|
||||
body = _gemini_request_body("gemini-2.5-flash", {}, request_override=True)
|
||||
|
||||
generation_config = body["generationConfig"]
|
||||
assert "response_schema" not in generation_config
|
||||
|
|
|
|||
|
|
@ -200,7 +200,9 @@ CLIENT_SCHEMA = {
|
|||
(None, None, "gemini-2.5-flash", True),
|
||||
(None, None, "gemini-1.5-pro", False),
|
||||
(False, None, "gemini-2.5-flash", False),
|
||||
(True, None, "gemini-1.5-pro", True),
|
||||
(True, None, "gemini-flash-latest", True),
|
||||
(True, None, "gemini-1.5-pro", False),
|
||||
(None, True, "gemini-1.5-pro", False),
|
||||
(False, True, "gemini-2.5-flash", True),
|
||||
(True, False, "gemini-2.5-flash", False),
|
||||
],
|
||||
|
|
@ -208,7 +210,10 @@ CLIENT_SCHEMA = {
|
|||
def test_should_use_response_json_schema_precedence(
|
||||
monkeypatch, global_setting, request_override, model, expected
|
||||
):
|
||||
"""Per request override beats litellm.vertex_ai_use_response_json_schema, which beats the model heuristic"""
|
||||
"""
|
||||
Per request override beats litellm.vertex_ai_use_response_json_schema, which beats the model
|
||||
heuristic, and neither can put a schema on a channel Gemini 1.x has no field for
|
||||
"""
|
||||
monkeypatch.setattr(litellm, "vertex_ai_use_response_json_schema", global_setting)
|
||||
|
||||
assert should_use_response_json_schema(model, request_override) is expected
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue