mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
Make sure that media resolution is only for gemini 3 model
This commit is contained in:
parent
7227747a6f
commit
e49f21c918
2 changed files with 56 additions and 9 deletions
|
|
@ -28,7 +28,6 @@ from litellm.types.files import (
|
|||
get_file_type_from_extension,
|
||||
is_gemini_1_5_accepted_file_type,
|
||||
)
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.types.llms.openai import (
|
||||
AllMessageValues,
|
||||
ChatCompletionAssistantMessage,
|
||||
|
|
@ -48,7 +47,7 @@ from litellm.types.llms.vertex_ai import (
|
|||
ToolConfig,
|
||||
Tools,
|
||||
)
|
||||
from litellm.types.utils import GenericImageParsingChunk
|
||||
from litellm.types.utils import GenericImageParsingChunk, LlmProviders
|
||||
|
||||
from ..common_utils import (
|
||||
_check_text_in_content,
|
||||
|
|
@ -82,6 +81,7 @@ def _process_gemini_image(
|
|||
image_url: str,
|
||||
format: Optional[str] = None,
|
||||
media_resolution: Optional[Literal["low", "medium", "high"]] = None,
|
||||
model: Optional[str] = None,
|
||||
) -> PartType:
|
||||
"""
|
||||
Given an image URL, return the appropriate PartType for Gemini
|
||||
|
|
@ -118,16 +118,19 @@ def _process_gemini_image(
|
|||
# https links for unsupported mime types and base64 images
|
||||
image = convert_to_anthropic_image_obj(image_url, format=format)
|
||||
_blob: BlobType = {"data": image["data"], "mime_type": image["media_type"]}
|
||||
if media_resolution is not None:
|
||||
_blob["media_resolution"] = media_resolution
|
||||
# media_resolution on individual Part objects is exclusive to Gemini 3 models
|
||||
if media_resolution is not None and model is not None:
|
||||
from .vertex_and_google_ai_studio_gemini import VertexGeminiConfig
|
||||
if VertexGeminiConfig._is_gemini_3_or_newer(model):
|
||||
_blob["media_resolution"] = media_resolution
|
||||
|
||||
# Convert snake_case keys to camelCase for JSON serialization
|
||||
# The TypedDict uses snake_case, but the API expects camelCase
|
||||
_blob_dict = dict(_blob)
|
||||
if "media_resolution" in _blob_dict:
|
||||
_blob_dict["mediaResolution"] = _blob_dict.pop("media_resolution")
|
||||
_blob_dict["media_resolution"] = _blob_dict.pop("media_resolution")
|
||||
if "mime_type" in _blob_dict:
|
||||
_blob_dict["mimeType"] = _blob_dict.pop("mime_type")
|
||||
_blob_dict["mime_type"] = _blob_dict.pop("mime_type")
|
||||
|
||||
return PartType(inline_data=cast(BlobType, _blob_dict))
|
||||
raise Exception("Invalid image received - {}".format(image_url))
|
||||
|
|
@ -247,6 +250,7 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
|
|||
image_url=image_url,
|
||||
format=format,
|
||||
media_resolution=media_resolution,
|
||||
model=model,
|
||||
)
|
||||
_parts.append(_part)
|
||||
elif element["type"] == "input_audio":
|
||||
|
|
@ -271,6 +275,7 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
|
|||
_part = _process_gemini_image(
|
||||
image_url=openai_image_str,
|
||||
format=audio_format_modified,
|
||||
model=model,
|
||||
)
|
||||
_parts.append(_part)
|
||||
elif element["type"] == "file":
|
||||
|
|
@ -287,6 +292,7 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
|
|||
_part = _process_gemini_image(
|
||||
image_url=passed_file,
|
||||
format=format,
|
||||
model=model,
|
||||
)
|
||||
_parts.append(_part)
|
||||
except Exception:
|
||||
|
|
|
|||
|
|
@ -1795,7 +1795,9 @@ def test_media_resolution_from_detail_parameter():
|
|||
}
|
||||
]
|
||||
|
||||
contents = _gemini_convert_messages_with_history(messages=messages)
|
||||
contents = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-3-pro-preview"
|
||||
)
|
||||
|
||||
# Verify media_resolution is set in the inline_data
|
||||
# Note: Gemini adds a blank text part when there's no text, so we expect 2 parts
|
||||
|
|
@ -1837,7 +1839,9 @@ def test_media_resolution_low_detail():
|
|||
}
|
||||
]
|
||||
|
||||
contents = _gemini_convert_messages_with_history(messages=messages)
|
||||
contents = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-3-pro-preview"
|
||||
)
|
||||
|
||||
# Find the part with inline_data
|
||||
image_part = None
|
||||
|
|
@ -1951,7 +1955,9 @@ def test_media_resolution_per_part():
|
|||
}
|
||||
]
|
||||
|
||||
contents = _gemini_convert_messages_with_history(messages=messages)
|
||||
contents = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-3-pro-preview"
|
||||
)
|
||||
|
||||
# Should have one content with multiple parts
|
||||
assert len(contents) == 1
|
||||
|
|
@ -1968,6 +1974,41 @@ def test_media_resolution_per_part():
|
|||
assert image2_part["inline_data"]["mediaResolution"] == "high"
|
||||
|
||||
|
||||
def test_media_resolution_only_for_gemini_3_models():
|
||||
"""Ensure mediaResolution is not added for non-Gemini 3 models."""
|
||||
from litellm.llms.vertex_ai.gemini.transformation import (
|
||||
_gemini_convert_messages_with_history,
|
||||
)
|
||||
|
||||
base64_image = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg=="
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": base64_image,
|
||||
"detail": "high",
|
||||
},
|
||||
}
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
contents = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-2.5-pro"
|
||||
)
|
||||
image_part = None
|
||||
for part in contents[0]["parts"]:
|
||||
if "inline_data" in part:
|
||||
image_part = part
|
||||
break
|
||||
assert image_part is not None
|
||||
assert "inline_data" in image_part
|
||||
assert "mediaResolution" not in image_part["inline_data"]
|
||||
|
||||
|
||||
def test_gemini_3_image_models_no_thinking_config():
|
||||
"""
|
||||
Test that Gemini 3 image models do NOT receive automatic thinkingConfig.
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue