Make sure that media resolution is only for gemini 3 model

This commit is contained in:
Sameer Kankute 2025-11-26 18:57:57 +05:30
parent 7227747a6f
commit e49f21c918
2 changed files with 56 additions and 9 deletions

View file

@ -28,7 +28,6 @@ from litellm.types.files import (
get_file_type_from_extension,
is_gemini_1_5_accepted_file_type,
)
from litellm.types.utils import LlmProviders
from litellm.types.llms.openai import (
AllMessageValues,
ChatCompletionAssistantMessage,
@ -48,7 +47,7 @@ from litellm.types.llms.vertex_ai import (
ToolConfig,
Tools,
)
from litellm.types.utils import GenericImageParsingChunk
from litellm.types.utils import GenericImageParsingChunk, LlmProviders
from ..common_utils import (
_check_text_in_content,
@ -82,6 +81,7 @@ def _process_gemini_image(
image_url: str,
format: Optional[str] = None,
media_resolution: Optional[Literal["low", "medium", "high"]] = None,
model: Optional[str] = None,
) -> PartType:
"""
Given an image URL, return the appropriate PartType for Gemini
@ -118,16 +118,19 @@ def _process_gemini_image(
# https links for unsupported mime types and base64 images
image = convert_to_anthropic_image_obj(image_url, format=format)
_blob: BlobType = {"data": image["data"], "mime_type": image["media_type"]}
if media_resolution is not None:
_blob["media_resolution"] = media_resolution
# media_resolution on individual Part objects is exclusive to Gemini 3 models
if media_resolution is not None and model is not None:
from .vertex_and_google_ai_studio_gemini import VertexGeminiConfig
if VertexGeminiConfig._is_gemini_3_or_newer(model):
_blob["media_resolution"] = media_resolution
# Convert snake_case keys to camelCase for JSON serialization
# The TypedDict uses snake_case, but the API expects camelCase
_blob_dict = dict(_blob)
if "media_resolution" in _blob_dict:
_blob_dict["mediaResolution"] = _blob_dict.pop("media_resolution")
_blob_dict["media_resolution"] = _blob_dict.pop("media_resolution")
if "mime_type" in _blob_dict:
_blob_dict["mimeType"] = _blob_dict.pop("mime_type")
_blob_dict["mime_type"] = _blob_dict.pop("mime_type")
return PartType(inline_data=cast(BlobType, _blob_dict))
raise Exception("Invalid image received - {}".format(image_url))
@ -247,6 +250,7 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
image_url=image_url,
format=format,
media_resolution=media_resolution,
model=model,
)
_parts.append(_part)
elif element["type"] == "input_audio":
@ -271,6 +275,7 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
_part = _process_gemini_image(
image_url=openai_image_str,
format=audio_format_modified,
model=model,
)
_parts.append(_part)
elif element["type"] == "file":
@ -287,6 +292,7 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915
_part = _process_gemini_image(
image_url=passed_file,
format=format,
model=model,
)
_parts.append(_part)
except Exception:

View file

@ -1795,7 +1795,9 @@ def test_media_resolution_from_detail_parameter():
}
]
contents = _gemini_convert_messages_with_history(messages=messages)
contents = _gemini_convert_messages_with_history(
messages=messages, model="gemini-3-pro-preview"
)
# Verify media_resolution is set in the inline_data
# Note: Gemini adds a blank text part when there's no text, so we expect 2 parts
@ -1837,7 +1839,9 @@ def test_media_resolution_low_detail():
}
]
contents = _gemini_convert_messages_with_history(messages=messages)
contents = _gemini_convert_messages_with_history(
messages=messages, model="gemini-3-pro-preview"
)
# Find the part with inline_data
image_part = None
@ -1951,7 +1955,9 @@ def test_media_resolution_per_part():
}
]
contents = _gemini_convert_messages_with_history(messages=messages)
contents = _gemini_convert_messages_with_history(
messages=messages, model="gemini-3-pro-preview"
)
# Should have one content with multiple parts
assert len(contents) == 1
@ -1968,6 +1974,41 @@ def test_media_resolution_per_part():
assert image2_part["inline_data"]["mediaResolution"] == "high"
def test_media_resolution_only_for_gemini_3_models():
"""Ensure mediaResolution is not added for non-Gemini 3 models."""
from litellm.llms.vertex_ai.gemini.transformation import (
_gemini_convert_messages_with_history,
)
base64_image = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg=="
messages = [
{
"role": "user",
"content": [
{
"type": "image_url",
"image_url": {
"url": base64_image,
"detail": "high",
},
}
],
}
]
contents = _gemini_convert_messages_with_history(
messages=messages, model="gemini-2.5-pro"
)
image_part = None
for part in contents[0]["parts"]:
if "inline_data" in part:
image_part = part
break
assert image_part is not None
assert "inline_data" in image_part
assert "mediaResolution" not in image_part["inline_data"]
def test_gemini_3_image_models_no_thinking_config():
"""
Test that Gemini 3 image models do NOT receive automatic thinkingConfig.