diff --git a/litellm/llms/vertex_ai/gemini/transformation.py b/litellm/llms/vertex_ai/gemini/transformation.py index e4fcd35b954..04fde04f2c1 100644 --- a/litellm/llms/vertex_ai/gemini/transformation.py +++ b/litellm/llms/vertex_ai/gemini/transformation.py @@ -28,7 +28,6 @@ from litellm.types.files import ( get_file_type_from_extension, is_gemini_1_5_accepted_file_type, ) -from litellm.types.utils import LlmProviders from litellm.types.llms.openai import ( AllMessageValues, ChatCompletionAssistantMessage, @@ -48,7 +47,7 @@ from litellm.types.llms.vertex_ai import ( ToolConfig, Tools, ) -from litellm.types.utils import GenericImageParsingChunk +from litellm.types.utils import GenericImageParsingChunk, LlmProviders from ..common_utils import ( _check_text_in_content, @@ -82,6 +81,7 @@ def _process_gemini_image( image_url: str, format: Optional[str] = None, media_resolution: Optional[Literal["low", "medium", "high"]] = None, + model: Optional[str] = None, ) -> PartType: """ Given an image URL, return the appropriate PartType for Gemini @@ -118,16 +118,19 @@ def _process_gemini_image( # https links for unsupported mime types and base64 images image = convert_to_anthropic_image_obj(image_url, format=format) _blob: BlobType = {"data": image["data"], "mime_type": image["media_type"]} - if media_resolution is not None: - _blob["media_resolution"] = media_resolution + # media_resolution on individual Part objects is exclusive to Gemini 3 models + if media_resolution is not None and model is not None: + from .vertex_and_google_ai_studio_gemini import VertexGeminiConfig + if VertexGeminiConfig._is_gemini_3_or_newer(model): + _blob["media_resolution"] = media_resolution # Convert snake_case keys to camelCase for JSON serialization # The TypedDict uses snake_case, but the API expects camelCase _blob_dict = dict(_blob) if "media_resolution" in _blob_dict: - _blob_dict["mediaResolution"] = _blob_dict.pop("media_resolution") + _blob_dict["media_resolution"] = _blob_dict.pop("media_resolution") if "mime_type" in _blob_dict: - _blob_dict["mimeType"] = _blob_dict.pop("mime_type") + _blob_dict["mime_type"] = _blob_dict.pop("mime_type") return PartType(inline_data=cast(BlobType, _blob_dict)) raise Exception("Invalid image received - {}".format(image_url)) @@ -247,6 +250,7 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915 image_url=image_url, format=format, media_resolution=media_resolution, + model=model, ) _parts.append(_part) elif element["type"] == "input_audio": @@ -271,6 +275,7 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915 _part = _process_gemini_image( image_url=openai_image_str, format=audio_format_modified, + model=model, ) _parts.append(_part) elif element["type"] == "file": @@ -287,6 +292,7 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915 _part = _process_gemini_image( image_url=passed_file, format=format, + model=model, ) _parts.append(_part) except Exception: diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py index 2b305dbade1..6bd0fb52f1d 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py @@ -1795,7 +1795,9 @@ def test_media_resolution_from_detail_parameter(): } ] - contents = _gemini_convert_messages_with_history(messages=messages) + contents = _gemini_convert_messages_with_history( + messages=messages, model="gemini-3-pro-preview" + ) # Verify media_resolution is set in the inline_data # Note: Gemini adds a blank text part when there's no text, so we expect 2 parts @@ -1837,7 +1839,9 @@ def test_media_resolution_low_detail(): } ] - contents = _gemini_convert_messages_with_history(messages=messages) + contents = _gemini_convert_messages_with_history( + messages=messages, model="gemini-3-pro-preview" + ) # Find the part with inline_data image_part = None @@ -1951,7 +1955,9 @@ def test_media_resolution_per_part(): } ] - contents = _gemini_convert_messages_with_history(messages=messages) + contents = _gemini_convert_messages_with_history( + messages=messages, model="gemini-3-pro-preview" + ) # Should have one content with multiple parts assert len(contents) == 1 @@ -1968,6 +1974,41 @@ def test_media_resolution_per_part(): assert image2_part["inline_data"]["mediaResolution"] == "high" +def test_media_resolution_only_for_gemini_3_models(): + """Ensure mediaResolution is not added for non-Gemini 3 models.""" + from litellm.llms.vertex_ai.gemini.transformation import ( + _gemini_convert_messages_with_history, + ) + + base64_image = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg==" + messages = [ + { + "role": "user", + "content": [ + { + "type": "image_url", + "image_url": { + "url": base64_image, + "detail": "high", + }, + } + ], + } + ] + + contents = _gemini_convert_messages_with_history( + messages=messages, model="gemini-2.5-pro" + ) + image_part = None + for part in contents[0]["parts"]: + if "inline_data" in part: + image_part = part + break + assert image_part is not None + assert "inline_data" in image_part + assert "mediaResolution" not in image_part["inline_data"] + + def test_gemini_3_image_models_no_thinking_config(): """ Test that Gemini 3 image models do NOT receive automatic thinkingConfig.