diff --git a/litellm/google_genai/adapters/transformation.py b/litellm/google_genai/adapters/transformation.py index 10340eb7acf..d0651e32606 100644 --- a/litellm/google_genai/adapters/transformation.py +++ b/litellm/google_genai/adapters/transformation.py @@ -16,6 +16,7 @@ from litellm.types.llms.openai import ( AllMessageValues, ChatCompletionAssistantMessage, ChatCompletionAssistantToolCall, + ChatCompletionFileObject, ChatCompletionImageObject, ChatCompletionSystemMessage, ChatCompletionTextObject, @@ -24,6 +25,7 @@ from litellm.types.llms.openai import ( ChatCompletionToolMessage, ChatCompletionToolParam, ChatCompletionUserMessage, + ChatCompletionVideoObject, ) from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import ( @@ -73,6 +75,8 @@ class _GenAIRequestFunctionCall(TypedDict, total=False): class _GenAIContentPart(TypedDict, total=False): text: ReadOnly[str] inline_data: ReadOnly[Mapping[str, str]] + fileData: ReadOnly[Mapping[str, str]] + file_data: ReadOnly[Mapping[str, str]] functionResponse: ReadOnly[_GenAIFunctionResponse] functionCall: ReadOnly[_GenAIRequestFunctionCall] @@ -101,6 +105,35 @@ class _GenAISystemInstruction(TypedDict, total=False): _EMPTY_STR_MAPPING: Final[Mapping[str, str]] = MappingProxyType({}) +_YOUTUBE_HOSTS: Final = ("youtube.com/", "youtu.be/") +_UserContentPart: TypeAlias = ( + ChatCompletionTextObject | ChatCompletionImageObject | ChatCompletionVideoObject | ChatCompletionFileObject +) + + +def _file_data_to_content_part(file_data: Mapping[str, str]) -> _UserContentPart | None: + """Map a Gemini fileData part (a URI the model fetches itself) to the matching OpenAI content part + + Images go to image_url and videos (by mime type, or a YouTube link with no mime type) go to video_url, + which is what OpenRouter and other OpenAI-compatible providers accept for remote media. Anything else + goes to a file part that keeps the URI and mime type + """ + uri: Final = file_data.get("fileUri") or file_data.get("file_uri") + if not uri: + return None + mime_type: Final = file_data.get("mimeType") or file_data.get("mime_type") + if mime_type is not None and mime_type.startswith("image/"): + return ChatCompletionImageObject(type="image_url", image_url={"url": uri}) + if (mime_type is not None and mime_type.startswith("video/")) or ( + mime_type is None and any(host in uri for host in _YOUTUBE_HOSTS) + ): + return ChatCompletionVideoObject(type="video_url", video_url={"url": uri}) + file_part: Final = ChatCompletionFileObject(type="file", file={"file_data": uri}) + if mime_type is not None: + file_part["file"]["format"] = mime_type + return file_part + + _RESPONSE_MIME_TYPE_KEYS: Final = ("responseMimeType", "response_mime_type") _RESPONSE_SCHEMA_KEYS: Final = ("responseJsonSchema", "response_json_schema", "responseSchema", "response_schema") _TOOL_PARAMETERS_KEYS: Final = ("parametersJsonSchema", "parameters") @@ -488,7 +521,7 @@ class GoogleGenAIAdapter: if role == "user": # Handle user messages with potential function responses - content_parts: list[ChatCompletionTextObject | ChatCompletionImageObject] = [] + content_parts: list[_UserContentPart] = [] tool_messages: list[ChatCompletionToolMessage] = [] for part in parts: @@ -514,6 +547,11 @@ class GoogleGenAIAdapter: }, ) ) + elif "fileData" in part or "file_data" in part: + # Handle URI references (YouTube links, Files API URIs, gs:// or https:// media) + file_part = _file_data_to_content_part(part.get("fileData") or part.get("file_data") or {}) + if file_part is not None: + content_parts.append(file_part) elif "functionResponse" in part: # Transform function response to tool message func_response = part["functionResponse"] diff --git a/tests/unit/google_genai/test_google_genai_adapter.py b/tests/unit/google_genai/test_google_genai_adapter.py index 761ab7bac89..19375acd315 100644 --- a/tests/unit/google_genai/test_google_genai_adapter.py +++ b/tests/unit/google_genai/test_google_genai_adapter.py @@ -1641,3 +1641,67 @@ async def test_generate_content_sends_response_schema_and_tool_parameters_to_the } assert response["candidates"][0]["content"]["parts"] == [{"text": '{"park_name": "EPCOT"}'}] assert "text" not in response + + +@pytest.mark.parametrize( + "part, expected", + [ + pytest.param( + {"fileData": {"fileUri": "https://www.youtube.com/watch?v=abc123"}}, + {"type": "video_url", "video_url": {"url": "https://www.youtube.com/watch?v=abc123"}}, + id="youtube-link-without-mime-type", + ), + pytest.param( + {"fileData": {"fileUri": "https://youtu.be/abc123"}}, + {"type": "video_url", "video_url": {"url": "https://youtu.be/abc123"}}, + id="short-youtube-link", + ), + pytest.param( + {"file_data": {"file_uri": "gs://bucket/clip.mp4", "mime_type": "video/mp4"}}, + {"type": "video_url", "video_url": {"url": "gs://bucket/clip.mp4"}}, + id="snake-case-video-uri", + ), + pytest.param( + {"fileData": {"fileUri": "https://example.com/cat.png", "mimeType": "image/png"}}, + {"type": "image_url", "image_url": {"url": "https://example.com/cat.png"}}, + id="image-uri", + ), + pytest.param( + {"fileData": {"fileUri": "https://example.com/report.pdf", "mimeType": "application/pdf"}}, + {"type": "file", "file": {"file_data": "https://example.com/report.pdf", "format": "application/pdf"}}, + id="pdf-uri-keeps-mime-type", + ), + pytest.param( + {"fileData": {"fileUri": "https://generativelanguage.googleapis.com/v1beta/files/abc"}}, + {"type": "file", "file": {"file_data": "https://generativelanguage.googleapis.com/v1beta/files/abc"}}, + id="files-api-uri-without-mime-type", + ), + ], +) +def test_file_data_part_is_forwarded_instead_of_dropped(part, expected): + """A Gemini fileData part must reach the completion request next to the prompt text + + Before this was handled, the adapter silently dropped fileData, so models routed through the + completion adapter (e.g. OpenRouter) answered the prompt without ever seeing the video or file + """ + from litellm.google_genai.adapters.transformation import GoogleGenAIAdapter + + completion_request = GoogleGenAIAdapter().translate_generate_content_to_completion( + model="openrouter/google/gemini-3.8-flash", + contents=[{"role": "user", "parts": [part, {"text": "Summarize this"}]}], + ) + + assert completion_request["messages"] == [ + {"role": "user", "content": [expected, {"type": "text", "text": "Summarize this"}]} + ] + + +def test_file_data_part_without_uri_is_skipped(): + from litellm.google_genai.adapters.transformation import GoogleGenAIAdapter + + completion_request = GoogleGenAIAdapter().translate_generate_content_to_completion( + model="openrouter/google/gemini-3.8-flash", + contents=[{"role": "user", "parts": [{"fileData": {"mimeType": "video/mp4"}}, {"text": "hi"}]}], + ) + + assert completion_request["messages"] == [{"role": "user", "content": "hi"}]