diff --git a/litellm/google_genai/adapters/transformation.py b/litellm/google_genai/adapters/transformation.py index 10340eb7acf..e415d817dcc 100644 --- a/litellm/google_genai/adapters/transformation.py +++ b/litellm/google_genai/adapters/transformation.py @@ -2,6 +2,7 @@ import json from collections.abc import AsyncIterator, Callable, Iterator, Mapping, Sequence from types import MappingProxyType from typing import Any, Final, TypeAlias, TypeVar, cast +from urllib.parse import urlparse from pydantic import JsonValue, TypeAdapter, ValidationError from typing_extensions import ReadOnly, TypedDict @@ -12,10 +13,12 @@ from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider from litellm.litellm_core_utils.get_supported_openai_params import get_supported_openai_params from litellm.litellm_core_utils.json_validation_rule import normalize_json_schema_types, normalize_tool_schema from litellm.litellm_core_utils.prompt_templates.common_utils import filter_value_from_dict +from litellm.llms.vertex_ai.gemini.transformation import GEMINI_FILES_API_URI_PREFIX from litellm.types.llms.openai import ( AllMessageValues, ChatCompletionAssistantMessage, ChatCompletionAssistantToolCall, + ChatCompletionFileObject, ChatCompletionImageObject, ChatCompletionSystemMessage, ChatCompletionTextObject, @@ -24,6 +27,7 @@ from litellm.types.llms.openai import ( ChatCompletionToolMessage, ChatCompletionToolParam, ChatCompletionUserMessage, + ChatCompletionVideoObject, ) from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import ( @@ -73,6 +77,8 @@ class _GenAIRequestFunctionCall(TypedDict, total=False): class _GenAIContentPart(TypedDict, total=False): text: ReadOnly[str] inline_data: ReadOnly[Mapping[str, str]] + fileData: ReadOnly[object] + file_data: ReadOnly[object] functionResponse: ReadOnly[_GenAIFunctionResponse] functionCall: ReadOnly[_GenAIRequestFunctionCall] @@ -101,6 +107,65 @@ class _GenAISystemInstruction(TypedDict, total=False): _EMPTY_STR_MAPPING: Final[Mapping[str, str]] = MappingProxyType({}) +_YOUTUBE_HOSTS: Final = ("youtube.com/", "youtu.be/") +_FILE_DATA_FIELDS: Final = TypeAdapter(Mapping[str, str]) +_PDF_MIME_TYPE: Final = "application/pdf" +_UserContentPart: TypeAlias = ( + ChatCompletionTextObject | ChatCompletionImageObject | ChatCompletionVideoObject | ChatCompletionFileObject +) + + +def _is_fetchable_document_uri(uri: str, mime_type: str | None) -> bool: + if uri.startswith(GEMINI_FILES_API_URI_PREFIX): + return True + if not uri.startswith(("https://", "http://", "gs://")): + return False + if mime_type is not None: + return mime_type == _PDF_MIME_TYPE + return urlparse(uri).path.lower().endswith(".pdf") + + +def _file_data_to_content_part(file_data: object) -> _UserContentPart | None: + """Map a Gemini fileData part (a URI the model fetches itself) to the matching OpenAI content part + + Images go to image_url and videos (by mime type, or a YouTube link with no mime type) go to video_url, + which is what OpenRouter and other OpenAI-compatible providers accept for remote media. PDF URLs and + Gemini Files API URIs go to a file part with the URI as file_id, so downstream transforms can fetch it. + Anything else is rejected: other documents would be fetched and labelled as PDFs downstream, and opaque + IDs would be resolved as managed files without the proxy's ownership check + """ + fields: Final = _validated(_FILE_DATA_FIELDS, file_data) + if fields is None: + raise BadRequestError( + message=f"fileData must be an object of string fields (fileUri, mimeType), got {file_data!r}", + model=None, + llm_provider="google_genai", + ) + uri: Final = fields.get("fileUri") or fields.get("file_uri") + if not uri: + return None + mime_type: Final = fields.get("mimeType") or fields.get("mime_type") + if mime_type is not None and mime_type.startswith("image/"): + return ChatCompletionImageObject(type="image_url", image_url={"url": uri}) + if (mime_type is not None and mime_type.startswith("video/")) or ( + mime_type is None and any(host in uri for host in _YOUTUBE_HOSTS) + ): + return ChatCompletionVideoObject(type="video_url", video_url={"url": uri}) + if not _is_fetchable_document_uri(uri, mime_type): + raise BadRequestError( + message=( + "fileData on this model supports image, video and YouTube URIs, PDF URLs, and Gemini Files API " + f"URIs. Got fileUri={uri!r} mimeType={mime_type!r}" + ), + model=None, + llm_provider="google_genai", + ) + return ChatCompletionFileObject( + type="file", + file={"file_id": uri, "format": mime_type} if mime_type is not None else {"file_id": uri}, + ) + + _RESPONSE_MIME_TYPE_KEYS: Final = ("responseMimeType", "response_mime_type") _RESPONSE_SCHEMA_KEYS: Final = ("responseJsonSchema", "response_json_schema", "responseSchema", "response_schema") _TOOL_PARAMETERS_KEYS: Final = ("parametersJsonSchema", "parameters") @@ -488,7 +553,7 @@ class GoogleGenAIAdapter: if role == "user": # Handle user messages with potential function responses - content_parts: list[ChatCompletionTextObject | ChatCompletionImageObject] = [] + content_parts: list[_UserContentPart] = [] tool_messages: list[ChatCompletionToolMessage] = [] for part in parts: @@ -514,6 +579,10 @@ class GoogleGenAIAdapter: }, ) ) + elif "fileData" in part or "file_data" in part: + file_part = _file_data_to_content_part(part.get("fileData") or part.get("file_data") or {}) + if file_part is not None: + content_parts.append(file_part) elif "functionResponse" in part: # Transform function response to tool message func_response = part["functionResponse"] diff --git a/tests/unit/google_genai/test_google_genai_adapter.py b/tests/unit/google_genai/test_google_genai_adapter.py index 761ab7bac89..a0c31a58f38 100644 --- a/tests/unit/google_genai/test_google_genai_adapter.py +++ b/tests/unit/google_genai/test_google_genai_adapter.py @@ -1641,3 +1641,137 @@ async def test_generate_content_sends_response_schema_and_tool_parameters_to_the } assert response["candidates"][0]["content"]["parts"] == [{"text": '{"park_name": "EPCOT"}'}] assert "text" not in response + + +@pytest.mark.parametrize( + "part, expected", + [ + pytest.param( + {"fileData": {"fileUri": "https://www.youtube.com/watch?v=abc123"}}, + {"type": "video_url", "video_url": {"url": "https://www.youtube.com/watch?v=abc123"}}, + id="youtube-link-without-mime-type", + ), + pytest.param( + {"fileData": {"fileUri": "https://youtu.be/abc123"}}, + {"type": "video_url", "video_url": {"url": "https://youtu.be/abc123"}}, + id="short-youtube-link", + ), + pytest.param( + {"file_data": {"file_uri": "gs://bucket/clip.mp4", "mime_type": "video/mp4"}}, + {"type": "video_url", "video_url": {"url": "gs://bucket/clip.mp4"}}, + id="snake-case-video-uri", + ), + pytest.param( + {"fileData": {"fileUri": "https://example.com/cat.png", "mimeType": "image/png"}}, + {"type": "image_url", "image_url": {"url": "https://example.com/cat.png"}}, + id="image-uri", + ), + pytest.param( + {"fileData": {"fileUri": "https://example.com/report.pdf", "mimeType": "application/pdf"}}, + {"type": "file", "file": {"file_id": "https://example.com/report.pdf", "format": "application/pdf"}}, + id="pdf-uri-keeps-mime-type", + ), + pytest.param( + {"fileData": {"fileUri": "https://example.com/docs/Report.PDF?v=2"}}, + {"type": "file", "file": {"file_id": "https://example.com/docs/Report.PDF?v=2"}}, + id="pdf-url-without-mime-type", + ), + pytest.param( + {"fileData": {"fileUri": "https://generativelanguage.googleapis.com/v1beta/files/abc"}}, + {"type": "file", "file": {"file_id": "https://generativelanguage.googleapis.com/v1beta/files/abc"}}, + id="files-api-uri-without-mime-type", + ), + ], +) +def test_file_data_part_is_forwarded_instead_of_dropped(part, expected): + """A Gemini fileData part must reach the completion request next to the prompt text + + Before this was handled, the adapter silently dropped fileData, so models routed through the + completion adapter (e.g. OpenRouter) answered the prompt without ever seeing the video or file + """ + from litellm.google_genai.adapters.transformation import GoogleGenAIAdapter + + completion_request = GoogleGenAIAdapter().translate_generate_content_to_completion( + model="openrouter/google/gemini-3.8-flash", + contents=[{"role": "user", "parts": [part, {"text": "Summarize this"}]}], + ) + + assert completion_request["messages"] == [ + {"role": "user", "content": [expected, {"type": "text", "text": "Summarize this"}]} + ] + + +def test_file_data_part_without_uri_is_skipped(): + from litellm.google_genai.adapters.transformation import GoogleGenAIAdapter + + completion_request = GoogleGenAIAdapter().translate_generate_content_to_completion( + model="openrouter/google/gemini-3.8-flash", + contents=[{"role": "user", "parts": [{"fileData": {"mimeType": "video/mp4"}}, {"text": "hi"}]}], + ) + + assert completion_request["messages"] == [{"role": "user", "content": "hi"}] + + +def test_pdf_file_data_reaches_openai_transform_as_a_fetchable_url(): + """The OpenAI chat transform only downloads and base64-encodes PDF URLs passed as file_id""" + from litellm.google_genai.adapters.transformation import GoogleGenAIAdapter + from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig + + completion_request = GoogleGenAIAdapter().translate_generate_content_to_completion( + model="openrouter/google/gemini-3.8-flash", + contents=[ + { + "role": "user", + "parts": [{"fileData": {"fileUri": "https://example.com/report.pdf", "mimeType": "application/pdf"}}], + } + ], + ) + + file_part = completion_request["messages"][0]["content"][0] + assert OpenAIGPTConfig().contains_pdf_url(file_part["file"]) + + +@pytest.mark.parametrize( + "file_data", + [ + pytest.param("https://www.youtube.com/watch?v=abc123", id="string-instead-of-object"), + pytest.param({"fileUri": "https://example.com/a.pdf", "mimeType": 7}, id="non-string-mime-type"), + ], +) +def test_malformed_file_data_is_rejected_as_bad_request(file_data): + from litellm.exceptions import BadRequestError + from litellm.google_genai.adapters.transformation import GoogleGenAIAdapter + + with pytest.raises(BadRequestError, match="fileData must be an object"): + GoogleGenAIAdapter().translate_generate_content_to_completion( + model="openrouter/google/gemini-3.8-flash", + contents=[{"role": "user", "parts": [{"fileData": file_data}, {"text": "hi"}]}], + ) + + +@pytest.mark.parametrize( + "file_data", + [ + pytest.param( + {"fileUri": "https://example.com/notes.docx", "mimeType": "application/msword"}, + id="non-pdf-document-would-be-labelled-pdf", + ), + pytest.param({"fileUri": "https://example.com/notes.txt"}, id="non-pdf-url-without-mime-type"), + pytest.param( + {"fileUri": "bGl0ZWxsbV9wcm94eTphcHBsaWNhdGlvbi9wZGY7dW5pZmllZF9pZCxhYmM", "mimeType": "application/pdf"}, + id="opaque-managed-file-id", + ), + pytest.param({"fileUri": "file-abc123"}, id="opaque-provider-file-id"), + ], +) +def test_file_data_that_downstream_would_mishandle_is_rejected(file_data): + """Non-PDF documents would be downloaded and renamed my_file.pdf downstream, and opaque IDs would be + resolved as managed files without the proxy's ownership check, so neither may become a file_id""" + from litellm.exceptions import BadRequestError + from litellm.google_genai.adapters.transformation import GoogleGenAIAdapter + + with pytest.raises(BadRequestError, match="fileData on this model supports"): + GoogleGenAIAdapter().translate_generate_content_to_completion( + model="openrouter/google/gemini-3.8-flash", + contents=[{"role": "user", "parts": [{"fileData": file_data}, {"text": "hi"}]}], + )