mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
fix(google_genai): only forward PDF and Gemini Files URIs as file_id
Non-PDF documents were downloaded and labelled my_file.pdf by the OpenAI transform, and opaque IDs were resolved as managed files without the proxy's ownership check on generateContent. Reject both with a 400 and reuse the shared Gemini Files API prefix constant
This commit is contained in:
parent
be0f9edf74
commit
6b19f17d99
2 changed files with 59 additions and 2 deletions
|
|
@ -2,6 +2,7 @@ import json
|
|||
from collections.abc import AsyncIterator, Callable, Iterator, Mapping, Sequence
|
||||
from types import MappingProxyType
|
||||
from typing import Any, Final, TypeAlias, TypeVar, cast
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from pydantic import JsonValue, TypeAdapter, ValidationError
|
||||
from typing_extensions import ReadOnly, TypedDict
|
||||
|
|
@ -12,6 +13,7 @@ from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
|||
from litellm.litellm_core_utils.get_supported_openai_params import get_supported_openai_params
|
||||
from litellm.litellm_core_utils.json_validation_rule import normalize_json_schema_types, normalize_tool_schema
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import filter_value_from_dict
|
||||
from litellm.llms.vertex_ai.gemini.transformation import GEMINI_FILES_API_URI_PREFIX
|
||||
from litellm.types.llms.openai import (
|
||||
AllMessageValues,
|
||||
ChatCompletionAssistantMessage,
|
||||
|
|
@ -107,17 +109,30 @@ class _GenAISystemInstruction(TypedDict, total=False):
|
|||
_EMPTY_STR_MAPPING: Final[Mapping[str, str]] = MappingProxyType({})
|
||||
_YOUTUBE_HOSTS: Final = ("youtube.com/", "youtu.be/")
|
||||
_FILE_DATA_FIELDS: Final = TypeAdapter(Mapping[str, str])
|
||||
_PDF_MIME_TYPE: Final = "application/pdf"
|
||||
_UserContentPart: TypeAlias = (
|
||||
ChatCompletionTextObject | ChatCompletionImageObject | ChatCompletionVideoObject | ChatCompletionFileObject
|
||||
)
|
||||
|
||||
|
||||
def _is_fetchable_document_uri(uri: str, mime_type: str | None) -> bool:
|
||||
if uri.startswith(GEMINI_FILES_API_URI_PREFIX):
|
||||
return True
|
||||
if not uri.startswith(("https://", "http://", "gs://")):
|
||||
return False
|
||||
if mime_type is not None:
|
||||
return mime_type == _PDF_MIME_TYPE
|
||||
return urlparse(uri).path.lower().endswith(".pdf")
|
||||
|
||||
|
||||
def _file_data_to_content_part(file_data: object) -> _UserContentPart | None:
|
||||
"""Map a Gemini fileData part (a URI the model fetches itself) to the matching OpenAI content part
|
||||
|
||||
Images go to image_url and videos (by mime type, or a YouTube link with no mime type) go to video_url,
|
||||
which is what OpenRouter and other OpenAI-compatible providers accept for remote media. Anything else
|
||||
goes to a file part with the URI as file_id, so downstream transforms can fetch it
|
||||
which is what OpenRouter and other OpenAI-compatible providers accept for remote media. PDF URLs and
|
||||
Gemini Files API URIs go to a file part with the URI as file_id, so downstream transforms can fetch it.
|
||||
Anything else is rejected: other documents would be fetched and labelled as PDFs downstream, and opaque
|
||||
IDs would be resolved as managed files without the proxy's ownership check
|
||||
"""
|
||||
fields: Final = _validated(_FILE_DATA_FIELDS, file_data)
|
||||
if fields is None:
|
||||
|
|
@ -136,6 +151,15 @@ def _file_data_to_content_part(file_data: object) -> _UserContentPart | None:
|
|||
mime_type is None and any(host in uri for host in _YOUTUBE_HOSTS)
|
||||
):
|
||||
return ChatCompletionVideoObject(type="video_url", video_url={"url": uri})
|
||||
if not _is_fetchable_document_uri(uri, mime_type):
|
||||
raise BadRequestError(
|
||||
message=(
|
||||
"fileData on this model supports image, video and YouTube URIs, PDF URLs, and Gemini Files API "
|
||||
f"URIs. Got fileUri={uri!r} mimeType={mime_type!r}"
|
||||
),
|
||||
model=None,
|
||||
llm_provider="google_genai",
|
||||
)
|
||||
return ChatCompletionFileObject(
|
||||
type="file",
|
||||
file={"file_id": uri, "format": mime_type} if mime_type is not None else {"file_id": uri},
|
||||
|
|
|
|||
|
|
@ -1671,6 +1671,11 @@ async def test_generate_content_sends_response_schema_and_tool_parameters_to_the
|
|||
{"type": "file", "file": {"file_id": "https://example.com/report.pdf", "format": "application/pdf"}},
|
||||
id="pdf-uri-keeps-mime-type",
|
||||
),
|
||||
pytest.param(
|
||||
{"fileData": {"fileUri": "https://example.com/docs/Report.PDF?v=2"}},
|
||||
{"type": "file", "file": {"file_id": "https://example.com/docs/Report.PDF?v=2"}},
|
||||
id="pdf-url-without-mime-type",
|
||||
),
|
||||
pytest.param(
|
||||
{"fileData": {"fileUri": "https://generativelanguage.googleapis.com/v1beta/files/abc"}},
|
||||
{"type": "file", "file": {"file_id": "https://generativelanguage.googleapis.com/v1beta/files/abc"}},
|
||||
|
|
@ -1742,3 +1747,31 @@ def test_malformed_file_data_is_rejected_as_bad_request(file_data):
|
|||
model="openrouter/google/gemini-3.8-flash",
|
||||
contents=[{"role": "user", "parts": [{"fileData": file_data}, {"text": "hi"}]}],
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"file_data",
|
||||
[
|
||||
pytest.param(
|
||||
{"fileUri": "https://example.com/notes.docx", "mimeType": "application/msword"},
|
||||
id="non-pdf-document-would-be-labelled-pdf",
|
||||
),
|
||||
pytest.param({"fileUri": "https://example.com/notes.txt"}, id="non-pdf-url-without-mime-type"),
|
||||
pytest.param(
|
||||
{"fileUri": "bGl0ZWxsbV9wcm94eTphcHBsaWNhdGlvbi9wZGY7dW5pZmllZF9pZCxhYmM", "mimeType": "application/pdf"},
|
||||
id="opaque-managed-file-id",
|
||||
),
|
||||
pytest.param({"fileUri": "file-abc123"}, id="opaque-provider-file-id"),
|
||||
],
|
||||
)
|
||||
def test_file_data_that_downstream_would_mishandle_is_rejected(file_data):
|
||||
"""Non-PDF documents would be downloaded and renamed my_file.pdf downstream, and opaque IDs would be
|
||||
resolved as managed files without the proxy's ownership check, so neither may become a file_id"""
|
||||
from litellm.exceptions import BadRequestError
|
||||
from litellm.google_genai.adapters.transformation import GoogleGenAIAdapter
|
||||
|
||||
with pytest.raises(BadRequestError, match="fileData on this model supports"):
|
||||
GoogleGenAIAdapter().translate_generate_content_to_completion(
|
||||
model="openrouter/google/gemini-3.8-flash",
|
||||
contents=[{"role": "user", "parts": [{"fileData": file_data}, {"text": "hi"}]}],
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue