diff --git a/litellm/litellm_core_utils/prompt_templates/common_utils.py b/litellm/litellm_core_utils/prompt_templates/common_utils.py index cec61405ebb..3ce9a6a4eea 100644 --- a/litellm/litellm_core_utils/prompt_templates/common_utils.py +++ b/litellm/litellm_core_utils/prompt_templates/common_utils.py @@ -719,17 +719,32 @@ def extract_file_data(file_data: FileTypes) -> ExtractedFileData: with open(file_content, "rb") as f: content = f.read() elif isinstance(file_content, io.IOBase): - # If it's a file-like object - # Try to get filename from file handle if not already set + # If it's a file-like object, keep it as-is to avoid loading the entire + # file into memory. Callers that need bytes can call .read() themselves; + # callers streaming the data to an HTTP request should pass the object + # directly so the transfer is chunked. if not filename and hasattr(file_content, "name"): filename = Path(file_content.name).name - content = file_content.read() - - if isinstance(content, str): - content = content.encode("utf-8") - # Reset file pointer to beginning + # Compute file size via seek/tell so providers that need Content-Length + # (e.g. Gemini resumable upload) don't have to load all bytes. + file_content.seek(0, 2) + file_size: int = file_content.tell() file_content.seek(0) + + return ExtractedFileData( + filename=filename, + content=file_content, # type: ignore[typeddict-item] + content_type=content_type + or ( + mimetypes.guess_type(filename)[0] + if filename + else "application/octet-stream" + ) + or "application/octet-stream", + headers=file_headers, + file_size=file_size, + ) elif isinstance(file_content, bytes): content = file_content else: @@ -748,6 +763,7 @@ def extract_file_data(file_data: FileTypes) -> ExtractedFileData: content=content, content_type=content_type, headers=file_headers, + file_size=len(content), ) diff --git a/litellm/types/utils.py b/litellm/types/utils.py index e7f0cd77143..b191276c465 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2,6 +2,7 @@ import json import time from enum import Enum from typing import ( + IO, TYPE_CHECKING, Any, Dict, @@ -3477,15 +3478,23 @@ class ExtractedFileData(TypedDict): Attributes: filename: Name of the file if provided - content: The file content in bytes + content: The file content as bytes or a seekable IO[bytes] object. + When an IO[bytes] is returned the file pointer is positioned at offset 0. + Callers that need the full bytes (e.g. to compute a hash) should call + .read() themselves; callers that pass the content to an HTTP request + should prefer to pass the IO object directly so the data is streamed + rather than copied into memory. content_type: MIME type of the file headers: Any additional headers for the file + file_size: The size of the file in bytes (pre-computed via seek/tell to + avoid loading the file into memory just for its length). """ filename: Optional[str] - content: bytes + content: Union[bytes, "IO[bytes]"] content_type: Optional[str] headers: Mapping[str, str] + file_size: Optional[int] class SpecialEnums(Enum):