fix(types/utils): ExtractedFileData content supports IO[bytes] with file_size field

This commit is contained in:
Ishaan Jaffer 2026-03-23 09:49:49 -07:00
parent 24949a0140
commit 3e1518dcc4
2 changed files with 34 additions and 9 deletions

View file

@ -719,17 +719,32 @@ def extract_file_data(file_data: FileTypes) -> ExtractedFileData:
with open(file_content, "rb") as f:
content = f.read()
elif isinstance(file_content, io.IOBase):
# If it's a file-like object
# Try to get filename from file handle if not already set
# If it's a file-like object, keep it as-is to avoid loading the entire
# file into memory. Callers that need bytes can call .read() themselves;
# callers streaming the data to an HTTP request should pass the object
# directly so the transfer is chunked.
if not filename and hasattr(file_content, "name"):
filename = Path(file_content.name).name
content = file_content.read()
if isinstance(content, str):
content = content.encode("utf-8")
# Reset file pointer to beginning
# Compute file size via seek/tell so providers that need Content-Length
# (e.g. Gemini resumable upload) don't have to load all bytes.
file_content.seek(0, 2)
file_size: int = file_content.tell()
file_content.seek(0)
return ExtractedFileData(
filename=filename,
content=file_content, # type: ignore[typeddict-item]
content_type=content_type
or (
mimetypes.guess_type(filename)[0]
if filename
else "application/octet-stream"
)
or "application/octet-stream",
headers=file_headers,
file_size=file_size,
)
elif isinstance(file_content, bytes):
content = file_content
else:
@ -748,6 +763,7 @@ def extract_file_data(file_data: FileTypes) -> ExtractedFileData:
content=content,
content_type=content_type,
headers=file_headers,
file_size=len(content),
)

View file

@ -2,6 +2,7 @@ import json
import time
from enum import Enum
from typing import (
IO,
TYPE_CHECKING,
Any,
Dict,
@ -3477,15 +3478,23 @@ class ExtractedFileData(TypedDict):
Attributes:
filename: Name of the file if provided
content: The file content in bytes
content: The file content as bytes or a seekable IO[bytes] object.
When an IO[bytes] is returned the file pointer is positioned at offset 0.
Callers that need the full bytes (e.g. to compute a hash) should call
.read() themselves; callers that pass the content to an HTTP request
should prefer to pass the IO object directly so the data is streamed
rather than copied into memory.
content_type: MIME type of the file
headers: Any additional headers for the file
file_size: The size of the file in bytes (pre-computed via seek/tell to
avoid loading the file into memory just for its length).
"""
filename: Optional[str]
content: bytes
content: Union[bytes, "IO[bytes]"]
content_type: Optional[str]
headers: Mapping[str, str]
file_size: Optional[int]
class SpecialEnums(Enum):