mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(types/utils): ExtractedFileData content supports IO[bytes] with file_size field
This commit is contained in:
parent
24949a0140
commit
3e1518dcc4
2 changed files with 34 additions and 9 deletions
|
|
@ -719,17 +719,32 @@ def extract_file_data(file_data: FileTypes) -> ExtractedFileData:
|
|||
with open(file_content, "rb") as f:
|
||||
content = f.read()
|
||||
elif isinstance(file_content, io.IOBase):
|
||||
# If it's a file-like object
|
||||
# Try to get filename from file handle if not already set
|
||||
# If it's a file-like object, keep it as-is to avoid loading the entire
|
||||
# file into memory. Callers that need bytes can call .read() themselves;
|
||||
# callers streaming the data to an HTTP request should pass the object
|
||||
# directly so the transfer is chunked.
|
||||
if not filename and hasattr(file_content, "name"):
|
||||
filename = Path(file_content.name).name
|
||||
|
||||
content = file_content.read()
|
||||
|
||||
if isinstance(content, str):
|
||||
content = content.encode("utf-8")
|
||||
# Reset file pointer to beginning
|
||||
# Compute file size via seek/tell so providers that need Content-Length
|
||||
# (e.g. Gemini resumable upload) don't have to load all bytes.
|
||||
file_content.seek(0, 2)
|
||||
file_size: int = file_content.tell()
|
||||
file_content.seek(0)
|
||||
|
||||
return ExtractedFileData(
|
||||
filename=filename,
|
||||
content=file_content, # type: ignore[typeddict-item]
|
||||
content_type=content_type
|
||||
or (
|
||||
mimetypes.guess_type(filename)[0]
|
||||
if filename
|
||||
else "application/octet-stream"
|
||||
)
|
||||
or "application/octet-stream",
|
||||
headers=file_headers,
|
||||
file_size=file_size,
|
||||
)
|
||||
elif isinstance(file_content, bytes):
|
||||
content = file_content
|
||||
else:
|
||||
|
|
@ -748,6 +763,7 @@ def extract_file_data(file_data: FileTypes) -> ExtractedFileData:
|
|||
content=content,
|
||||
content_type=content_type,
|
||||
headers=file_headers,
|
||||
file_size=len(content),
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ import json
|
|||
import time
|
||||
from enum import Enum
|
||||
from typing import (
|
||||
IO,
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
Dict,
|
||||
|
|
@ -3477,15 +3478,23 @@ class ExtractedFileData(TypedDict):
|
|||
|
||||
Attributes:
|
||||
filename: Name of the file if provided
|
||||
content: The file content in bytes
|
||||
content: The file content as bytes or a seekable IO[bytes] object.
|
||||
When an IO[bytes] is returned the file pointer is positioned at offset 0.
|
||||
Callers that need the full bytes (e.g. to compute a hash) should call
|
||||
.read() themselves; callers that pass the content to an HTTP request
|
||||
should prefer to pass the IO object directly so the data is streamed
|
||||
rather than copied into memory.
|
||||
content_type: MIME type of the file
|
||||
headers: Any additional headers for the file
|
||||
file_size: The size of the file in bytes (pre-computed via seek/tell to
|
||||
avoid loading the file into memory just for its length).
|
||||
"""
|
||||
|
||||
filename: Optional[str]
|
||||
content: bytes
|
||||
content: Union[bytes, "IO[bytes]"]
|
||||
content_type: Optional[str]
|
||||
headers: Mapping[str, str]
|
||||
file_size: Optional[int]
|
||||
|
||||
|
||||
class SpecialEnums(Enum):
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue