fix: duck-type IO check for Python 3.9/3.10 SpooledTemporaryFile compat; fix zero-byte file_size guard

This commit is contained in:
Ishaan Jaffer 2026-03-23 10:05:17 -07:00
parent 8477facb95
commit 5d8dd3391e
4 changed files with 29 additions and 18 deletions

View file

@ -2,12 +2,12 @@
Common utility functions used for translating messages across providers
"""
import io
import mimetypes
import re
from os import PathLike
from pathlib import Path
from typing import (
IO,
TYPE_CHECKING,
Any,
Dict,
@ -718,19 +718,19 @@ def extract_file_data(file_data: FileTypes) -> ExtractedFileData:
filename = Path(str(file_content)).name
with open(file_content, "rb") as f:
content = f.read()
elif isinstance(file_content, io.IOBase):
# If it's a file-like object, keep it as-is to avoid loading the entire
# file into memory. Callers that need bytes can call .read() themselves;
# callers streaming the data to an HTTP request should pass the object
# directly so the transfer is chunked.
if not filename and hasattr(file_content, "name"):
filename = Path(file_content.name).name
elif hasattr(file_content, "read") and hasattr(file_content, "seek"):
# Duck-type check covers io.IOBase subclasses AND SpooledTemporaryFile on
# Python < 3.11 (which does not inherit from io.IOBase but is still
# file-like). Keep as-is to avoid loading the entire file into memory.
file_like = cast(IO[bytes], file_content)
if not filename and hasattr(file_like, "name"):
filename = Path(file_like.name).name
# Compute file size via seek/tell so providers that need Content-Length
# (e.g. Gemini resumable upload) don't have to load all bytes.
file_content.seek(0, 2)
file_size: int = file_content.tell()
file_content.seek(0)
file_like.seek(0, 2)
file_size: int = file_like.tell()
file_like.seek(0)
return ExtractedFileData(
filename=filename,

View file

@ -2986,7 +2986,7 @@ class BaseLLMHTTPHandler:
"headers": transformed_request["upload_request"]["headers"],
"timeout": timeout,
}
if isinstance(upload_data, io.IOBase):
if hasattr(upload_data, "read") and hasattr(upload_data, "seek"):
upload_kwargs["content"] = upload_data
else:
upload_kwargs["data"] = upload_data
@ -3023,7 +3023,9 @@ class BaseLLMHTTPHandler:
):
# Handle traditional file uploads (str, bytes, or IO[bytes] for streaming)
http_method = provider_config.file_upload_http_method.upper()
if isinstance(transformed_request, io.IOBase):
if hasattr(transformed_request, "read") and hasattr(
transformed_request, "seek"
):
# Stream the IO object without loading it fully into memory
upload_kwargs: Dict[str, Any] = {
"url": api_base,
@ -3154,7 +3156,7 @@ class BaseLLMHTTPHandler:
"headers": transformed_request["upload_request"]["headers"],
"timeout": timeout,
}
if isinstance(upload_data, io.IOBase):
if hasattr(upload_data, "read") and hasattr(upload_data, "seek"):
async_upload_kwargs["content"] = upload_data
else:
async_upload_kwargs["data"] = upload_data
@ -3192,7 +3194,9 @@ class BaseLLMHTTPHandler:
):
# Handle traditional file uploads (str, bytes, or IO[bytes] for streaming)
http_method = provider_config.file_upload_http_method.upper()
if isinstance(transformed_request, io.IOBase):
if hasattr(transformed_request, "read") and hasattr(
transformed_request, "seek"
):
# Stream the IO object without loading it fully into memory
async_upload_kwargs: Dict[str, Any] = {
"url": api_base,

View file

@ -126,7 +126,13 @@ class GoogleAIStudioFilesHandler(GeminiModelInfo, BaseFilesConfig):
# Get file size — prefer the pre-computed value (available when content is
# an IO[bytes] object so we didn't have to load all bytes into memory).
file_size = extracted_data.get("file_size") or len(extracted_data["content"]) # type: ignore[arg-type]
# Use explicit `is not None` to avoid treating a 0-byte file as falsy.
_precomputed = extracted_data.get("file_size")
file_size = (
_precomputed
if _precomputed is not None
else len(extracted_data["content"]) # type: ignore[arg-type]
)
# Step 1: Initial resumable upload request
headers = {

View file

@ -1,4 +1,3 @@
import io
import json
import os
import time
@ -293,7 +292,9 @@ class VertexAIFilesConfig(VertexBase, BaseFilesConfig):
return "\n".join(json.dumps(item) for item in vertex_jsonl_content)
elif isinstance(extracted_file_data_content, bytes):
return extracted_file_data_content
elif isinstance(extracted_file_data_content, io.IOBase):
elif hasattr(extracted_file_data_content, "read") and hasattr(
extracted_file_data_content, "seek"
):
return extracted_file_data_content # type: ignore[return-value]
else:
raise ValueError("Unsupported file content type")