mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
[Fix] - Reliability fix OOMs with image url handling#19257
This commit is contained in:
parent
c6e358222c
commit
2962afbff6
7 changed files with 310 additions and 55 deletions
|
|
@ -744,6 +744,7 @@ router_settings:
|
|||
| LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD | If true, prints the standard logging payload to the console - useful for debugging
|
||||
| LITELM_ENVIRONMENT | Environment for LiteLLM Instance. This is currently only logged to DeepEval to determine the environment for DeepEval integration.
|
||||
| LOGFIRE_TOKEN | Token for Logfire logging service
|
||||
| LOGFIRE_BASE_URL | Base URL for Logfire logging service (useful for self hosted deployments)
|
||||
| LOGGING_WORKER_CONCURRENCY | Maximum number of concurrent coroutine slots for the logging worker on the asyncio event loop. Default is 100. Setting too high will flood the event loop with logging tasks which will lower the overall latency of the requests.
|
||||
| LOGGING_WORKER_MAX_QUEUE_SIZE | Maximum size of the logging worker queue. When the queue is full, the worker aggressively clears tasks to make room instead of dropping logs. Default is 50,000
|
||||
| LOGGING_WORKER_MAX_TIME_PER_COROUTINE | Maximum time in seconds allowed for each coroutine in the logging worker before timing out. Default is 20.0
|
||||
|
|
@ -754,6 +755,7 @@ router_settings:
|
|||
| LOGGING_WORKER_AGGRESSIVE_CLEAR_COOLDOWN_SECONDS | Cooldown time in seconds before allowing another aggressive clear operation when the queue is full. Default is 0.5
|
||||
| MAX_STRING_LENGTH_PROMPT_IN_DB | Maximum length for strings in spend logs when sanitizing request bodies. Strings longer than this will be truncated. Default is 1000
|
||||
| MAX_IN_MEMORY_QUEUE_FLUSH_COUNT | Maximum count for in-memory queue flush operations. Default is 1000
|
||||
| MAX_IMAGE_URL_DOWNLOAD_SIZE_MB | Maximum size in MB for downloading images from URLs. Prevents memory issues from downloading very large images. Images exceeding this limit will be rejected before download. Set to 0 to completely disable image URL handling (all image_url requests will be blocked). Default is 50MB (matching [OpenAI's limit](https://platform.openai.com/docs/guides/images-vision?api-mode=chat#image-input-requirements))
|
||||
| MAX_LONG_SIDE_FOR_IMAGE_HIGH_RES | Maximum length for the long side of high-resolution images. Default is 2000
|
||||
| MAX_REDIS_BUFFER_DEQUEUE_COUNT | Maximum count for Redis buffer dequeue operations. Default is 100
|
||||
| MAX_SHORT_SIDE_FOR_IMAGE_HIGH_RES | Maximum length for the short side of high-resolution images. Default is 768
|
||||
|
|
@ -771,10 +773,18 @@ router_settings:
|
|||
| MINIMUM_PROMPT_CACHE_TOKEN_COUNT | Minimum token count for caching a prompt. Default is 1024
|
||||
| MISTRAL_API_BASE | Base URL for Mistral API. Default is https://api.mistral.ai
|
||||
| MISTRAL_API_KEY | API key for Mistral API
|
||||
| MICROSOFT_AUTHORIZATION_ENDPOINT | Custom authorization endpoint URL for Microsoft SSO (overrides default Microsoft OAuth authorization endpoint)
|
||||
| MICROSOFT_CLIENT_ID | Client ID for Microsoft services
|
||||
| MICROSOFT_CLIENT_SECRET | Client secret for Microsoft services
|
||||
| MICROSOFT_TENANT | Tenant ID for Microsoft Azure
|
||||
| MICROSOFT_SERVICE_PRINCIPAL_ID | Service Principal ID for Microsoft Enterprise Application. (This is an advanced feature if you want litellm to auto-assign members to Litellm Teams based on their Microsoft Entra ID Groups)
|
||||
| MICROSOFT_TENANT | Tenant ID for Microsoft Azure
|
||||
| MICROSOFT_TOKEN_ENDPOINT | Custom token endpoint URL for Microsoft SSO (overrides default Microsoft OAuth token endpoint)
|
||||
| MICROSOFT_USER_DISPLAY_NAME_ATTRIBUTE | Field name for user display name in Microsoft SSO response. Default is `displayName`
|
||||
| MICROSOFT_USER_EMAIL_ATTRIBUTE | Field name for user email in Microsoft SSO response. Default is `userPrincipalName`
|
||||
| MICROSOFT_USER_FIRST_NAME_ATTRIBUTE | Field name for user first name in Microsoft SSO response. Default is `givenName`
|
||||
| MICROSOFT_USER_ID_ATTRIBUTE | Field name for user ID in Microsoft SSO response. Default is `id`
|
||||
| MICROSOFT_USER_LAST_NAME_ATTRIBUTE | Field name for user last name in Microsoft SSO response. Default is `surname`
|
||||
| MICROSOFT_USERINFO_ENDPOINT | Custom userinfo endpoint URL for Microsoft SSO (overrides default Microsoft Graph userinfo endpoint)
|
||||
| NO_DOCS | Flag to disable Swagger UI documentation
|
||||
| NO_REDOC | Flag to disable Redoc documentation
|
||||
| NO_PROXY | List of addresses to bypass proxy
|
||||
|
|
@ -856,7 +866,6 @@ router_settings:
|
|||
| SECRET_MANAGER_REFRESH_INTERVAL | Refresh interval in seconds for secret manager. Default is 86400 (24 hours)
|
||||
| SEPARATE_HEALTH_APP | If set to '1', runs health endpoints on a separate ASGI app and port. Default: '0'.
|
||||
| SEPARATE_HEALTH_PORT | Port for the separate health endpoints app. Only used if SEPARATE_HEALTH_APP=1. Default: 4001.
|
||||
| SUPERVISORD_STOPWAITSECS | Upper bound timeout in seconds for graceful shutdown when SEPARATE_HEALTH_APP=1. Default: 3600 (1 hour).
|
||||
| SERVER_ROOT_PATH | Root path for the server application
|
||||
| SEND_USER_API_KEY_ALIAS | Flag to send user API key alias to Zscaler AI Guard. Default is False
|
||||
| SEND_USER_API_KEY_TEAM_ID | Flag to send user API key team ID to Zscaler AI Guard. Default is False
|
||||
|
|
|
|||
|
|
@ -48,6 +48,11 @@ DEFAULT_REPLICATE_POLLING_DELAY_SECONDS = int(
|
|||
DEFAULT_IMAGE_TOKEN_COUNT = int(os.getenv("DEFAULT_IMAGE_TOKEN_COUNT", 250))
|
||||
DEFAULT_IMAGE_WIDTH = int(os.getenv("DEFAULT_IMAGE_WIDTH", 300))
|
||||
DEFAULT_IMAGE_HEIGHT = int(os.getenv("DEFAULT_IMAGE_HEIGHT", 300))
|
||||
# Maximum size for image URL downloads in MB (default 50MB, set to 0 to disable limit)
|
||||
# This prevents memory issues from downloading very large images
|
||||
# Maps to OpenAI's 50 MB payload limit - requests with images exceeding this size will be rejected
|
||||
# Set MAX_IMAGE_URL_DOWNLOAD_SIZE_MB=0 to disable image URL handling entirely
|
||||
MAX_IMAGE_URL_DOWNLOAD_SIZE_MB = float(os.getenv("MAX_IMAGE_URL_DOWNLOAD_SIZE_MB", 50))
|
||||
MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB = int(
|
||||
os.getenv("MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB", 1024)
|
||||
) # 1MB = 1024KB
|
||||
|
|
@ -1073,6 +1078,13 @@ LITELLM_TRUNCATED_PAYLOAD_FIELD = "litellm_truncated"
|
|||
|
||||
########################### LiteLLM Proxy Specific Constants ###########################
|
||||
########################################################################################
|
||||
|
||||
# Standard headers that are always checked for customer/end-user ID (no configuration required)
|
||||
# These headers work out-of-the-box for tools like Claude Code that support custom headers
|
||||
STANDARD_CUSTOMER_ID_HEADERS = [
|
||||
"x-litellm-customer-id",
|
||||
"x-litellm-end-user-id",
|
||||
]
|
||||
MAX_SPENDLOG_ROWS_TO_QUERY = int(
|
||||
os.getenv("MAX_SPENDLOG_ROWS_TO_QUERY", 1_000_000)
|
||||
) # if spendLogs has more than 1M rows, do not query the DB
|
||||
|
|
@ -1285,3 +1297,20 @@ COROUTINE_CHECKER_MAX_SIZE_IN_MEMORY = int(
|
|||
########################### RAG Text Splitter Constants ###########################
|
||||
DEFAULT_CHUNK_SIZE = int(os.getenv("DEFAULT_CHUNK_SIZE", 1000))
|
||||
DEFAULT_CHUNK_OVERLAP = int(os.getenv("DEFAULT_CHUNK_OVERLAP", 200))
|
||||
|
||||
########################### Microsoft SSO Constants ###########################
|
||||
MICROSOFT_USER_EMAIL_ATTRIBUTE = str(
|
||||
os.getenv("MICROSOFT_USER_EMAIL_ATTRIBUTE", "userPrincipalName")
|
||||
)
|
||||
MICROSOFT_USER_DISPLAY_NAME_ATTRIBUTE = str(
|
||||
os.getenv("MICROSOFT_USER_DISPLAY_NAME_ATTRIBUTE", "displayName")
|
||||
)
|
||||
MICROSOFT_USER_ID_ATTRIBUTE = str(
|
||||
os.getenv("MICROSOFT_USER_ID_ATTRIBUTE", "id")
|
||||
)
|
||||
MICROSOFT_USER_FIRST_NAME_ATTRIBUTE = str(
|
||||
os.getenv("MICROSOFT_USER_FIRST_NAME_ATTRIBUTE", "givenName")
|
||||
)
|
||||
MICROSOFT_USER_LAST_NAME_ATTRIBUTE = str(
|
||||
os.getenv("MICROSOFT_USER_LAST_NAME_ATTRIBUTE", "surname")
|
||||
)
|
||||
|
|
|
|||
|
|
@ -902,22 +902,22 @@ def convert_to_anthropic_image_obj(
|
|||
media_type=media_type,
|
||||
data=base64_data,
|
||||
)
|
||||
except litellm.ImageFetchError:
|
||||
raise
|
||||
except Exception as e:
|
||||
if "Error: Unable to fetch image from URL" in str(e):
|
||||
raise e
|
||||
raise Exception(
|
||||
"""Image url not in expected format. Example Expected input - "image_url": "data:image/jpeg;base64,{base64_image}". Supported formats - ['image/jpeg', 'image/png', 'image/gif', 'image/webp']."""
|
||||
f"""Image url not in expected format. Example Expected input - "image_url": "data:image/jpeg;base64,{{base64_image}}". Supported formats - ['image/jpeg', 'image/png', 'image/gif', 'image/webp']. Error: {str(e)}"""
|
||||
)
|
||||
|
||||
|
||||
def create_anthropic_image_param(
|
||||
image_url_input: Union[str, dict],
|
||||
image_url_input: Union[str, dict],
|
||||
format: Optional[str] = None,
|
||||
is_bedrock_invoke: bool = False
|
||||
is_bedrock_invoke: bool = False,
|
||||
) -> AnthropicMessagesImageParam:
|
||||
"""
|
||||
Create an AnthropicMessagesImageParam from an image URL input.
|
||||
|
||||
|
||||
Supports both URL references (for HTTP/HTTPS URLs) and base64 encoding.
|
||||
"""
|
||||
# Extract URL and format from input
|
||||
|
|
@ -927,7 +927,7 @@ def create_anthropic_image_param(
|
|||
image_url = image_url_input.get("url", "")
|
||||
if format is None:
|
||||
format = image_url_input.get("format")
|
||||
|
||||
|
||||
# Check if the image URL is an HTTP/HTTPS URL
|
||||
if image_url.startswith("http://") or image_url.startswith("https://"):
|
||||
# For Bedrock invoke and Vertex AI Anthropic, always convert URLs to base64
|
||||
|
|
@ -1071,8 +1071,14 @@ def anthropic_messages_pt_xml(messages: list):
|
|||
if isinstance(messages[msg_i]["content"], list):
|
||||
for m in messages[msg_i]["content"]:
|
||||
if m.get("type", "") == "image_url":
|
||||
format = m["image_url"].get("format") if isinstance(m["image_url"], dict) else None
|
||||
image_param = create_anthropic_image_param(m["image_url"], format=format)
|
||||
format = (
|
||||
m["image_url"].get("format")
|
||||
if isinstance(m["image_url"], dict)
|
||||
else None
|
||||
)
|
||||
image_param = create_anthropic_image_param(
|
||||
m["image_url"], format=format
|
||||
)
|
||||
# Convert to dict format for XML version
|
||||
source = image_param["source"]
|
||||
if isinstance(source, dict) and source.get("type") == "url":
|
||||
|
|
@ -1381,10 +1387,10 @@ def convert_to_gemini_tool_call_invoke(
|
|||
if tool_calls is not None:
|
||||
for idx, tool in enumerate(tool_calls):
|
||||
if "function" in tool:
|
||||
gemini_function_call: Optional[
|
||||
VertexFunctionCall
|
||||
] = _gemini_tool_call_invoke_helper(
|
||||
function_call_params=tool["function"]
|
||||
gemini_function_call: Optional[VertexFunctionCall] = (
|
||||
_gemini_tool_call_invoke_helper(
|
||||
function_call_params=tool["function"]
|
||||
)
|
||||
)
|
||||
if gemini_function_call is not None:
|
||||
part_dict: VertexPartType = {
|
||||
|
|
@ -1484,10 +1490,10 @@ def convert_to_gemini_tool_call_result(
|
|||
}
|
||||
"""
|
||||
from litellm.types.llms.vertex_ai import BlobType
|
||||
|
||||
|
||||
content_str: str = ""
|
||||
inline_data: Optional[BlobType] = None
|
||||
|
||||
|
||||
if "content" in message:
|
||||
if isinstance(message["content"], str):
|
||||
content_str = message["content"]
|
||||
|
|
@ -1500,15 +1506,21 @@ def convert_to_gemini_tool_call_result(
|
|||
elif content_type in ("input_image", "image_url"):
|
||||
# Extract image for inline_data (for Computer Use screenshots and tool results)
|
||||
image_url_data = content.get("image_url", "")
|
||||
image_url = image_url_data.get("url", "") if isinstance(image_url_data, dict) else image_url_data
|
||||
|
||||
image_url = (
|
||||
image_url_data.get("url", "")
|
||||
if isinstance(image_url_data, dict)
|
||||
else image_url_data
|
||||
)
|
||||
|
||||
if image_url:
|
||||
# Convert image to base64 blob format for Gemini
|
||||
try:
|
||||
image_obj = convert_to_anthropic_image_obj(image_url, format=None)
|
||||
image_obj = convert_to_anthropic_image_obj(
|
||||
image_url, format=None
|
||||
)
|
||||
inline_data = BlobType(
|
||||
data=image_obj["data"],
|
||||
mime_type=image_obj["media_type"]
|
||||
mime_type=image_obj["media_type"],
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_logger.warning(
|
||||
|
|
@ -1540,7 +1552,6 @@ def convert_to_gemini_tool_call_result(
|
|||
# For Computer Use, the response should contain structured data like {"url": "..."}
|
||||
response_data: dict
|
||||
try:
|
||||
import json
|
||||
if content_str.strip().startswith("{") or content_str.strip().startswith("["):
|
||||
# Try to parse as JSON (for Computer Use structured responses)
|
||||
parsed = json.loads(content_str)
|
||||
|
|
@ -1553,7 +1564,7 @@ def convert_to_gemini_tool_call_result(
|
|||
except (json.JSONDecodeError, ValueError):
|
||||
# Not valid JSON, wrap in content field
|
||||
response_data = {"content": content_str}
|
||||
|
||||
|
||||
# We can't determine from openai message format whether it's a successful or
|
||||
# error call result so default to the successful result template
|
||||
_function_response = VertexFunctionResponse(
|
||||
|
|
@ -1562,7 +1573,7 @@ def convert_to_gemini_tool_call_result(
|
|||
|
||||
# Create part with function_response, and optionally inline_data for images (Computer Use)
|
||||
_part: VertexPartType = {"function_response": _function_response}
|
||||
|
||||
|
||||
# For Computer Use, if we have an image, we need separate parts:
|
||||
# - One part with function_response
|
||||
# - One part with inline_data
|
||||
|
|
@ -1570,19 +1581,19 @@ def convert_to_gemini_tool_call_result(
|
|||
if inline_data:
|
||||
image_part: VertexPartType = {"inline_data": inline_data}
|
||||
return [_part, image_part]
|
||||
|
||||
|
||||
return _part
|
||||
|
||||
|
||||
def _sanitize_anthropic_tool_use_id(tool_use_id: str) -> str:
|
||||
"""
|
||||
Sanitize tool_use_id to match Anthropic's required pattern: ^[a-zA-Z0-9_-]+$
|
||||
|
||||
|
||||
Anthropic requires tool_use_id to only contain alphanumeric characters, underscores, and hyphens.
|
||||
This function replaces any invalid characters with underscores.
|
||||
"""
|
||||
# Replace any character that's not alphanumeric, underscore, or hyphen with underscore
|
||||
sanitized = re.sub(r'[^a-zA-Z0-9_-]', '_', tool_use_id)
|
||||
sanitized = re.sub(r"[^a-zA-Z0-9_-]", "_", tool_use_id)
|
||||
# Ensure it's not empty (fallback to a default if needed)
|
||||
if not sanitized:
|
||||
sanitized = "tool_use_id"
|
||||
|
|
@ -1644,13 +1655,19 @@ def convert_to_anthropic_tool_result(
|
|||
)
|
||||
)
|
||||
elif content["type"] == "image_url":
|
||||
format = content["image_url"].get("format") if isinstance(content["image_url"], dict) else None
|
||||
_anthropic_image_param = create_anthropic_image_param(content["image_url"], format=format)
|
||||
format = (
|
||||
content["image_url"].get("format")
|
||||
if isinstance(content["image_url"], dict)
|
||||
else None
|
||||
)
|
||||
_anthropic_image_param = create_anthropic_image_param(
|
||||
content["image_url"], format=format
|
||||
)
|
||||
_anthropic_image_param = add_cache_control_to_content(
|
||||
anthropic_content_element=_anthropic_image_param,
|
||||
original_content_element=content,
|
||||
)
|
||||
anthropic_content_list.append(_anthropic_image_param)
|
||||
anthropic_content_list.append(cast(AnthropicMessagesImageParam, _anthropic_image_param))
|
||||
|
||||
anthropic_content = anthropic_content_list
|
||||
anthropic_tool_result: Optional[AnthropicMessagesToolResultParam] = None
|
||||
|
|
@ -1665,7 +1682,9 @@ def convert_to_anthropic_tool_result(
|
|||
# We can't determine from openai message format whether it's a successful or
|
||||
# error call result so default to the successful result template
|
||||
anthropic_tool_result = AnthropicMessagesToolResultParam(
|
||||
type="tool_result", tool_use_id=sanitized_tool_use_id, content=anthropic_content
|
||||
type="tool_result",
|
||||
tool_use_id=sanitized_tool_use_id,
|
||||
content=anthropic_content,
|
||||
)
|
||||
|
||||
if message["role"] == "function":
|
||||
|
|
@ -1674,7 +1693,9 @@ def convert_to_anthropic_tool_result(
|
|||
# Sanitize tool_use_id to match Anthropic's pattern requirement: ^[a-zA-Z0-9_-]+$
|
||||
sanitized_tool_use_id = _sanitize_anthropic_tool_use_id(tool_call_id)
|
||||
anthropic_tool_result = AnthropicMessagesToolResultParam(
|
||||
type="tool_result", tool_use_id=sanitized_tool_use_id, content=anthropic_content
|
||||
type="tool_result",
|
||||
tool_use_id=sanitized_tool_use_id,
|
||||
content=anthropic_content,
|
||||
)
|
||||
|
||||
if anthropic_tool_result is None:
|
||||
|
|
@ -1690,12 +1711,15 @@ def convert_function_to_anthropic_tool_invoke(
|
|||
try:
|
||||
_name = get_attribute_or_key(function_call, "name") or ""
|
||||
_arguments = get_attribute_or_key(function_call, "arguments")
|
||||
|
||||
tool_input = json.loads(_arguments) if _arguments else {}
|
||||
|
||||
anthropic_tool_invoke = [
|
||||
AnthropicMessagesToolUseParam(
|
||||
type="tool_use",
|
||||
id=str(uuid.uuid4()),
|
||||
name=_name,
|
||||
input=json.loads(_arguments) if _arguments else {},
|
||||
input=tool_input,
|
||||
)
|
||||
]
|
||||
return anthropic_tool_invoke
|
||||
|
|
@ -1749,7 +1773,9 @@ def convert_to_anthropic_tool_invoke(
|
|||
|
||||
Fixes: https://github.com/BerriAI/litellm/issues/17737
|
||||
"""
|
||||
anthropic_tool_invoke: List[Union[AnthropicMessagesToolUseParam, Dict[str, Any]]] = []
|
||||
anthropic_tool_invoke: List[
|
||||
Union[AnthropicMessagesToolUseParam, Dict[str, Any]]
|
||||
] = []
|
||||
|
||||
for tool in tool_calls:
|
||||
if not get_attribute_or_key(tool, "type") == "function":
|
||||
|
|
@ -2015,11 +2041,17 @@ def anthropic_messages_pt( # noqa: PLR0915
|
|||
for m in user_message_types_block["content"]:
|
||||
if m.get("type", "") == "image_url":
|
||||
m = cast(ChatCompletionImageObject, m)
|
||||
format = m["image_url"].get("format") if isinstance(m["image_url"], dict) else None
|
||||
format = (
|
||||
m["image_url"].get("format")
|
||||
if isinstance(m["image_url"], dict)
|
||||
else None
|
||||
)
|
||||
# Convert ChatCompletionImageUrlObject to dict if needed
|
||||
image_url_value = m["image_url"]
|
||||
if isinstance(image_url_value, str):
|
||||
image_url_input: Union[str, dict[str, Any]] = image_url_value
|
||||
image_url_input: Union[str, dict[str, Any]] = (
|
||||
image_url_value
|
||||
)
|
||||
else:
|
||||
# ChatCompletionImageUrlObject or dict case - convert to dict
|
||||
image_url_input = {
|
||||
|
|
@ -2029,20 +2061,26 @@ def anthropic_messages_pt( # noqa: PLR0915
|
|||
# Bedrock invoke models have format: invoke/...
|
||||
# Vertex AI Anthropic also doesn't support URL sources for images
|
||||
is_bedrock_invoke = model.lower().startswith("invoke/")
|
||||
is_vertex_ai = llm_provider.startswith("vertex_ai") if llm_provider else False
|
||||
is_vertex_ai = (
|
||||
llm_provider.startswith("vertex_ai")
|
||||
if llm_provider
|
||||
else False
|
||||
)
|
||||
force_base64 = is_bedrock_invoke or is_vertex_ai
|
||||
_anthropic_content_element = create_anthropic_image_param(
|
||||
image_url_input, format=format, is_bedrock_invoke=force_base64
|
||||
)
|
||||
image_url_input,
|
||||
format=format,
|
||||
is_bedrock_invoke=force_base64,
|
||||
)
|
||||
_content_element = add_cache_control_to_content(
|
||||
anthropic_content_element=_anthropic_content_element,
|
||||
original_content_element=dict(m),
|
||||
)
|
||||
|
||||
if "cache_control" in _content_element:
|
||||
_anthropic_content_element[
|
||||
"cache_control"
|
||||
] = _content_element["cache_control"]
|
||||
_anthropic_content_element["cache_control"] = (
|
||||
_content_element["cache_control"]
|
||||
)
|
||||
user_content.append(_anthropic_content_element)
|
||||
elif m.get("type", "") == "text":
|
||||
m = cast(ChatCompletionTextObject, m)
|
||||
|
|
@ -2080,9 +2118,9 @@ def anthropic_messages_pt( # noqa: PLR0915
|
|||
)
|
||||
|
||||
if "cache_control" in _content_element:
|
||||
_anthropic_content_text_element[
|
||||
"cache_control"
|
||||
] = _content_element["cache_control"]
|
||||
_anthropic_content_text_element["cache_control"] = (
|
||||
_content_element["cache_control"]
|
||||
)
|
||||
|
||||
user_content.append(_anthropic_content_text_element)
|
||||
|
||||
|
|
@ -2178,18 +2216,27 @@ def anthropic_messages_pt( # noqa: PLR0915
|
|||
): # support assistant tool invoke conversion
|
||||
# Get web_search_results from provider_specific_fields for server_tool_use reconstruction
|
||||
# Fixes: https://github.com/BerriAI/litellm/issues/17737
|
||||
_provider_specific_fields_raw = assistant_content_block.get("provider_specific_fields")
|
||||
_provider_specific_fields_raw = assistant_content_block.get(
|
||||
"provider_specific_fields"
|
||||
)
|
||||
_provider_specific_fields: Dict[str, Any] = {}
|
||||
if isinstance(_provider_specific_fields_raw, dict):
|
||||
_provider_specific_fields = cast(Dict[str, Any], _provider_specific_fields_raw)
|
||||
_web_search_results = _provider_specific_fields.get("web_search_results")
|
||||
_provider_specific_fields = cast(
|
||||
Dict[str, Any], _provider_specific_fields_raw
|
||||
)
|
||||
_web_search_results = _provider_specific_fields.get(
|
||||
"web_search_results"
|
||||
)
|
||||
tool_invoke_results = convert_to_anthropic_tool_invoke(
|
||||
assistant_tool_calls,
|
||||
web_search_results=_web_search_results,
|
||||
)
|
||||
# AnthropicMessagesAssistantMessageValues includes AnthropicMessagesToolUseParam
|
||||
assistant_content.extend(
|
||||
cast(List[AnthropicMessagesAssistantMessageValues], tool_invoke_results)
|
||||
cast(
|
||||
List[AnthropicMessagesAssistantMessageValues],
|
||||
tool_invoke_results,
|
||||
)
|
||||
)
|
||||
|
||||
assistant_function_call = assistant_content_block.get("function_call")
|
||||
|
|
@ -3252,14 +3299,18 @@ def _convert_to_bedrock_tool_call_result(
|
|||
"""
|
||||
-
|
||||
"""
|
||||
tool_result_content_blocks:List[BedrockToolResultContentBlock] = []
|
||||
tool_result_content_blocks: List[BedrockToolResultContentBlock] = []
|
||||
if isinstance(message["content"], str):
|
||||
tool_result_content_blocks.append(BedrockToolResultContentBlock(text=message["content"]))
|
||||
tool_result_content_blocks.append(
|
||||
BedrockToolResultContentBlock(text=message["content"])
|
||||
)
|
||||
elif isinstance(message["content"], List):
|
||||
content_list = message["content"]
|
||||
for content in content_list:
|
||||
if content["type"] == "text":
|
||||
tool_result_content_blocks.append(BedrockToolResultContentBlock(text=content["text"]))
|
||||
tool_result_content_blocks.append(
|
||||
BedrockToolResultContentBlock(text=content["text"])
|
||||
)
|
||||
elif content["type"] == "image_url":
|
||||
format: Optional[str] = None
|
||||
if isinstance(content["image_url"], dict):
|
||||
|
|
@ -3267,12 +3318,14 @@ def _convert_to_bedrock_tool_call_result(
|
|||
format = content["image_url"].get("format")
|
||||
else:
|
||||
image_url = content["image_url"]
|
||||
_block:BedrockContentBlock = BedrockImageProcessor.process_image_sync(
|
||||
_block: BedrockContentBlock = BedrockImageProcessor.process_image_sync(
|
||||
image_url=image_url,
|
||||
format=format,
|
||||
)
|
||||
if "image" in _block:
|
||||
tool_result_content_blocks.append(BedrockToolResultContentBlock(image=_block["image"]))
|
||||
tool_result_content_blocks.append(
|
||||
BedrockToolResultContentBlock(image=_block["image"])
|
||||
)
|
||||
|
||||
message.get("name", "")
|
||||
id = str(message.get("tool_call_id", str(uuid.uuid4())))
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ from httpx import Response
|
|||
import litellm
|
||||
from litellm import verbose_logger
|
||||
from litellm.caching.caching import InMemoryCache
|
||||
from litellm.constants import MAX_IMAGE_URL_DOWNLOAD_SIZE_MB
|
||||
|
||||
MAX_IMGS_IN_MEMORY = 10
|
||||
|
||||
|
|
@ -21,7 +22,25 @@ def _process_image_response(response: Response, url: str) -> str:
|
|||
f"Error: Unable to fetch image from URL. Status code: {response.status_code}, url={url}"
|
||||
)
|
||||
|
||||
# Check size before downloading if Content-Length header is present
|
||||
content_length = response.headers.get("Content-Length")
|
||||
if content_length is not None:
|
||||
size_mb = int(content_length) / (1024 * 1024)
|
||||
if size_mb > MAX_IMAGE_URL_DOWNLOAD_SIZE_MB:
|
||||
raise litellm.ImageFetchError(
|
||||
f"Error: Image size ({size_mb:.2f}MB) exceeds maximum allowed size ({MAX_IMAGE_URL_DOWNLOAD_SIZE_MB}MB). url={url}"
|
||||
)
|
||||
|
||||
image_bytes = response.content
|
||||
|
||||
# Check actual size after download if Content-Length was not available
|
||||
if content_length is None:
|
||||
size_mb = len(image_bytes) / (1024 * 1024)
|
||||
if size_mb > MAX_IMAGE_URL_DOWNLOAD_SIZE_MB:
|
||||
raise litellm.ImageFetchError(
|
||||
f"Error: Image size ({size_mb:.2f}MB) exceeds maximum allowed size ({MAX_IMAGE_URL_DOWNLOAD_SIZE_MB}MB). url={url}"
|
||||
)
|
||||
|
||||
base64_image = base64.b64encode(image_bytes).decode("utf-8")
|
||||
|
||||
image_type = response.headers.get("Content-Type")
|
||||
|
|
@ -48,6 +67,12 @@ def _process_image_response(response: Response, url: str) -> str:
|
|||
|
||||
|
||||
async def async_convert_url_to_base64(url: str) -> str:
|
||||
# If MAX_IMAGE_URL_DOWNLOAD_SIZE_MB is 0, block all image downloads
|
||||
if MAX_IMAGE_URL_DOWNLOAD_SIZE_MB == 0:
|
||||
raise litellm.ImageFetchError(
|
||||
f"Error: Image URL download is disabled (MAX_IMAGE_URL_DOWNLOAD_SIZE_MB=0). url={url}"
|
||||
)
|
||||
|
||||
cached_result = in_memory_cache.get_cache(url)
|
||||
if cached_result:
|
||||
return cached_result
|
||||
|
|
@ -67,6 +92,12 @@ async def async_convert_url_to_base64(url: str) -> str:
|
|||
|
||||
|
||||
def convert_url_to_base64(url: str) -> str:
|
||||
# If MAX_IMAGE_URL_DOWNLOAD_SIZE_MB is 0, block all image downloads
|
||||
if MAX_IMAGE_URL_DOWNLOAD_SIZE_MB == 0:
|
||||
raise litellm.ImageFetchError(
|
||||
f"Error: Image URL download is disabled (MAX_IMAGE_URL_DOWNLOAD_SIZE_MB=0). url={url}"
|
||||
)
|
||||
|
||||
cached_result = in_memory_cache.get_cache(url)
|
||||
if cached_result:
|
||||
return cached_result
|
||||
|
|
|
|||
|
|
@ -5,4 +5,3 @@ model_list:
|
|||
- model_name: openai/*
|
||||
litellm_params:
|
||||
model: openai/*
|
||||
|
||||
|
|
|
|||
|
|
@ -1401,3 +1401,38 @@ def test_anthropic_thinking_param_via_map_openai_params():
|
|||
assert "thinkingLevel" not in thinking_config_2, "Should NOT have thinkingLevel for Gemini 2"
|
||||
assert thinking_config_2["includeThoughts"] is True
|
||||
assert thinking_config_2["thinkingBudget"] == 10000
|
||||
|
||||
|
||||
|
||||
def test_gemini_image_size_limit_exceeded():
|
||||
"""
|
||||
Test that large images exceeding MAX_IMAGE_URL_DOWNLOAD_SIZE_MB are rejected.
|
||||
|
||||
This validates that the 50MB default limit prevents downloading very large images
|
||||
that could cause memory issues and pod crashes.
|
||||
"""
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "What is in this image?"
|
||||
},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": "https://upload.wikimedia.org/wikipedia/commons/5/51/Blue_Marble_2002.jpg"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
|
||||
with pytest.raises(litellm.ImageFetchError) as excinfo:
|
||||
completion(
|
||||
model="gemini/gemini-2.5-flash-lite",
|
||||
messages=messages
|
||||
)
|
||||
|
||||
error_message = str(excinfo.value)
|
||||
assert "Image size" in error_message
|
||||
assert "exceeds maximum allowed size" in error_message
|
||||
|
|
|
|||
|
|
@ -1,7 +1,10 @@
|
|||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
from httpx import Request, Response
|
||||
|
||||
import litellm
|
||||
from litellm import constants
|
||||
from litellm.litellm_core_utils.prompt_templates.image_handling import (
|
||||
convert_url_to_base64,
|
||||
)
|
||||
|
|
@ -39,3 +42,99 @@ def test_completion_with_invalid_image_url(monkeypatch):
|
|||
)
|
||||
assert excinfo.value.status_code == 400
|
||||
assert "Unable to fetch image" in str(excinfo.value)
|
||||
|
||||
|
||||
class LargeImageClient:
|
||||
"""
|
||||
Client that returns a large image exceeding size limit.
|
||||
"""
|
||||
|
||||
def __init__(self, size_mb=100, include_content_length=True):
|
||||
self.size_mb = size_mb
|
||||
self.include_content_length = include_content_length
|
||||
|
||||
def get(self, url, follow_redirects=True):
|
||||
size_bytes = int(self.size_mb * 1024 * 1024)
|
||||
headers = {"Content-Type": "image/jpeg"}
|
||||
if self.include_content_length:
|
||||
headers["Content-Length"] = str(size_bytes)
|
||||
return Response(
|
||||
status_code=200,
|
||||
headers=headers,
|
||||
content=b"x" * size_bytes,
|
||||
request=Request("GET", url),
|
||||
)
|
||||
|
||||
|
||||
def test_image_exceeds_size_limit_with_content_length(monkeypatch):
|
||||
"""
|
||||
Test that images exceeding MAX_IMAGE_URL_DOWNLOAD_SIZE_MB are rejected when Content-Length header is present.
|
||||
"""
|
||||
monkeypatch.setattr(litellm, "module_level_client", LargeImageClient(size_mb=100))
|
||||
|
||||
with pytest.raises(litellm.ImageFetchError) as excinfo:
|
||||
convert_url_to_base64("https://example.com/large-image.jpg")
|
||||
|
||||
assert "exceeds maximum allowed size" in str(excinfo.value)
|
||||
assert "100.00MB" in str(excinfo.value)
|
||||
assert "50.0MB" in str(excinfo.value)
|
||||
|
||||
|
||||
def test_image_exceeds_size_limit_without_content_length(monkeypatch):
|
||||
"""
|
||||
Test that images exceeding MAX_IMAGE_URL_DOWNLOAD_SIZE_MB are rejected even without Content-Length header.
|
||||
"""
|
||||
monkeypatch.setattr(
|
||||
litellm, "module_level_client", LargeImageClient(size_mb=100, include_content_length=False)
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.ImageFetchError) as excinfo:
|
||||
convert_url_to_base64("https://example.com/large-image.jpg")
|
||||
|
||||
assert "exceeds maximum allowed size" in str(excinfo.value)
|
||||
|
||||
|
||||
class SmallImageClient:
|
||||
"""
|
||||
Client that returns a small valid image.
|
||||
"""
|
||||
|
||||
def get(self, url, follow_redirects=True):
|
||||
size_bytes = 1024
|
||||
headers = {
|
||||
"Content-Type": "image/jpeg",
|
||||
"Content-Length": str(size_bytes),
|
||||
}
|
||||
return Response(
|
||||
status_code=200,
|
||||
headers=headers,
|
||||
content=b"x" * size_bytes,
|
||||
request=Request("GET", url),
|
||||
)
|
||||
|
||||
|
||||
def test_image_within_size_limit(monkeypatch):
|
||||
"""
|
||||
Test that images within size limit are processed successfully.
|
||||
"""
|
||||
monkeypatch.setattr(litellm, "module_level_client", SmallImageClient())
|
||||
|
||||
result = convert_url_to_base64("https://example.com/small-image.jpg")
|
||||
|
||||
assert result.startswith("data:image/jpeg;base64,")
|
||||
|
||||
|
||||
def test_image_size_limit_disabled(monkeypatch):
|
||||
"""
|
||||
Test that setting MAX_IMAGE_URL_DOWNLOAD_SIZE_MB to 0 disables all image URL downloads.
|
||||
"""
|
||||
import litellm.litellm_core_utils.prompt_templates.image_handling as image_handling
|
||||
|
||||
monkeypatch.setattr(litellm, "module_level_client", SmallImageClient())
|
||||
monkeypatch.setattr(image_handling, "MAX_IMAGE_URL_DOWNLOAD_SIZE_MB", 0)
|
||||
|
||||
with pytest.raises(litellm.ImageFetchError) as excinfo:
|
||||
convert_url_to_base64("https://example.com/image.jpg")
|
||||
|
||||
assert "Image URL download is disabled" in str(excinfo.value)
|
||||
assert "MAX_IMAGE_URL_DOWNLOAD_SIZE_MB=0" in str(excinfo.value)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue