mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(responses): propagate message cache_control safely through objects and models
This commit is contained in:
parent
5abe5f82e1
commit
18101345a0
4 changed files with 348 additions and 351 deletions
|
|
@ -62,23 +62,27 @@ def extract_ttl_from_cached_messages(messages: List[AllMessageValues]) -> Option
|
|||
if not is_cached_message(message):
|
||||
continue
|
||||
|
||||
content = message.get("content")
|
||||
content = message.get("content") if isinstance(message, dict) else getattr(message, "content", None)
|
||||
if not content or isinstance(content, str):
|
||||
continue
|
||||
|
||||
for content_item in content:
|
||||
# Type check to ensure content_item is a dictionary before calling .get()
|
||||
if not isinstance(content_item, dict):
|
||||
# Check if content_item is dict or object model
|
||||
if isinstance(content_item, dict):
|
||||
cache_control = content_item.get("cache_control")
|
||||
else:
|
||||
cache_control = getattr(content_item, "cache_control", None)
|
||||
|
||||
if not cache_control:
|
||||
continue
|
||||
|
||||
cache_control = content_item.get("cache_control")
|
||||
if not cache_control or not isinstance(cache_control, dict):
|
||||
cc_type = (
|
||||
cache_control.get("type") if isinstance(cache_control, dict) else getattr(cache_control, "type", None)
|
||||
)
|
||||
if cc_type != "ephemeral":
|
||||
continue
|
||||
|
||||
if cache_control.get("type") != "ephemeral":
|
||||
continue
|
||||
|
||||
ttl = cache_control.get("ttl")
|
||||
ttl = cache_control.get("ttl") if isinstance(cache_control, dict) else getattr(cache_control, "ttl", None)
|
||||
if ttl and _is_valid_ttl_format(ttl):
|
||||
return str(ttl)
|
||||
|
||||
|
|
|
|||
|
|
@ -892,26 +892,38 @@ class LiteLLMCompletionResponsesConfig:
|
|||
function_call=input_item
|
||||
)
|
||||
else:
|
||||
content = input_item.get("content")
|
||||
content = (
|
||||
input_item.get("content") if isinstance(input_item, dict) else getattr(input_item, "content", None)
|
||||
)
|
||||
# Handle None content: Responses API allows None content, but GenericChatCompletionMessage requires content
|
||||
# Since guardrails skip None content anyway, we return empty list to exclude it from structured messages
|
||||
if content is None:
|
||||
return []
|
||||
return [
|
||||
GenericChatCompletionMessage(
|
||||
role=input_item.get("role") or "user",
|
||||
content=LiteLLMCompletionResponsesConfig._transform_responses_api_content_to_chat_completion_content(
|
||||
content
|
||||
),
|
||||
)
|
||||
]
|
||||
|
||||
role = input_item.get("role") if isinstance(input_item, dict) else getattr(input_item, "role", None)
|
||||
cache_control = (
|
||||
input_item.get("cache_control")
|
||||
if isinstance(input_item, dict)
|
||||
else getattr(input_item, "cache_control", None)
|
||||
)
|
||||
|
||||
msg = GenericChatCompletionMessage(
|
||||
role=role or "user",
|
||||
content=LiteLLMCompletionResponsesConfig._transform_responses_api_content_to_chat_completion_content(
|
||||
content
|
||||
),
|
||||
)
|
||||
if cache_control is not None:
|
||||
msg["cache_control"] = cache_control
|
||||
return [msg]
|
||||
|
||||
@staticmethod
|
||||
def _is_input_item_tool_call_output(input_item: Any) -> bool:
|
||||
"""
|
||||
Check if the input item is a tool call output
|
||||
"""
|
||||
return input_item.get("type") in [
|
||||
val = input_item.get("type") if isinstance(input_item, dict) else getattr(input_item, "type", None)
|
||||
return val in [
|
||||
"function_call_output",
|
||||
"custom_tool_call_output",
|
||||
"web_search_call",
|
||||
|
|
@ -926,7 +938,8 @@ class LiteLLMCompletionResponsesConfig:
|
|||
Both need to be reconstructed as assistant tool_calls for Chat
|
||||
Completions providers.
|
||||
"""
|
||||
return input_item.get("type") in ("function_call", "custom_tool_call")
|
||||
val = input_item.get("type") if isinstance(input_item, dict) else getattr(input_item, "type", None)
|
||||
return val in ("function_call", "custom_tool_call")
|
||||
|
||||
@staticmethod
|
||||
def _transform_responses_api_tool_call_output_to_chat_completion_message(
|
||||
|
|
@ -1155,6 +1168,31 @@ class LiteLLMCompletionResponsesConfig:
|
|||
|
||||
return ChatCompletionImageObject(type="image_url", image_url=image_url_obj)
|
||||
|
||||
@staticmethod
|
||||
def _normalize_responses_api_object_to_dict(item: Any) -> dict[str, Any]:
|
||||
"""
|
||||
Normalize a Responses API object (Pydantic model or custom class) to a dictionary
|
||||
"""
|
||||
if hasattr(item, "model_dump"):
|
||||
return item.model_dump()
|
||||
elif hasattr(item, "dict"):
|
||||
return item.dict()
|
||||
|
||||
item_dict = {}
|
||||
for attr in [
|
||||
"type",
|
||||
"text",
|
||||
"cache_control",
|
||||
"file_id",
|
||||
"file_data",
|
||||
"file_url",
|
||||
"image_url",
|
||||
"detail",
|
||||
]:
|
||||
if hasattr(item, attr):
|
||||
item_dict[attr] = getattr(item, attr)
|
||||
return item_dict
|
||||
|
||||
@staticmethod
|
||||
def _transform_responses_api_content_to_chat_completion_content(
|
||||
content: Any,
|
||||
|
|
@ -1176,7 +1214,10 @@ class LiteLLMCompletionResponsesConfig:
|
|||
for item in content:
|
||||
if isinstance(item, str):
|
||||
content_list.append(item)
|
||||
elif isinstance(item, dict):
|
||||
elif isinstance(item, dict) or (item is not None and not isinstance(item, (str, int, float, bool))):
|
||||
if not isinstance(item, dict):
|
||||
item = LiteLLMCompletionResponsesConfig._normalize_responses_api_object_to_dict(item)
|
||||
|
||||
if item.get("type") == "input_file":
|
||||
content_list.append(
|
||||
LiteLLMCompletionResponsesConfig._transform_input_file_item_to_file_item(item)
|
||||
|
|
|
|||
|
|
@ -7209,36 +7209,51 @@ def is_cached_message(message: AllMessageValues) -> bool:
|
|||
return False
|
||||
|
||||
# Check message-level cache_control (set by cache_control_injection_points hook for string content)
|
||||
message_level_cache_control = message.get("cache_control")
|
||||
if (
|
||||
message_level_cache_control is not None
|
||||
and isinstance(message_level_cache_control, dict)
|
||||
and message_level_cache_control.get("type") == "ephemeral"
|
||||
):
|
||||
return True
|
||||
message_level_cache_control = (
|
||||
message.get("cache_control")
|
||||
if isinstance(message, dict)
|
||||
else getattr(message, "cache_control", None)
|
||||
)
|
||||
if message_level_cache_control is not None:
|
||||
cc_type = (
|
||||
message_level_cache_control.get("type")
|
||||
if isinstance(message_level_cache_control, dict)
|
||||
else getattr(message_level_cache_control, "type", None)
|
||||
)
|
||||
if cc_type == "ephemeral":
|
||||
return True
|
||||
|
||||
if "content" not in message:
|
||||
return False
|
||||
|
||||
content = message["content"]
|
||||
if isinstance(message, dict):
|
||||
if "content" not in message:
|
||||
return False
|
||||
content = message["content"]
|
||||
else:
|
||||
content = getattr(message, "content", None)
|
||||
|
||||
# Handle non-list content types (None, str, etc.)
|
||||
if not isinstance(content, list):
|
||||
return False
|
||||
|
||||
for content_item in content:
|
||||
# Ensure content_item is a dictionary before accessing keys
|
||||
if not isinstance(content_item, dict):
|
||||
continue
|
||||
# Check if content_item is dict or object model
|
||||
if isinstance(content_item, dict):
|
||||
cache_control = content_item.get("cache_control")
|
||||
item_type = content_item.get("type")
|
||||
else:
|
||||
cache_control = getattr(content_item, "cache_control", None)
|
||||
item_type = getattr(content_item, "type", None)
|
||||
|
||||
cache_control = content_item.get("cache_control")
|
||||
if (
|
||||
content_item.get("type") == "text"
|
||||
item_type == "text"
|
||||
and cache_control is not None
|
||||
and isinstance(cache_control, dict)
|
||||
and cache_control.get("type") == "ephemeral"
|
||||
):
|
||||
return True
|
||||
cc_type = (
|
||||
cache_control.get("type")
|
||||
if isinstance(cache_control, dict)
|
||||
else getattr(cache_control, "type", None)
|
||||
)
|
||||
if cc_type == "ephemeral":
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
Loading…
Add table
Reference in a new issue