fix(responses): propagate message cache_control safely through objects and models

This commit is contained in:
Elliott de Launay 2026-07-07 17:30:19 -04:00
parent 5abe5f82e1
commit 18101345a0
4 changed files with 348 additions and 351 deletions

View file

@ -62,23 +62,27 @@ def extract_ttl_from_cached_messages(messages: List[AllMessageValues]) -> Option
if not is_cached_message(message):
continue
content = message.get("content")
content = message.get("content") if isinstance(message, dict) else getattr(message, "content", None)
if not content or isinstance(content, str):
continue
for content_item in content:
# Type check to ensure content_item is a dictionary before calling .get()
if not isinstance(content_item, dict):
# Check if content_item is dict or object model
if isinstance(content_item, dict):
cache_control = content_item.get("cache_control")
else:
cache_control = getattr(content_item, "cache_control", None)
if not cache_control:
continue
cache_control = content_item.get("cache_control")
if not cache_control or not isinstance(cache_control, dict):
cc_type = (
cache_control.get("type") if isinstance(cache_control, dict) else getattr(cache_control, "type", None)
)
if cc_type != "ephemeral":
continue
if cache_control.get("type") != "ephemeral":
continue
ttl = cache_control.get("ttl")
ttl = cache_control.get("ttl") if isinstance(cache_control, dict) else getattr(cache_control, "ttl", None)
if ttl and _is_valid_ttl_format(ttl):
return str(ttl)

View file

@ -892,26 +892,38 @@ class LiteLLMCompletionResponsesConfig:
function_call=input_item
)
else:
content = input_item.get("content")
content = (
input_item.get("content") if isinstance(input_item, dict) else getattr(input_item, "content", None)
)
# Handle None content: Responses API allows None content, but GenericChatCompletionMessage requires content
# Since guardrails skip None content anyway, we return empty list to exclude it from structured messages
if content is None:
return []
return [
GenericChatCompletionMessage(
role=input_item.get("role") or "user",
content=LiteLLMCompletionResponsesConfig._transform_responses_api_content_to_chat_completion_content(
content
),
)
]
role = input_item.get("role") if isinstance(input_item, dict) else getattr(input_item, "role", None)
cache_control = (
input_item.get("cache_control")
if isinstance(input_item, dict)
else getattr(input_item, "cache_control", None)
)
msg = GenericChatCompletionMessage(
role=role or "user",
content=LiteLLMCompletionResponsesConfig._transform_responses_api_content_to_chat_completion_content(
content
),
)
if cache_control is not None:
msg["cache_control"] = cache_control
return [msg]
@staticmethod
def _is_input_item_tool_call_output(input_item: Any) -> bool:
"""
Check if the input item is a tool call output
"""
return input_item.get("type") in [
val = input_item.get("type") if isinstance(input_item, dict) else getattr(input_item, "type", None)
return val in [
"function_call_output",
"custom_tool_call_output",
"web_search_call",
@ -926,7 +938,8 @@ class LiteLLMCompletionResponsesConfig:
Both need to be reconstructed as assistant tool_calls for Chat
Completions providers.
"""
return input_item.get("type") in ("function_call", "custom_tool_call")
val = input_item.get("type") if isinstance(input_item, dict) else getattr(input_item, "type", None)
return val in ("function_call", "custom_tool_call")
@staticmethod
def _transform_responses_api_tool_call_output_to_chat_completion_message(
@ -1155,6 +1168,31 @@ class LiteLLMCompletionResponsesConfig:
return ChatCompletionImageObject(type="image_url", image_url=image_url_obj)
@staticmethod
def _normalize_responses_api_object_to_dict(item: Any) -> dict[str, Any]:
"""
Normalize a Responses API object (Pydantic model or custom class) to a dictionary
"""
if hasattr(item, "model_dump"):
return item.model_dump()
elif hasattr(item, "dict"):
return item.dict()
item_dict = {}
for attr in [
"type",
"text",
"cache_control",
"file_id",
"file_data",
"file_url",
"image_url",
"detail",
]:
if hasattr(item, attr):
item_dict[attr] = getattr(item, attr)
return item_dict
@staticmethod
def _transform_responses_api_content_to_chat_completion_content(
content: Any,
@ -1176,7 +1214,10 @@ class LiteLLMCompletionResponsesConfig:
for item in content:
if isinstance(item, str):
content_list.append(item)
elif isinstance(item, dict):
elif isinstance(item, dict) or (item is not None and not isinstance(item, (str, int, float, bool))):
if not isinstance(item, dict):
item = LiteLLMCompletionResponsesConfig._normalize_responses_api_object_to_dict(item)
if item.get("type") == "input_file":
content_list.append(
LiteLLMCompletionResponsesConfig._transform_input_file_item_to_file_item(item)

View file

@ -7209,36 +7209,51 @@ def is_cached_message(message: AllMessageValues) -> bool:
return False
# Check message-level cache_control (set by cache_control_injection_points hook for string content)
message_level_cache_control = message.get("cache_control")
if (
message_level_cache_control is not None
and isinstance(message_level_cache_control, dict)
and message_level_cache_control.get("type") == "ephemeral"
):
return True
message_level_cache_control = (
message.get("cache_control")
if isinstance(message, dict)
else getattr(message, "cache_control", None)
)
if message_level_cache_control is not None:
cc_type = (
message_level_cache_control.get("type")
if isinstance(message_level_cache_control, dict)
else getattr(message_level_cache_control, "type", None)
)
if cc_type == "ephemeral":
return True
if "content" not in message:
return False
content = message["content"]
if isinstance(message, dict):
if "content" not in message:
return False
content = message["content"]
else:
content = getattr(message, "content", None)
# Handle non-list content types (None, str, etc.)
if not isinstance(content, list):
return False
for content_item in content:
# Ensure content_item is a dictionary before accessing keys
if not isinstance(content_item, dict):
continue
# Check if content_item is dict or object model
if isinstance(content_item, dict):
cache_control = content_item.get("cache_control")
item_type = content_item.get("type")
else:
cache_control = getattr(content_item, "cache_control", None)
item_type = getattr(content_item, "type", None)
cache_control = content_item.get("cache_control")
if (
content_item.get("type") == "text"
item_type == "text"
and cache_control is not None
and isinstance(cache_control, dict)
and cache_control.get("type") == "ephemeral"
):
return True
cc_type = (
cache_control.get("type")
if isinstance(cache_control, dict)
else getattr(cache_control, "type", None)
)
if cc_type == "ephemeral":
return True
return False