From ccf03ee7dfa096a04d6106375478010e16a25543 Mon Sep 17 00:00:00 2001 From: yucheng Date: Fri, 2 Oct 2026 03:13:32 +0000 Subject: [PATCH] fix(responses): keep cache breakpoints out of non-bridge converter callers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../transformation.py | 28 +++-- ...responses_transformation_transformation.py | 115 ++++++++++++++++++ 2 files changed, 134 insertions(+), 9 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 367c5b31321..d69b2915067 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -393,6 +393,20 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): return None, index def convert_chat_completion_messages_to_responses_api( + self, + messages: list["AllMessageValues"], + *, + keep_prompt_cache_breakpoints: bool = False, + ) -> tuple[list[object], str | None]: + converted_input_items, instructions = self._convert_chat_completion_messages_to_responses_input(messages) + return ( + converted_input_items + if keep_prompt_cache_breakpoints + else _strip_prompt_cache_breakpoints(converted_input_items), + instructions, + ) + + def _convert_chat_completion_messages_to_responses_input( self, messages: list["AllMessageValues"] ) -> tuple[list[object], str | None]: input_items: Final[list[object]] = [] @@ -623,23 +637,19 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): litellm_logging_obj: "LiteLLMLoggingObj", client: object | None = None, ) -> dict: - converted_input_items, converted_instructions = self.convert_chat_completion_messages_to_responses_api(messages) base_model: Final = litellm_params.get("base_model") supports_prompt_cache_breakpoint: Final = supports_openai_prompt_cache_breakpoint(model) or ( isinstance(base_model, str) and bool(base_model) and supports_openai_prompt_cache_breakpoint(base_model) ) - input_items_without_unsupported_markers: Final = ( - converted_input_items - if supports_prompt_cache_breakpoint - else _strip_prompt_cache_breakpoints(converted_input_items) + converted_input_items, converted_instructions = self.convert_chat_completion_messages_to_responses_api( + messages, + keep_prompt_cache_breakpoints=supports_prompt_cache_breakpoint, ) # OpenAI's Responses API rejects an empty input. For a system-only # request, carry the system message as a system-role input item instead # of instructions, mirroring how non-string system content is already # handled in convert_chat_completion_messages_to_responses_api. - is_system_only_request: Final = ( - not input_items_without_unsupported_markers and converted_instructions is not None - ) + is_system_only_request: Final = not converted_input_items and converted_instructions is not None input_items: Final = ( [ { @@ -649,7 +659,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): } ] if is_system_only_request - else input_items_without_unsupported_markers + else converted_input_items ) instructions: Final = None if is_system_only_request else converted_instructions diff --git a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index a45a56a5226..fcb44958ffb 100644 --- a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -1,3 +1,4 @@ +import copy import datetime import json import os @@ -4333,6 +4334,120 @@ def test_prompt_cache_breakpoints_are_dropped_from_function_call_output_for_unsu ] +def test_convert_chat_completion_messages_to_responses_api_drops_prompt_cache_breakpoints_unless_kept() -> None: + handler: Final = LiteLLMResponsesTransformationHandler() + cache_breakpoint: Final = {"mode": "explicit"} + image_data_url: Final = "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg==" + file_data: Final = "data:application/pdf;base64,JVBERi0xLjQK" + messages: Final = cast( + list[AllMessageValues], + [ + { + "role": "user", + "content": [ + {"type": "text", "text": "Review these inputs", "prompt_cache_breakpoint": cache_breakpoint}, + { + "type": "image_url", + "image_url": {"url": image_data_url}, + "prompt_cache_breakpoint": cache_breakpoint, + }, + { + "type": "file", + "file": {"file_data": file_data, "filename": "input.pdf"}, + "prompt_cache_breakpoint": cache_breakpoint, + }, + ], + }, + { + "role": "assistant", + "tool_calls": [ + { + "id": "call_1", + "type": "function", + "function": {"name": "lookup", "arguments": "{}"}, + } + ], + }, + { + "role": "tool", + "tool_call_id": "call_1", + "content": [ + {"type": "text", "text": "Tool result", "prompt_cache_breakpoint": cache_breakpoint} + ], + }, + ], + ) + messages_before: Final = copy.deepcopy(messages) + + default_input, default_instructions = handler.convert_chat_completion_messages_to_responses_api(messages) + kept_input, kept_instructions = handler.convert_chat_completion_messages_to_responses_api( + messages, + keep_prompt_cache_breakpoints=True, + ) + + assert default_instructions is None + assert default_input == [ + { + "type": "message", + "role": "user", + "content": [ + {"type": "input_text", "text": "Review these inputs"}, + {"type": "input_image", "image_url": image_data_url, "detail": "auto"}, + {"type": "input_file", "file_data": file_data, "filename": "input.pdf"}, + ], + }, + { + "type": "function_call", + "call_id": "call_1", + "name": "lookup", + "arguments": "{}", + }, + { + "type": "function_call_output", + "call_id": "call_1", + "output": [{"type": "input_text", "text": "Tool result"}], + }, + ] + assert kept_instructions is None + assert kept_input == [ + { + "type": "message", + "role": "user", + "content": [ + { + "type": "input_text", + "text": "Review these inputs", + "prompt_cache_breakpoint": cache_breakpoint, + }, + { + "type": "input_image", + "image_url": image_data_url, + "detail": "auto", + "prompt_cache_breakpoint": cache_breakpoint, + }, + { + "type": "input_file", + "file_data": file_data, + "filename": "input.pdf", + "prompt_cache_breakpoint": cache_breakpoint, + }, + ], + }, + { + "type": "function_call", + "call_id": "call_1", + "name": "lookup", + "arguments": "{}", + }, + { + "type": "function_call_output", + "call_id": "call_1", + "output": [{"type": "input_text", "text": "Tool result", "prompt_cache_breakpoint": cache_breakpoint}], + }, + ] + assert messages == messages_before + + @pytest.mark.parametrize( ("litellm_params", "keep_marker"), (({"base_model": "gpt-5.6"}, True), ({}, False)),