From f57f994629c8450fab37abc8f7bbf15312e802ba Mon Sep 17 00:00:00 2001 From: Simon Sorg Date: Mon, 21 Sep 2026 21:06:52 +0200 Subject: [PATCH] fix(responses): preserve prompt cache breakpoints in chat bridge --- .../transformation.py | 6 ++- ...responses_transformation_transformation.py | 39 +++++++++++++++++++ 2 files changed, 44 insertions(+), 1 deletion(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 31af5a144eb..e2fe45102d4 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -26,6 +26,7 @@ from litellm import ModelResponse from litellm._logging import verbose_logger from litellm.litellm_core_utils.prompt_templates.common_utils import ( responses_reasoning_items_from_thinking_blocks, + with_prompt_cache_breakpoint, ) from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator from litellm.llms.base_llm.bridges.completion_transformation import ( @@ -1060,7 +1061,10 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): # Handle multimodal content original_type = item.get("type") if original_type == "text": - converted = self._convert_content_str_to_input_text(item.get("text", ""), role) + converted = with_prompt_cache_breakpoint( + self._convert_content_str_to_input_text(item.get("text", ""), role), + item.get("prompt_cache_breakpoint"), + ) result.append(converted) verbose_logger.debug("Chat provider: text -> %s", converted) elif original_type == "image_url": diff --git a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index d0f9bad795d..78aac112b2a 100644 --- a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -4185,6 +4185,45 @@ def _system_input_item(text: str) -> dict[str, object]: return {"type": "message", "role": "system", "content": [{"type": "input_text", "text": text}]} +def test_prompt_cache_breakpoint_survives_chat_to_responses_conversion() -> None: + handler: Final = LiteLLMResponsesTransformationHandler() + cache_breakpoint: Final = {"mode": "explicit"} + + request: Final = handler.transform_request( + model="gpt-5.6-sol", + messages=[ + { + "role": "system", + "content": [ + { + "type": "text", + "text": "Stable prefix", + "prompt_cache_breakpoint": cache_breakpoint, + } + ], + }, + {"role": "user", "content": "Use a tool"}, + ], + optional_params={"prompt_cache_options": cache_breakpoint}, + litellm_params={}, + headers={}, + litellm_logging_obj=Mock(), + ) + + assert request["input"][0] == { + "type": "message", + "role": "system", + "content": [ + { + "type": "input_text", + "text": "Stable prefix", + "prompt_cache_breakpoint": cache_breakpoint, + } + ], + } + assert request["prompt_cache_options"] == cache_breakpoint + + def test_mid_conversation_system_string_stays_in_input_after_a_user_turn(): handler: Final = LiteLLMResponsesTransformationHandler()