From 0c630cef3b464e0d93fe070b41ec14a83a48bd9c Mon Sep 17 00:00:00 2001 From: Tan Nguyen Date: Wed, 23 Sep 2026 11:13:54 +0700 Subject: [PATCH 1/7] fix(anthropic): write the cache_control marker on a block that accepts one The marker went on the last content block whatever it was. Anthropic accepts one on text, image, tool_use, tool_result and document blocks; a thinking block is not among them, and an empty text block is replaced by a placeholder before the request goes out, taking the marker with it. Either way the configured breakpoint is spent and the turn is not cached. Walk back to the last block that accepts a marker, the way the OpenAI prompt cache path in this file already does. A tool message is the exception: its empty text block is kept, nested in the tool_result the conversion builds, so the marker stays on it. --- .../anthropic_cache_control_hook.py | 35 +++- .../test_anthropic_cache_control_hook.py | 160 ++++++++++++++++++ 2 files changed, 192 insertions(+), 3 deletions(-) diff --git a/litellm/integrations/anthropic_cache_control_hook.py b/litellm/integrations/anthropic_cache_control_hook.py index 1db144b5fdc..4ab3e0d35cf 100644 --- a/litellm/integrations/anthropic_cache_control_hook.py +++ b/litellm/integrations/anthropic_cache_control_hook.py @@ -67,6 +67,10 @@ _GPT_VERSION_PATTERN: Final = re.compile(r"^gpt-(\d+)(?:\.(\d+))?") OPENAI_PROMPT_CACHE_BREAKPOINT_BLOCK_TYPES: Final = frozenset( {"text", "image", "image_url", "file", "input_audio", "input_text", "input_image", "input_file"} ) +# Anthropic lists the block types a cache_control marker may sit on: text, image, +# tool_use, tool_result and document. A list of refused types rather than accepted +# ones so a block type this code has not been told about still takes a marker. +ANTHROPIC_BLOCK_TYPES_WITHOUT_CACHE_CONTROL: Final = frozenset({"thinking", "redacted_thinking"}) OPENAI_API_HOST: Final = "api.openai.com" OPENAI_API_BASE_ENV_VARS: Final = ("OPENAI_BASE_URL", "OPENAI_API_BASE") _OBJECT_MAPPING_ADAPTER: Final = TypeAdapter(dict[object, object]) @@ -139,6 +143,28 @@ def _accepts_prompt_cache_breakpoint(block: object) -> bool: return isinstance(block, dict) and block.get("type") in OPENAI_PROMPT_CACHE_BREAKPOINT_BLOCK_TYPES +def _index_of_block_accepting_cache_control(content: list[object], on_a_tool_message: bool) -> int | None: + """Position of the last block a cache_control marker can be written on, or None. + + Searched from the end: a marker caches everything up to and including its own + block, so the last one caches the most. + + An empty text block is refused because the rewrites this hook runs ahead of drop + it, and the marker goes with it. ``on_a_tool_message`` lifts that refusal: a tool + message keeps its empty block, nested in the tool_result the rewrite builds. + """ + for index in range(len(content) - 1, -1, -1): + block = content[index] + if not isinstance(block, dict): + continue + if block.get("type") in ANTHROPIC_BLOCK_TYPES_WITHOUT_CACHE_CONTROL: + continue + if block.get("type") == "text" and not block.get("text") and not on_a_tool_message: + continue + return index + return None + + # Set by a caller whose message list is not the one that goes upstream -- today the # Responses API layer, whose `instructions` only becomes a system message further down. # Tells this hook to hand role-targeted points to the pass holding the final messages @@ -507,10 +533,13 @@ class AnthropicCacheControlHook(CustomPromptManagement): # 1. if string, insert cache control in the message if isinstance(message_content, str): message["cache_control"] = control - # 2. list of objects - only apply to last item per Anthropic spec + # 2. list of objects - the last block that accepts a marker, per Anthropic spec elif isinstance(message_content, list): - if len(message_content) > 0 and isinstance(message_content[-1], dict): - message_content[-1]["cache_control"] = control # pyright: ignore[reportGeneralTypeIssues] # loose runtime dict + target_index: Final = _index_of_block_accepting_cache_control( + message_content, on_a_tool_message=message.get("role") == "tool" + ) + if target_index is not None: + message_content[target_index]["cache_control"] = control # pyright: ignore[reportGeneralTypeIssues] # loose runtime dict return message @staticmethod diff --git a/tests/unit/integrations/test_anthropic_cache_control_hook.py b/tests/unit/integrations/test_anthropic_cache_control_hook.py index 1d70a21af7b..3adf0e18640 100644 --- a/tests/unit/integrations/test_anthropic_cache_control_hook.py +++ b/tests/unit/integrations/test_anthropic_cache_control_hook.py @@ -3642,3 +3642,163 @@ class TestRecordGatewayInjection: custom_llm_provider="anthropic", ) assert self.KEY not in kwargs["litellm_metadata"] + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_skips_a_thinking_block(monkeypatch: pytest.MonkeyPatch): + """ + A cache_control marker on a thinking block is spent on a block Anthropic does not + accept one on, so the turn it was meant to cache is not cached. The marker goes on + the last block of the message that accepts one instead. + """ + monkeypatch.setenv("ANTHROPIC_API_KEY", "fake_anthropic_key") + anthropic_cache_control_hook = AnthropicCacheControlHook() + monkeypatch.setattr(litellm, "callbacks", [anthropic_cache_control_hook]) + + mock_response = MagicMock() + mock_response.json.return_value = { + "id": "msg_01", + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5", + "content": [{"type": "text", "text": "Because two plus two is four."}], + "stop_reason": "end_turn", + "usage": {"input_tokens": 10, "output_tokens": 20}, + } + mock_response.status_code = 200 + + client = AsyncHTTPHandler() + with patch.object(client, "post", return_value=mock_response) as mock_post: + await litellm.acompletion( + model="anthropic/claude-sonnet-4-5", + messages=[ + {"role": "user", "content": "What is 2 + 2?"}, + { + "role": "assistant", + "content": [ + {"type": "text", "text": "The answer is 4."}, + {"type": "thinking", "thinking": "Adding two and two.", "signature": "sig"}, + {"type": "redacted_thinking", "data": "redacted"}, + ], + }, + {"role": "user", "content": "Why?"}, + ], + cache_control_injection_points=[{"location": "message", "index": 1}], + client=client, + ) + + request_body = mock_post.call_args.kwargs["json"] + + assert request_body["messages"][1] == { + "role": "assistant", + "content": [ + {"type": "text", "text": "The answer is 4.", "cache_control": {"type": "ephemeral"}}, + {"type": "thinking", "thinking": "Adding two and two.", "signature": "sig"}, + ], + } + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_skips_an_empty_text_block(monkeypatch: pytest.MonkeyPatch): + """ + An empty text block is replaced by a placeholder before the request goes out, and a + cache_control marker written on it is replaced with it. The marker goes on the last + block that survives instead. + """ + monkeypatch.setenv("ANTHROPIC_API_KEY", "fake_anthropic_key") + monkeypatch.setattr(litellm, "callbacks", [AnthropicCacheControlHook()]) + + mock_response = MagicMock() + mock_response.json.return_value = { + "id": "msg_01", + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5", + "content": [{"type": "text", "text": "Because two plus two is four."}], + "stop_reason": "end_turn", + "usage": {"input_tokens": 10, "output_tokens": 20}, + } + mock_response.status_code = 200 + + client = AsyncHTTPHandler() + with patch.object(client, "post", return_value=mock_response) as mock_post: + await litellm.acompletion( + model="anthropic/claude-sonnet-4-5", + messages=[ + {"role": "user", "content": "What is 2 + 2?"}, + { + "role": "assistant", + "content": [ + {"type": "text", "text": "The answer is 4."}, + {"type": "text", "text": ""}, + ], + }, + {"role": "user", "content": "Why?"}, + ], + cache_control_injection_points=[{"location": "message", "index": 1}], + client=client, + ) + + request_body = mock_post.call_args.kwargs["json"] + + assert request_body["messages"][1] == { + "role": "assistant", + "content": [ + {"type": "text", "text": "The answer is 4.", "cache_control": {"type": "ephemeral"}}, + {"type": "text", "text": "[System: Empty message content sanitised to satisfy protocol]"}, + ], + } + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_marks_an_empty_tool_result(monkeypatch: pytest.MonkeyPatch): + """ + A tool message keeps its empty text block - it goes out nested in the tool_result the + conversion builds - so the marker stays on it rather than walking off the message. + """ + monkeypatch.setenv("ANTHROPIC_API_KEY", "fake_anthropic_key") + monkeypatch.setattr(litellm, "callbacks", [AnthropicCacheControlHook()]) + + mock_response = MagicMock() + mock_response.json.return_value = { + "id": "msg_01", + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5", + "content": [{"type": "text", "text": "Nothing came back."}], + "stop_reason": "end_turn", + "usage": {"input_tokens": 10, "output_tokens": 20}, + } + mock_response.status_code = 200 + + client = AsyncHTTPHandler() + with patch.object(client, "post", return_value=mock_response) as mock_post: + await litellm.acompletion( + model="anthropic/claude-sonnet-4-5", + messages=[ + {"role": "user", "content": "Search for it."}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + {"id": "call_1", "type": "function", "function": {"name": "search", "arguments": "{}"}} + ], + }, + {"role": "tool", "tool_call_id": "call_1", "content": [{"type": "text", "text": ""}]}, + ], + cache_control_injection_points=[{"location": "message", "index": 2}], + client=client, + ) + + request_body = mock_post.call_args.kwargs["json"] + + assert request_body["messages"][-1] == { + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "call_1", + "content": [{"type": "text", "text": "", "cache_control": {"type": "ephemeral"}}], + } + ], + } From 03e6a97639a4e81573d5ede0f73552cfda2ff785 Mon Sep 17 00:00:00 2001 From: Tan Nguyen Date: Wed, 23 Sep 2026 12:31:31 +0700 Subject: [PATCH 2/7] fix(anthropic): count an injection point's index within the role it names {"role": "assistant", "index": -1} read as "the last message", because the index branch returned before the role was looked at. On a conversation ending with a user turn it marked that turn; the assistant turn the point named went uncached. The message it lands on may also have nowhere to write a marker. An assistant turn that said everything through tool_calls has no content, and it is the newest assistant turn at every step of an agent loop, so the point was spent and nothing was written. Count the index within the role's turns, and walk back from there to the nearest earlier turn that accepts a marker. The walk stays inside the role the point named. --- .../anthropic_cache_control_hook.py | 63 ++++- .../test_anthropic_cache_control_hook.py | 259 ++++++++++++++++++ 2 files changed, 309 insertions(+), 13 deletions(-) diff --git a/litellm/integrations/anthropic_cache_control_hook.py b/litellm/integrations/anthropic_cache_control_hook.py index 4ab3e0d35cf..c54b42bf0ca 100644 --- a/litellm/integrations/anthropic_cache_control_hook.py +++ b/litellm/integrations/anthropic_cache_control_hook.py @@ -165,6 +165,25 @@ def _index_of_block_accepting_cache_control(content: list[object], on_a_tool_mes return None +def _message_accepts_cache_control(message: object) -> bool: + """Whether a cache_control marker written on this message reaches the provider. + + Reads the same block rule as `_safe_insert_cache_control_in_message`, so the two + cannot disagree about where a marker goes. A message whose content is empty -- + an assistant turn that said everything through ``tool_calls`` -- has nowhere to + put one. + """ + if not isinstance(message, dict): + return False + on_a_tool_message: Final = message.get("role") == "tool" + content: Final = message.get("content") + if isinstance(content, str): + return content != "" or on_a_tool_message + if isinstance(content, list): + return _index_of_block_accepting_cache_control(content, on_a_tool_message) is not None + return False + + # Set by a caller whose message list is not the one that goes upstream -- today the # Responses API layer, whose `instructions` only becomes a system message further down. # Tells this hook to hand role-targeted points to the pass holding the final messages @@ -471,28 +490,46 @@ class AnthropicCacheControlHook(CustomPromptManagement): else: targetted_index = _targetted_index - # Case 1: Target by specific index - if targetted_index is not None: - original_index: Final = targetted_index - if targetted_index < 0: - targetted_index += len(messages) + # The messages the point names. A point with a role counts its index within + # that role's turns, so {role: assistant, index: -1} is the last assistant + # turn rather than the last message. + targetted_role: Final = point.get("role", None) + candidates: Final = [ + index + for index, message in enumerate(messages) + if targetted_role is None or message.get("role") == targetted_role + ] - if 0 <= targetted_index < len(messages): - return [targetted_index] + # Case 1: Target by role alone + if targetted_index is None: + if targetted_role is None: + return [] + return [index for index in candidates if _message_accepts_cache_control(messages[index])] + # Case 2: Target by index, within the role the point named + original_index: Final = targetted_index + if targetted_index < 0: + targetted_index += len(candidates) + + if not 0 <= targetted_index < len(candidates): verbose_logger.warning( "AnthropicCacheControlHook: Provided index %s is out of bounds for message list of length %s. Targeted index was %s. Skipping cache control injection for this point.", original_index, - len(messages), + len(candidates), targetted_index, ) return [] - # Case 2: Target by role - targetted_role: Final = point.get("role", None) - if targetted_role is not None: - return [idx for idx, msg in enumerate(messages) if msg.get("role") == targetted_role] - + # The message it arrives at may have nowhere to write a marker -- an assistant + # turn that said everything through tool_calls is the common one -- so walk back + # to the nearest earlier turn that does. Stopping there would spend the point + # and write nothing. + position = targetted_index + while position >= 0: + index = candidates[position] + if _message_accepts_cache_control(messages[index]): + return [index] + position -= 1 return [] @staticmethod diff --git a/tests/unit/integrations/test_anthropic_cache_control_hook.py b/tests/unit/integrations/test_anthropic_cache_control_hook.py index 3adf0e18640..c9af51e0392 100644 --- a/tests/unit/integrations/test_anthropic_cache_control_hook.py +++ b/tests/unit/integrations/test_anthropic_cache_control_hook.py @@ -3802,3 +3802,262 @@ async def test_anthropic_cache_control_hook_marks_an_empty_tool_result(monkeypat } ], } + + +def _anthropic_response_mock() -> MagicMock: + mock_response = MagicMock() + mock_response.json.return_value = { + "id": "msg_01", + "type": "message", + "role": "assistant", + "model": "claude-sonnet-4-5", + "content": [{"type": "text", "text": "Sure."}], + "stop_reason": "end_turn", + "usage": {"input_tokens": 10, "output_tokens": 20}, + } + mock_response.status_code = 200 + return mock_response + + +async def _marked_messages(messages, points, monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("ANTHROPIC_API_KEY", "fake_anthropic_key") + monkeypatch.setattr(litellm, "callbacks", [AnthropicCacheControlHook()]) + client = AsyncHTTPHandler() + with patch.object(client, "post", return_value=_anthropic_response_mock()) as mock_post: + await litellm.acompletion( + model="anthropic/claude-sonnet-4-5", + messages=messages, + cache_control_injection_points=points, + client=client, + ) + return mock_post.call_args.kwargs["json"]["messages"] + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_counts_the_index_within_the_role(monkeypatch: pytest.MonkeyPatch): + """ + {"role": "assistant", "index": -1} means the last assistant turn. Counting the index + over every message instead marks whatever message happens to be last, so the turn the + point named is not cached. + """ + marked = await _marked_messages( + [ + {"role": "user", "content": "What is 2 + 2?"}, + {"role": "assistant", "content": "The answer is 4."}, + {"role": "user", "content": "Why?"}, + ], + [{"location": "message", "role": "assistant", "index": -1}], + monkeypatch, + ) + + assert marked == [ + {"role": "user", "content": [{"type": "text", "text": "What is 2 + 2?"}]}, + { + "role": "assistant", + "content": [{"type": "text", "text": "The answer is 4.", "cache_control": {"type": "ephemeral"}}], + }, + {"role": "user", "content": [{"type": "text", "text": "Why?"}]}, + ] + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_walks_back_off_a_tool_call_turn(monkeypatch: pytest.MonkeyPatch): + """ + An assistant turn that said everything through tool_calls has no content to mark, and + it is the newest assistant turn at every step of an agent loop. Stopping there spends + the point and writes nothing, so the walk goes back to the turn before it. + """ + marked = await _marked_messages( + [ + {"role": "user", "content": "Open the file."}, + {"role": "assistant", "content": "Opening it now."}, + {"role": "user", "content": "Thanks."}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + {"id": "call_1", "type": "function", "function": {"name": "open_file", "arguments": "{}"}} + ], + }, + {"role": "tool", "tool_call_id": "call_1", "content": "file contents"}, + ], + [{"location": "message", "role": "assistant", "index": -1}], + monkeypatch, + ) + + assert marked[1] == { + "role": "assistant", + "content": [{"type": "text", "text": "Opening it now.", "cache_control": {"type": "ephemeral"}}], + } + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_walk_back_stays_in_the_role(monkeypatch: pytest.MonkeyPatch): + """ + The walk stays inside the role the point named, so an empty last user turn marks an + earlier user turn rather than the assistant turn between them. + """ + marked = await _marked_messages( + [ + {"role": "user", "content": "first question"}, + {"role": "assistant", "content": "first answer"}, + {"role": "user", "content": ""}, + ], + [{"location": "message", "role": "user", "index": -1}], + monkeypatch, + ) + + assert marked[0] == { + "role": "user", + "content": [{"type": "text", "text": "first question", "cache_control": {"type": "ephemeral"}}], + } + assert marked[1] == {"role": "assistant", "content": [{"type": "text", "text": "first answer"}]} + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_bounds_the_index_by_the_role(monkeypatch: pytest.MonkeyPatch): + """ + An index is in bounds when the role has that many turns, not when the message list + does. -3 against two assistant turns names no turn, so the point marks nothing. + """ + marked = await _marked_messages( + [ + {"role": "user", "content": "u1"}, + {"role": "assistant", "content": "a1"}, + {"role": "user", "content": "u2"}, + {"role": "assistant", "content": "a2"}, + ], + [{"location": "message", "role": "assistant", "index": -3}], + monkeypatch, + ) + + assert marked == [ + {"role": "user", "content": [{"type": "text", "text": "u1"}]}, + {"role": "assistant", "content": [{"type": "text", "text": "a1"}]}, + {"role": "user", "content": [{"type": "text", "text": "u2"}]}, + {"role": "assistant", "content": [{"type": "text", "text": "a2"}]}, + ] + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_bounds_a_positive_index_by_the_role(monkeypatch: pytest.MonkeyPatch): + """ + Index 3 names a fourth turn of the role. Two assistant turns among five messages is + out of bounds, and bounding it by the message list instead reads past the end of the + role's turns. + """ + marked = await _marked_messages( + [ + {"role": "user", "content": "u1"}, + {"role": "assistant", "content": "a1"}, + {"role": "user", "content": "u2"}, + {"role": "assistant", "content": "a2"}, + {"role": "user", "content": "u3"}, + ], + [{"location": "message", "role": "assistant", "index": 3}], + monkeypatch, + ) + + assert marked == [ + {"role": "user", "content": [{"type": "text", "text": "u1"}]}, + {"role": "assistant", "content": [{"type": "text", "text": "a1"}]}, + {"role": "user", "content": [{"type": "text", "text": "u2"}]}, + {"role": "assistant", "content": [{"type": "text", "text": "a2"}]}, + {"role": "user", "content": [{"type": "text", "text": "u3"}]}, + ] + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_role_alone_skips_a_turn_with_nowhere_to_write( + monkeypatch: pytest.MonkeyPatch, +): + """ + A point naming a role and no index marks every turn of that role that accepts a + marker. An empty turn would spend a breakpoint on a block the conversion replaces. + """ + marked = await _marked_messages( + [ + {"role": "user", "content": "u1"}, + {"role": "assistant", "content": "a1"}, + {"role": "user", "content": ""}, + {"role": "assistant", "content": "a2"}, + {"role": "user", "content": "u3"}, + ], + [{"location": "message", "role": "user"}], + monkeypatch, + ) + + assert marked[0] == { + "role": "user", + "content": [{"type": "text", "text": "u1", "cache_control": {"type": "ephemeral"}}], + } + assert marked[2] == { + "role": "user", + "content": [{"type": "text", "text": "[System: Empty message content sanitised to satisfy protocol]"}], + } + assert marked[4] == { + "role": "user", + "content": [{"type": "text", "text": "u3", "cache_control": {"type": "ephemeral"}}], + } + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_marks_a_tool_turn_that_returned_nothing( + monkeypatch: pytest.MonkeyPatch, +): + """ + A tool that returned nothing is the newest turn of an agent loop. Its empty content + reaches the provider inside the tool_result, so the marker goes on it rather than + walking back and leaving the tool_use outside the cached prefix. + """ + marked = await _marked_messages( + [ + {"role": "user", "content": "Search for it."}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + {"id": "call_1", "type": "function", "function": {"name": "search", "arguments": "{}"}} + ], + }, + {"role": "tool", "tool_call_id": "call_1", "content": ""}, + ], + [{"location": "message", "index": 2}], + monkeypatch, + ) + + assert marked[-1] == { + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "call_1", + "content": "", + "cache_control": {"type": "ephemeral"}, + } + ], + } + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_walks_back_off_a_thinking_only_turn(monkeypatch: pytest.MonkeyPatch): + """ + A turn whose only block is a thinking block has no block that accepts a marker, so + the walk goes back to the turn before it rather than spending the point there. + """ + marked = await _marked_messages( + [ + {"role": "user", "content": "q1"}, + {"role": "assistant", "content": "a1"}, + {"role": "user", "content": "q2"}, + {"role": "assistant", "content": [{"type": "thinking", "thinking": "t", "signature": "s"}]}, + ], + [{"location": "message", "role": "assistant", "index": -1}], + monkeypatch, + ) + + assert marked[1] == { + "role": "assistant", + "content": [{"type": "text", "text": "a1", "cache_control": {"type": "ephemeral"}}], + } + assert marked[3] == {"role": "assistant", "content": [{"type": "thinking", "thinking": "t", "signature": "s"}]} From b6a22917d99b56847c91cabd097bb6d759cff4a0 Mon Sep 17 00:00:00 2001 From: Tan Nguyen Date: Wed, 23 Sep 2026 12:40:20 +0700 Subject: [PATCH 3/7] fix(anthropic): keep two injection points as two cache breakpoints The second point resolved to the message the first one had just marked, found it marked and was dropped. Two configured breakpoints became one, and the four exist so that a prefix which stops matching at one can still match at an earlier one. Carry the messages already marked in this request into the walk, so a point that arrives on one keeps going back instead of being spent there. --- .../anthropic_cache_control_hook.py | 19 ++++++++++--- .../test_anthropic_cache_control_hook.py | 27 +++++++++++++++++++ 2 files changed, 42 insertions(+), 4 deletions(-) diff --git a/litellm/integrations/anthropic_cache_control_hook.py b/litellm/integrations/anthropic_cache_control_hook.py index c54b42bf0ca..9e95678af93 100644 --- a/litellm/integrations/anthropic_cache_control_hook.py +++ b/litellm/integrations/anthropic_cache_control_hook.py @@ -439,6 +439,7 @@ class AnthropicCacheControlHook(CustomPromptManagement): """ used_blocks = AnthropicCacheControlHook.count_request_cache_breakpoints(messages) + taken: set[int] = set() limit_reached = False for point in points: if used_blocks >= max_blocks: @@ -449,7 +450,9 @@ class AnthropicCacheControlHook(CustomPromptManagement): type="ephemeral" ) - for target_index in AnthropicCacheControlHook._resolve_target_indices(point=point, messages=messages): + for target_index in AnthropicCacheControlHook._resolve_target_indices( + point=point, messages=messages, taken=taken + ): if used_blocks >= max_blocks: limit_reached = True break @@ -463,6 +466,7 @@ class AnthropicCacheControlHook(CustomPromptManagement): ) if AnthropicCacheControlHook._message_has_cache_control(messages[target_index]): used_blocks += 1 + taken.add(target_index) if limit_reached: break @@ -477,9 +481,16 @@ class AnthropicCacheControlHook(CustomPromptManagement): @staticmethod def _resolve_target_indices( - point: CacheControlMessageInjectionPoint, messages: list[AllMessageValues] + point: CacheControlMessageInjectionPoint, messages: list[AllMessageValues], taken: set[int] | None = None ) -> list[int]: - """Resolve which message indices an injection point targets.""" + """Resolve which message indices an injection point targets. + + ``taken`` is the messages an earlier point in the same request already marked. + The walk goes past them: arriving on one, finding it marked and dropping the + point turns two configured breakpoints into one, and the four exist so that a + prefix which stops matching at one can still match at an earlier one. + """ + already_marked: Final = taken or set() _targetted_index: Final[int | str | None] = point.get("index", None) targetted_index: int | None = None if isinstance(_targetted_index, str): @@ -527,7 +538,7 @@ class AnthropicCacheControlHook(CustomPromptManagement): position = targetted_index while position >= 0: index = candidates[position] - if _message_accepts_cache_control(messages[index]): + if index not in already_marked and _message_accepts_cache_control(messages[index]): return [index] position -= 1 return [] diff --git a/tests/unit/integrations/test_anthropic_cache_control_hook.py b/tests/unit/integrations/test_anthropic_cache_control_hook.py index c9af51e0392..ed0ef328fe4 100644 --- a/tests/unit/integrations/test_anthropic_cache_control_hook.py +++ b/tests/unit/integrations/test_anthropic_cache_control_hook.py @@ -4061,3 +4061,30 @@ async def test_anthropic_cache_control_hook_walks_back_off_a_thinking_only_turn( "content": [{"type": "text", "text": "a1", "cache_control": {"type": "ephemeral"}}], } assert marked[3] == {"role": "assistant", "content": [{"type": "thinking", "thinking": "t", "signature": "s"}]} + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_two_points_stay_two_breakpoints(monkeypatch: pytest.MonkeyPatch): + """ + The second point walks past the message the first one marked. Arriving on it, + finding it marked and dropping the point turns two configured breakpoints into one, + and the four exist so a prefix that stops matching at one can match at an earlier one. + """ + marked = await _marked_messages( + [ + {"role": "user", "content": "open"}, + {"role": "assistant", "content": "ack"}, + {"role": "user", "content": ""}, + ], + [{"location": "message", "index": -1}, {"location": "message", "index": -2}], + monkeypatch, + ) + + assert marked == [ + {"role": "user", "content": [{"type": "text", "text": "open", "cache_control": {"type": "ephemeral"}}]}, + {"role": "assistant", "content": [{"type": "text", "text": "ack", "cache_control": {"type": "ephemeral"}}]}, + { + "role": "user", + "content": [{"type": "text", "text": "[System: Empty message content sanitised to satisfy protocol]"}], + }, + ] From d45077ea824b34e92bbc80aa713f370071c3001b Mon Sep 17 00:00:00 2001 From: Tan Nguyen Date: Wed, 23 Sep 2026 13:39:11 +0700 Subject: [PATCH 4/7] fix(anthropic): keep the type-discipline budget flat Build the candidate indices as a tuple, take the already-marked messages as a Container with a frozenset default, and say why the one remaining set is mutable. --- .../integrations/anthropic_cache_control_hook.py | 13 ++++++------- 1 file changed, 6 insertions(+), 7 deletions(-) diff --git a/litellm/integrations/anthropic_cache_control_hook.py b/litellm/integrations/anthropic_cache_control_hook.py index 9e95678af93..24932591f70 100644 --- a/litellm/integrations/anthropic_cache_control_hook.py +++ b/litellm/integrations/anthropic_cache_control_hook.py @@ -12,7 +12,7 @@ Supported for both `v1/chat/completions` (via the prompt-management hook) and import copy import os import re -from collections.abc import Iterable, Mapping, Sequence +from collections.abc import Container, Iterable, Mapping, Sequence from typing import TYPE_CHECKING, Any, Final, cast from urllib.parse import urlparse @@ -439,7 +439,7 @@ class AnthropicCacheControlHook(CustomPromptManagement): """ used_blocks = AnthropicCacheControlHook.count_request_cache_breakpoints(messages) - taken: set[int] = set() + taken: set[int] = set() # mutable-ok: the messages this request has already marked limit_reached = False for point in points: if used_blocks >= max_blocks: @@ -481,7 +481,7 @@ class AnthropicCacheControlHook(CustomPromptManagement): @staticmethod def _resolve_target_indices( - point: CacheControlMessageInjectionPoint, messages: list[AllMessageValues], taken: set[int] | None = None + point: CacheControlMessageInjectionPoint, messages: list[AllMessageValues], taken: Container[int] = frozenset() ) -> list[int]: """Resolve which message indices an injection point targets. @@ -490,7 +490,6 @@ class AnthropicCacheControlHook(CustomPromptManagement): point turns two configured breakpoints into one, and the four exist so that a prefix which stops matching at one can still match at an earlier one. """ - already_marked: Final = taken or set() _targetted_index: Final[int | str | None] = point.get("index", None) targetted_index: int | None = None if isinstance(_targetted_index, str): @@ -505,11 +504,11 @@ class AnthropicCacheControlHook(CustomPromptManagement): # that role's turns, so {role: assistant, index: -1} is the last assistant # turn rather than the last message. targetted_role: Final = point.get("role", None) - candidates: Final = [ + candidates: Final = tuple( index for index, message in enumerate(messages) if targetted_role is None or message.get("role") == targetted_role - ] + ) # Case 1: Target by role alone if targetted_index is None: @@ -538,7 +537,7 @@ class AnthropicCacheControlHook(CustomPromptManagement): position = targetted_index while position >= 0: index = candidates[position] - if index not in already_marked and _message_accepts_cache_control(messages[index]): + if index not in taken and _message_accepts_cache_control(messages[index]): return [index] position -= 1 return [] From 464ee4ab2699a6fd50e3d34c6dfbc86c8fa0a96c Mon Sep 17 00:00:00 2001 From: Tan Nguyen Date: Wed, 23 Sep 2026 14:19:15 +0700 Subject: [PATCH 5/7] fix(anthropic): refuse a marker on a tool_reference block, and drop the mutable state The tool_result conversion rebuilds a tool_reference from its type and tool_name alone, so a marker written on one is dropped before the request goes out. Refuse that block type the way thinking blocks are refused. Resolve an injection point against the messages that carry no breakpoint yet instead of carrying a mutable set of the ones this request marked, so the walk needs no reassignment and the caller's own markers are walked past too. --- .../anthropic_cache_control_hook.py | 85 ++++++------------- .../test_anthropic_cache_control_hook.py | 51 ++++++++++- 2 files changed, 78 insertions(+), 58 deletions(-) diff --git a/litellm/integrations/anthropic_cache_control_hook.py b/litellm/integrations/anthropic_cache_control_hook.py index 24932591f70..733937039c5 100644 --- a/litellm/integrations/anthropic_cache_control_hook.py +++ b/litellm/integrations/anthropic_cache_control_hook.py @@ -12,7 +12,7 @@ Supported for both `v1/chat/completions` (via the prompt-management hook) and import copy import os import re -from collections.abc import Container, Iterable, Mapping, Sequence +from collections.abc import Iterable, Mapping, Sequence from typing import TYPE_CHECKING, Any, Final, cast from urllib.parse import urlparse @@ -67,10 +67,12 @@ _GPT_VERSION_PATTERN: Final = re.compile(r"^gpt-(\d+)(?:\.(\d+))?") OPENAI_PROMPT_CACHE_BREAKPOINT_BLOCK_TYPES: Final = frozenset( {"text", "image", "image_url", "file", "input_audio", "input_text", "input_image", "input_file"} ) -# Anthropic lists the block types a cache_control marker may sit on: text, image, -# tool_use, tool_result and document. A list of refused types rather than accepted -# ones so a block type this code has not been told about still takes a marker. -ANTHROPIC_BLOCK_TYPES_WITHOUT_CACHE_CONTROL: Final = frozenset({"thinking", "redacted_thinking"}) +# Block types a marker never reaches the provider on. Anthropic accepts one on text, +# image, tool_use, tool_result and document, so a thinking block is refused there; a +# tool_reference is rebuilt from its type and tool_name alone by the tool_result +# conversion, which drops everything else. A list of refused types rather than +# accepted ones, so a block type this code has not been told about still takes one. +ANTHROPIC_BLOCK_TYPES_WITHOUT_CACHE_CONTROL: Final = frozenset({"thinking", "redacted_thinking", "tool_reference"}) OPENAI_API_HOST: Final = "api.openai.com" OPENAI_API_BASE_ENV_VARS: Final = ("OPENAI_BASE_URL", "OPENAI_API_BASE") _OBJECT_MAPPING_ADAPTER: Final = TypeAdapter(dict[object, object]) @@ -144,14 +146,10 @@ def _accepts_prompt_cache_breakpoint(block: object) -> bool: def _index_of_block_accepting_cache_control(content: list[object], on_a_tool_message: bool) -> int | None: - """Position of the last block a cache_control marker can be written on, or None. + """Last block a marker written here reaches the provider on, searched from the end. - Searched from the end: a marker caches everything up to and including its own - block, so the last one caches the most. - - An empty text block is refused because the rewrites this hook runs ahead of drop - it, and the marker goes with it. ``on_a_tool_message`` lifts that refusal: a tool - message keeps its empty block, nested in the tool_result the rewrite builds. + An empty text block is replaced by a placeholder unless it sits on a tool message, + where it goes out inside the tool_result. """ for index in range(len(content) - 1, -1, -1): block = content[index] @@ -166,13 +164,7 @@ def _index_of_block_accepting_cache_control(content: list[object], on_a_tool_mes def _message_accepts_cache_control(message: object) -> bool: - """Whether a cache_control marker written on this message reaches the provider. - - Reads the same block rule as `_safe_insert_cache_control_in_message`, so the two - cannot disagree about where a marker goes. A message whose content is empty -- - an assistant turn that said everything through ``tool_calls`` -- has nowhere to - put one. - """ + """Whether a marker written on this message reaches the provider.""" if not isinstance(message, dict): return False on_a_tool_message: Final = message.get("role") == "tool" @@ -439,7 +431,6 @@ class AnthropicCacheControlHook(CustomPromptManagement): """ used_blocks = AnthropicCacheControlHook.count_request_cache_breakpoints(messages) - taken: set[int] = set() # mutable-ok: the messages this request has already marked limit_reached = False for point in points: if used_blocks >= max_blocks: @@ -450,9 +441,7 @@ class AnthropicCacheControlHook(CustomPromptManagement): type="ephemeral" ) - for target_index in AnthropicCacheControlHook._resolve_target_indices( - point=point, messages=messages, taken=taken - ): + for target_index in AnthropicCacheControlHook._resolve_target_indices(point=point, messages=messages): if used_blocks >= max_blocks: limit_reached = True break @@ -466,7 +455,6 @@ class AnthropicCacheControlHook(CustomPromptManagement): ) if AnthropicCacheControlHook._message_has_cache_control(messages[target_index]): used_blocks += 1 - taken.add(target_index) if limit_reached: break @@ -481,15 +469,9 @@ class AnthropicCacheControlHook(CustomPromptManagement): @staticmethod def _resolve_target_indices( - point: CacheControlMessageInjectionPoint, messages: list[AllMessageValues], taken: Container[int] = frozenset() + point: CacheControlMessageInjectionPoint, messages: list[AllMessageValues] ) -> list[int]: - """Resolve which message indices an injection point targets. - - ``taken`` is the messages an earlier point in the same request already marked. - The walk goes past them: arriving on one, finding it marked and dropping the - point turns two configured breakpoints into one, and the four exist so that a - prefix which stops matching at one can still match at an earlier one. - """ + """Resolve which message indices an injection point targets.""" _targetted_index: Final[int | str | None] = point.get("index", None) targetted_index: int | None = None if isinstance(_targetted_index, str): @@ -500,47 +482,36 @@ class AnthropicCacheControlHook(CustomPromptManagement): else: targetted_index = _targetted_index - # The messages the point names. A point with a role counts its index within - # that role's turns, so {role: assistant, index: -1} is the last assistant - # turn rather than the last message. targetted_role: Final = point.get("role", None) candidates: Final = tuple( index for index, message in enumerate(messages) if targetted_role is None or message.get("role") == targetted_role ) + free: Final = frozenset( + index + for index in candidates + if not AnthropicCacheControlHook._message_has_cache_control(messages[index]) + and _message_accepts_cache_control(messages[index]) + ) # Case 1: Target by role alone if targetted_index is None: - if targetted_role is None: - return [] - return [index for index in candidates if _message_accepts_cache_control(messages[index])] + return [] if targetted_role is None else [index for index in candidates if index in free] - # Case 2: Target by index, within the role the point named - original_index: Final = targetted_index - if targetted_index < 0: - targetted_index += len(candidates) - - if not 0 <= targetted_index < len(candidates): + # Case 2: Target by index, counted within the role the point named + position: Final = targetted_index + len(candidates) if targetted_index < 0 else targetted_index + if not 0 <= position < len(candidates): verbose_logger.warning( "AnthropicCacheControlHook: Provided index %s is out of bounds for message list of length %s. Targeted index was %s. Skipping cache control injection for this point.", - original_index, - len(candidates), targetted_index, + len(candidates), + position, ) return [] - # The message it arrives at may have nowhere to write a marker -- an assistant - # turn that said everything through tool_calls is the common one -- so walk back - # to the nearest earlier turn that does. Stopping there would spend the point - # and write nothing. - position = targetted_index - while position >= 0: - index = candidates[position] - if index not in taken and _message_accepts_cache_control(messages[index]): - return [index] - position -= 1 - return [] + landing: Final = next((candidates[step] for step in range(position, -1, -1) if candidates[step] in free), None) + return [] if landing is None else [landing] @staticmethod def _count_cache_control_blocks(message: object) -> int: diff --git a/tests/unit/integrations/test_anthropic_cache_control_hook.py b/tests/unit/integrations/test_anthropic_cache_control_hook.py index ed0ef328fe4..13842332a5c 100644 --- a/tests/unit/integrations/test_anthropic_cache_control_hook.py +++ b/tests/unit/integrations/test_anthropic_cache_control_hook.py @@ -16,6 +16,7 @@ from litellm.integrations.anthropic_cache_control_hook import ( supports_openai_prompt_cache_breakpoint, ) from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler +from litellm.types.integrations.anthropic_cache_control_hook import CacheControlMessageInjectionPoint from litellm.types.llms.openai import AllMessageValues @@ -3819,7 +3820,11 @@ def _anthropic_response_mock() -> MagicMock: return mock_response -async def _marked_messages(messages, points, monkeypatch: pytest.MonkeyPatch): +async def _marked_messages( + messages: list[AllMessageValues], + points: list[CacheControlMessageInjectionPoint], + monkeypatch: pytest.MonkeyPatch, +) -> list[AllMessageValues]: monkeypatch.setenv("ANTHROPIC_API_KEY", "fake_anthropic_key") monkeypatch.setattr(litellm, "callbacks", [AnthropicCacheControlHook()]) client = AsyncHTTPHandler() @@ -4088,3 +4093,47 @@ async def test_anthropic_cache_control_hook_two_points_stay_two_breakpoints(monk "content": [{"type": "text", "text": "[System: Empty message content sanitised to satisfy protocol]"}], }, ] + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_skips_a_tool_reference_block(monkeypatch: pytest.MonkeyPatch): + """ + The tool_result conversion rebuilds a tool_reference from its type and tool_name + alone, so a marker written on one never reaches the provider. + """ + marked = await _marked_messages( + [ + {"role": "user", "content": "find a tool"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + {"id": "call_1", "type": "function", "function": {"name": "tool_search", "arguments": "{}"}} + ], + }, + { + "role": "tool", + "tool_call_id": "call_1", + "content": [ + {"type": "text", "text": "found"}, + {"type": "tool_reference", "tool_name": "get_weather"}, + ], + }, + ], + [{"location": "message", "index": 2}], + monkeypatch, + ) + + assert marked[-1] == { + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "call_1", + "content": [ + {"type": "text", "text": "found", "cache_control": {"type": "ephemeral"}}, + {"type": "tool_reference", "tool_name": "get_weather"}, + ], + } + ], + } From 817f823a0490f3f9c3d3bff2f059e07404b784e8 Mon Sep 17 00:00:00 2001 From: Tan Nguyen Date: Wed, 23 Sep 2026 16:02:51 +0700 Subject: [PATCH 6/7] test(anthropic): cover the guards that refuse a non-object block or message A bare string in a content list is skipped by the walk-back, so the request still goes out; a bare string in the message list is left to litellm's own validation. Removing either guard raises AttributeError out of the hook. --- .../test_anthropic_cache_control_hook.py | 32 +++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/tests/unit/integrations/test_anthropic_cache_control_hook.py b/tests/unit/integrations/test_anthropic_cache_control_hook.py index 13842332a5c..b3414a7e9f2 100644 --- a/tests/unit/integrations/test_anthropic_cache_control_hook.py +++ b/tests/unit/integrations/test_anthropic_cache_control_hook.py @@ -4137,3 +4137,35 @@ async def test_anthropic_cache_control_hook_skips_a_tool_reference_block(monkeyp } ], } + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_skips_a_block_that_is_not_an_object(monkeypatch: pytest.MonkeyPatch): + """A caller can put a bare string in a content list; reading .type off it raises.""" + marked = await _marked_messages( + [ + {"role": "user", "content": "What is 2 + 2?"}, + {"role": "assistant", "content": ["The answer is 4."]}, + ], + [{"location": "message", "role": "assistant", "index": -1}], + monkeypatch, + ) + assert marked == [{"role": "user", "content": [{"type": "text", "text": "What is 2 + 2?"}]}] + + +@pytest.mark.asyncio +async def test_anthropic_cache_control_hook_leaves_a_message_that_is_not_an_object_to_litellm( + monkeypatch: pytest.MonkeyPatch, +): + """A bare string in the message list belongs to litellm's own validation, not to this hook.""" + monkeypatch.setenv("ANTHROPIC_API_KEY", "fake_anthropic_key") + monkeypatch.setattr(litellm, "callbacks", [AnthropicCacheControlHook()]) + client = AsyncHTTPHandler() + with patch.object(client, "post", return_value=_anthropic_response_mock()): + with pytest.raises(litellm.APIConnectionError): + await litellm.acompletion( + model="anthropic/claude-sonnet-4-5", + messages=[{"role": "user", "content": "What is 2 + 2?"}, "The answer is 4."], + cache_control_injection_points=[{"location": "message", "index": -1}], + client=client, + ) From ed02947f7b31742ea1749b9b0e1c41c2a2125ec1 Mon Sep 17 00:00:00 2001 From: Tan Nguyen Date: Thu, 1 Oct 2026 21:17:54 +0700 Subject: [PATCH 7/7] fix(anthropic): return the resolved indices as a tuple and type the narrowed message dict The resolver built three lists that nothing mutates, and the message a type guard narrowed to dict carried unknown key and value types into the block walk-back. Both sit under the repo's type-discipline and basedpyright budgets. --- .../integrations/anthropic_cache_control_hook.py | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/litellm/integrations/anthropic_cache_control_hook.py b/litellm/integrations/anthropic_cache_control_hook.py index 733937039c5..730d0a1f204 100644 --- a/litellm/integrations/anthropic_cache_control_hook.py +++ b/litellm/integrations/anthropic_cache_control_hook.py @@ -167,8 +167,9 @@ def _message_accepts_cache_control(message: object) -> bool: """Whether a marker written on this message reaches the provider.""" if not isinstance(message, dict): return False - on_a_tool_message: Final = message.get("role") == "tool" - content: Final = message.get("content") + fields: Final = cast(dict[str, object], message) # cast-ok: a runtime dict whose value types are not known here + on_a_tool_message: Final = fields.get("role") == "tool" + content: Final = fields.get("content") if isinstance(content, str): return content != "" or on_a_tool_message if isinstance(content, list): @@ -470,7 +471,7 @@ class AnthropicCacheControlHook(CustomPromptManagement): @staticmethod def _resolve_target_indices( point: CacheControlMessageInjectionPoint, messages: list[AllMessageValues] - ) -> list[int]: + ) -> tuple[int, ...]: """Resolve which message indices an injection point targets.""" _targetted_index: Final[int | str | None] = point.get("index", None) targetted_index: int | None = None @@ -497,7 +498,7 @@ class AnthropicCacheControlHook(CustomPromptManagement): # Case 1: Target by role alone if targetted_index is None: - return [] if targetted_role is None else [index for index in candidates if index in free] + return () if targetted_role is None else tuple(index for index in candidates if index in free) # Case 2: Target by index, counted within the role the point named position: Final = targetted_index + len(candidates) if targetted_index < 0 else targetted_index @@ -508,10 +509,10 @@ class AnthropicCacheControlHook(CustomPromptManagement): len(candidates), position, ) - return [] + return () landing: Final = next((candidates[step] for step in range(position, -1, -1) if candidates[step] in free), None) - return [] if landing is None else [landing] + return () if landing is None else (landing,) @staticmethod def _count_cache_control_blocks(message: object) -> int: