fix(anthropic): write the cache_control marker on a block that accepts one

The marker went on the last content block whatever it was. Anthropic accepts one
on text, image, tool_use, tool_result and document blocks; a thinking block is not
among them, and an empty text block is replaced by a placeholder before the request
goes out, taking the marker with it. Either way the configured breakpoint is spent
and the turn is not cached.

Walk back to the last block that accepts a marker, the way the OpenAI prompt cache
path in this file already does. A tool message is the exception: its empty text
block is kept, nested in the tool_result the conversion builds, so the marker stays
on it.
This commit is contained in:
Tan Nguyen 2026-09-23 11:13:54 +07:00
parent b4bb2a77a2
commit 0c630cef3b
2 changed files with 192 additions and 3 deletions

View file

@ -67,6 +67,10 @@ _GPT_VERSION_PATTERN: Final = re.compile(r"^gpt-(\d+)(?:\.(\d+))?")
OPENAI_PROMPT_CACHE_BREAKPOINT_BLOCK_TYPES: Final = frozenset(
{"text", "image", "image_url", "file", "input_audio", "input_text", "input_image", "input_file"}
)
# Anthropic lists the block types a cache_control marker may sit on: text, image,
# tool_use, tool_result and document. A list of refused types rather than accepted
# ones so a block type this code has not been told about still takes a marker.
ANTHROPIC_BLOCK_TYPES_WITHOUT_CACHE_CONTROL: Final = frozenset({"thinking", "redacted_thinking"})
OPENAI_API_HOST: Final = "api.openai.com"
OPENAI_API_BASE_ENV_VARS: Final = ("OPENAI_BASE_URL", "OPENAI_API_BASE")
_OBJECT_MAPPING_ADAPTER: Final = TypeAdapter(dict[object, object])
@ -139,6 +143,28 @@ def _accepts_prompt_cache_breakpoint(block: object) -> bool:
return isinstance(block, dict) and block.get("type") in OPENAI_PROMPT_CACHE_BREAKPOINT_BLOCK_TYPES
def _index_of_block_accepting_cache_control(content: list[object], on_a_tool_message: bool) -> int | None:
"""Position of the last block a cache_control marker can be written on, or None.
Searched from the end: a marker caches everything up to and including its own
block, so the last one caches the most.
An empty text block is refused because the rewrites this hook runs ahead of drop
it, and the marker goes with it. ``on_a_tool_message`` lifts that refusal: a tool
message keeps its empty block, nested in the tool_result the rewrite builds.
"""
for index in range(len(content) - 1, -1, -1):
block = content[index]
if not isinstance(block, dict):
continue
if block.get("type") in ANTHROPIC_BLOCK_TYPES_WITHOUT_CACHE_CONTROL:
continue
if block.get("type") == "text" and not block.get("text") and not on_a_tool_message:
continue
return index
return None
# Set by a caller whose message list is not the one that goes upstream -- today the
# Responses API layer, whose `instructions` only becomes a system message further down.
# Tells this hook to hand role-targeted points to the pass holding the final messages
@ -507,10 +533,13 @@ class AnthropicCacheControlHook(CustomPromptManagement):
# 1. if string, insert cache control in the message
if isinstance(message_content, str):
message["cache_control"] = control
# 2. list of objects - only apply to last item per Anthropic spec
# 2. list of objects - the last block that accepts a marker, per Anthropic spec
elif isinstance(message_content, list):
if len(message_content) > 0 and isinstance(message_content[-1], dict):
message_content[-1]["cache_control"] = control # pyright: ignore[reportGeneralTypeIssues] # loose runtime dict
target_index: Final = _index_of_block_accepting_cache_control(
message_content, on_a_tool_message=message.get("role") == "tool"
)
if target_index is not None:
message_content[target_index]["cache_control"] = control # pyright: ignore[reportGeneralTypeIssues] # loose runtime dict
return message
@staticmethod

View file

@ -3642,3 +3642,163 @@ class TestRecordGatewayInjection:
custom_llm_provider="anthropic",
)
assert self.KEY not in kwargs["litellm_metadata"]
@pytest.mark.asyncio
async def test_anthropic_cache_control_hook_skips_a_thinking_block(monkeypatch: pytest.MonkeyPatch):
"""
A cache_control marker on a thinking block is spent on a block Anthropic does not
accept one on, so the turn it was meant to cache is not cached. The marker goes on
the last block of the message that accepts one instead.
"""
monkeypatch.setenv("ANTHROPIC_API_KEY", "fake_anthropic_key")
anthropic_cache_control_hook = AnthropicCacheControlHook()
monkeypatch.setattr(litellm, "callbacks", [anthropic_cache_control_hook])
mock_response = MagicMock()
mock_response.json.return_value = {
"id": "msg_01",
"type": "message",
"role": "assistant",
"model": "claude-sonnet-4-5",
"content": [{"type": "text", "text": "Because two plus two is four."}],
"stop_reason": "end_turn",
"usage": {"input_tokens": 10, "output_tokens": 20},
}
mock_response.status_code = 200
client = AsyncHTTPHandler()
with patch.object(client, "post", return_value=mock_response) as mock_post:
await litellm.acompletion(
model="anthropic/claude-sonnet-4-5",
messages=[
{"role": "user", "content": "What is 2 + 2?"},
{
"role": "assistant",
"content": [
{"type": "text", "text": "The answer is 4."},
{"type": "thinking", "thinking": "Adding two and two.", "signature": "sig"},
{"type": "redacted_thinking", "data": "redacted"},
],
},
{"role": "user", "content": "Why?"},
],
cache_control_injection_points=[{"location": "message", "index": 1}],
client=client,
)
request_body = mock_post.call_args.kwargs["json"]
assert request_body["messages"][1] == {
"role": "assistant",
"content": [
{"type": "text", "text": "The answer is 4.", "cache_control": {"type": "ephemeral"}},
{"type": "thinking", "thinking": "Adding two and two.", "signature": "sig"},
],
}
@pytest.mark.asyncio
async def test_anthropic_cache_control_hook_skips_an_empty_text_block(monkeypatch: pytest.MonkeyPatch):
"""
An empty text block is replaced by a placeholder before the request goes out, and a
cache_control marker written on it is replaced with it. The marker goes on the last
block that survives instead.
"""
monkeypatch.setenv("ANTHROPIC_API_KEY", "fake_anthropic_key")
monkeypatch.setattr(litellm, "callbacks", [AnthropicCacheControlHook()])
mock_response = MagicMock()
mock_response.json.return_value = {
"id": "msg_01",
"type": "message",
"role": "assistant",
"model": "claude-sonnet-4-5",
"content": [{"type": "text", "text": "Because two plus two is four."}],
"stop_reason": "end_turn",
"usage": {"input_tokens": 10, "output_tokens": 20},
}
mock_response.status_code = 200
client = AsyncHTTPHandler()
with patch.object(client, "post", return_value=mock_response) as mock_post:
await litellm.acompletion(
model="anthropic/claude-sonnet-4-5",
messages=[
{"role": "user", "content": "What is 2 + 2?"},
{
"role": "assistant",
"content": [
{"type": "text", "text": "The answer is 4."},
{"type": "text", "text": ""},
],
},
{"role": "user", "content": "Why?"},
],
cache_control_injection_points=[{"location": "message", "index": 1}],
client=client,
)
request_body = mock_post.call_args.kwargs["json"]
assert request_body["messages"][1] == {
"role": "assistant",
"content": [
{"type": "text", "text": "The answer is 4.", "cache_control": {"type": "ephemeral"}},
{"type": "text", "text": "[System: Empty message content sanitised to satisfy protocol]"},
],
}
@pytest.mark.asyncio
async def test_anthropic_cache_control_hook_marks_an_empty_tool_result(monkeypatch: pytest.MonkeyPatch):
"""
A tool message keeps its empty text block - it goes out nested in the tool_result the
conversion builds - so the marker stays on it rather than walking off the message.
"""
monkeypatch.setenv("ANTHROPIC_API_KEY", "fake_anthropic_key")
monkeypatch.setattr(litellm, "callbacks", [AnthropicCacheControlHook()])
mock_response = MagicMock()
mock_response.json.return_value = {
"id": "msg_01",
"type": "message",
"role": "assistant",
"model": "claude-sonnet-4-5",
"content": [{"type": "text", "text": "Nothing came back."}],
"stop_reason": "end_turn",
"usage": {"input_tokens": 10, "output_tokens": 20},
}
mock_response.status_code = 200
client = AsyncHTTPHandler()
with patch.object(client, "post", return_value=mock_response) as mock_post:
await litellm.acompletion(
model="anthropic/claude-sonnet-4-5",
messages=[
{"role": "user", "content": "Search for it."},
{
"role": "assistant",
"content": None,
"tool_calls": [
{"id": "call_1", "type": "function", "function": {"name": "search", "arguments": "{}"}}
],
},
{"role": "tool", "tool_call_id": "call_1", "content": [{"type": "text", "text": ""}]},
],
cache_control_injection_points=[{"location": "message", "index": 2}],
client=client,
)
request_body = mock_post.call_args.kwargs["json"]
assert request_body["messages"][-1] == {
"role": "user",
"content": [
{
"type": "tool_result",
"tool_use_id": "call_1",
"content": [{"type": "text", "text": "", "cache_control": {"type": "ephemeral"}}],
}
],
}