fix(anthropic-messages): wire cache_control_injection_points into /v1/messages

The OpenAI-shaped /v1/chat/completions path picks up
cache_control_injection_points via litellm_logging_obj.async_get_chat_completion_prompt
inside litellm.completion, so model-config prompt caching works as
documented there.

The native Anthropic /v1/messages handler
(litellm/llms/anthropic/experimental_pass_through/messages/handler.py)
forwards Anthropic-format payloads directly to
base_llm_http_handler.anthropic_messages_handler without ever invoking
the prompt-management hook. Result: any Claude Code customer pointing
ANTHROPIC_BASE_URL at the proxy and routed to anthropic/* or
bedrock/*claude* deployments saw silently-dropped caching:
cache_creation_input_tokens=0 and cache_read_input_tokens=0 despite a
valid-looking cache_control_injection_points config that already worked
on /v1/chat/completions.

This commit:

1. Adds location: "system" and location: "tools" to the
   CacheControlInjectionPoint schema. /v1/messages keeps system and
   tools as top-level parameters (not entries in the messages array),
   so they need their own injection-point shapes. The big win for Claude
   Code — caching the long system prompt and the tool list — was
   unexpressible before this.

2. Adds AnthropicCacheControlHook.apply_to_anthropic_messages_request,
   which writes cache_control at the block level (Anthropic rejects
   message-level cache_control on /v1/messages, the existing
   _safe_insert_cache_control_in_message helper writes at the message
   level and is preserved for the OpenAI-shaped path).

3. Auto-translates the very common Claude Code config
   {location: message, role: system} to {location: system} when the
   request has no system-role message in the array but does have a
   top-level system prompt. This makes the customer's exact unmodified
   config produce cache hits on /v1/messages without a config edit.

4. Wires the new entrypoint into anthropic_messages_handler before
   dispatch, mutating messages / system / tools and popping the param
   from kwargs so it does not leak as an unknown field to upstream.

The OpenAI-shaped chat/completions adapter branch and the OpenAI
Responses API branch are unchanged — both still go through
litellm.acompletion / litellm.responses, where the existing hook
handles the OpenAI-format case.

Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-05-18 22:47:00 +00:00
parent 8c6625216b
commit af2562c6f6
No known key found for this signature in database
4 changed files with 888 additions and 0 deletions

View file

@ -241,6 +241,254 @@ class AnthropicCacheControlHook(CustomPromptManagement):
return True
return False
# ------------------------------------------------------------------
# Anthropic /v1/messages (native Messages API) processing
#
# The OpenAI-shaped path (``get_chat_completion_prompt`` above) writes
# ``cache_control`` at the message level for string content so it can be
# detected later by both OpenAI- and Anthropic-style downstream
# transformers. Anthropic's /v1/messages does NOT accept message-level
# ``cache_control`` — it must live on a content block. Additionally,
# /v1/messages separates ``system`` and ``tools`` from ``messages``, so
# those need their own injection paths.
# ------------------------------------------------------------------
@staticmethod
def apply_to_anthropic_messages_request(
messages: List[Dict[str, Any]],
system: Optional[Union[str, List[Dict[str, Any]]]],
tools: Optional[List[Dict[str, Any]]],
non_default_params: Dict[str, Any],
) -> Tuple[
List[Dict[str, Any]],
Optional[Union[str, List[Dict[str, Any]]]],
Optional[List[Dict[str, Any]]],
]:
"""Apply ``cache_control_injection_points`` to an Anthropic /v1/messages
request.
Unlike :meth:`get_chat_completion_prompt`, this entrypoint:
- inserts ``cache_control`` at the *block* level (Anthropic's API
rejects message-level cache_control),
- handles the top-level ``system`` parameter (which is not a
message in /v1/messages), and
- handles ``tools`` (cache_control on the final tool covers the
whole list).
The provided ``non_default_params`` is mutated to remove the
``cache_control_injection_points`` key so it is not forwarded
downstream as an unknown field.
"""
injection_points: List[CacheControlInjectionPoint] = non_default_params.pop(
"cache_control_injection_points", []
)
if not injection_points:
return messages, system, tools
processed_messages = copy.deepcopy(messages)
processed_system: Optional[Union[str, List[Dict[str, Any]]]] = (
copy.deepcopy(system) if system is not None else None
)
processed_tools: Optional[List[Dict[str, Any]]] = (
copy.deepcopy(tools) if tools is not None else None
)
for point in injection_points:
location = point.get("location")
control: ChatCompletionCachedContent = point.get(
"control", None
) or ChatCompletionCachedContent(type="ephemeral")
if location == "message":
# Backward-compat sugar for the very common Claude Code config
# ``location: message, role: system``. On /v1/chat/completions
# that targets the system message in the ``messages`` array;
# on /v1/messages there IS no system message (system is a
# top-level param) so the original config silently no-op'd
# for customers (this PR's motivating bug report). When we
# detect that exact mismatch — role=system requested, no
# system-role message present, but a top-level system param
# IS — treat it as ``location: system``.
if (
point.get("role") == "system"
and point.get("index") is None
and processed_system is not None
and not any(m.get("role") == "system" for m in processed_messages)
):
processed_system = AnthropicCacheControlHook._insert_cache_control_in_anthropic_system(
system=processed_system, control=control
)
continue
processed_messages = (
AnthropicCacheControlHook._process_anthropic_message_injection(
point=cast(CacheControlMessageInjectionPoint, point),
messages=processed_messages,
)
)
elif location == "system":
processed_system = (
AnthropicCacheControlHook._insert_cache_control_in_anthropic_system(
system=processed_system, control=control
)
)
elif location == "tools":
processed_tools = (
AnthropicCacheControlHook._insert_cache_control_in_anthropic_tools(
tools=processed_tools, control=control
)
)
else:
# Unknown location (e.g. "tool_config" for Bedrock Converse) –
# not applicable on the Anthropic Messages API path. Leave
# everything untouched and log so the user can debug silent
# misconfiguration.
verbose_logger.debug(
"AnthropicCacheControlHook: ignoring injection point "
f"with location={location!r} on /v1/messages path."
)
return processed_messages, processed_system, processed_tools
@staticmethod
def _process_anthropic_message_injection(
point: CacheControlMessageInjectionPoint,
messages: List[Dict[str, Any]],
) -> List[Dict[str, Any]]:
"""``location: "message"`` for Anthropic /v1/messages format.
Differs from :meth:`_process_message_injection` in that cache_control
is *always* written at the block level, never on the message dict
itself. String-content messages are upgraded to a single-block list
so the cache_control field has somewhere valid to live.
Note: on /v1/messages the system prompt is a separate top-level
parameter, NOT a message with ``role: "system"``. The customer report
in https://github.com/BerriAI/litellm/issues (Claude Code) hit this
exact pitfall. We still honor ``role: "system"`` here for
compatibility, but it will be a no-op when the request has no
system-role message (which is the case for /v1/messages).
"""
control: ChatCompletionCachedContent = point.get(
"control", None
) or ChatCompletionCachedContent(type="ephemeral")
_targetted_index: Optional[Union[int, str]] = point.get("index", None)
targetted_index: Optional[int] = None
if isinstance(_targetted_index, str):
try:
targetted_index = int(_targetted_index)
except ValueError:
pass
else:
targetted_index = _targetted_index
targetted_role = point.get("role", None)
if targetted_index is not None:
original_index = targetted_index
if targetted_index < 0:
targetted_index += len(messages)
if 0 <= targetted_index < len(messages):
messages[targetted_index] = (
AnthropicCacheControlHook._safe_insert_cache_control_in_anthropic_message(
message=messages[targetted_index], control=control
)
)
else:
verbose_logger.warning(
f"AnthropicCacheControlHook: Provided index {original_index} is out of bounds for message list of length {len(messages)}. "
f"Targeted index was {targetted_index}. Skipping cache control injection for this point."
)
elif targetted_role is not None:
for i, msg in enumerate(messages):
if msg.get("role") == targetted_role:
messages[i] = (
AnthropicCacheControlHook._safe_insert_cache_control_in_anthropic_message(
message=msg, control=control
)
)
return messages
@staticmethod
def _safe_insert_cache_control_in_anthropic_message(
message: Dict[str, Any], control: ChatCompletionCachedContent
) -> Dict[str, Any]:
"""Block-level cache_control insertion for an Anthropic message.
- ``content: str`` → wrap into ``[{"type": "text", "text": str, "cache_control": control}]``.
Anthropic accepts this shape and treats it identically to the
original string body for token-counting purposes, while making the
cache marker valid.
- ``content: list`` → set ``cache_control`` on the last block. Per
Anthropic's spec only the final block in a sequence carries the
marker; everything preceding it is covered.
"""
content = message.get("content")
if isinstance(content, str):
message["content"] = [
{
"type": "text",
"text": content,
"cache_control": control,
}
]
elif isinstance(content, list) and content:
last = content[-1]
if isinstance(last, dict):
last["cache_control"] = control
return message
@staticmethod
def _insert_cache_control_in_anthropic_system(
system: Optional[Union[str, List[Dict[str, Any]]]],
control: ChatCompletionCachedContent,
) -> Optional[Union[str, List[Dict[str, Any]]]]:
"""Apply ``cache_control`` to the Anthropic /v1/messages ``system``.
Returns a list-of-blocks shape even when the input was a string,
because cache_control on the string form is not valid Anthropic
request syntax. Returns ``None`` unchanged if there is no system
prompt (caller should not have requested ``location: "system"`` in
that case, but we no-op gracefully).
"""
if system is None:
verbose_logger.warning(
"AnthropicCacheControlHook: cache_control_injection_points "
"requested location='system' but request has no system prompt. "
"Skipping."
)
return None
if isinstance(system, str):
return [{"type": "text", "text": system, "cache_control": control}]
if isinstance(system, list) and system:
last = system[-1]
if isinstance(last, dict):
last["cache_control"] = control
return system
return system
@staticmethod
def _insert_cache_control_in_anthropic_tools(
tools: Optional[List[Dict[str, Any]]],
control: ChatCompletionCachedContent,
) -> Optional[List[Dict[str, Any]]]:
"""Apply ``cache_control`` to the final tool definition.
Anthropic only honors ``cache_control`` on the last entry of the
``tools`` array; doing so caches all preceding tool definitions as a
single chunk. No-ops gracefully when there are no tools (logged at
debug level — this is a common shape for non-tool requests).
"""
if not tools:
verbose_logger.debug(
"AnthropicCacheControlHook: cache_control_injection_points "
"requested location='tools' but request has no tools. Skipping."
)
return tools
last = tools[-1]
if isinstance(last, dict):
last["cache_control"] = control
return tools
@staticmethod
def get_custom_logger_for_anthropic_cache_control_hook(
non_default_params: Dict,

View file

@ -11,6 +11,7 @@ from functools import partial
from typing import Any, AsyncIterator, Coroutine, Dict, List, Optional, Union, cast
import litellm
from litellm.integrations.anthropic_cache_control_hook import AnthropicCacheControlHook
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.llms.anthropic.common_utils import (
strip_empty_text_blocks_from_anthropic_messages,
@ -474,6 +475,35 @@ def anthropic_messages_handler(
"display": "summarized",
}
# Apply `cache_control_injection_points` (model-config prompt caching) on the
# native Anthropic /v1/messages path. The OpenAI-shaped chat/completions
# path picks this up automatically via litellm_logging_obj.async_get_chat_completion_prompt
# in litellm.completion, but the /v1/messages handler does not pass through
# that hook — historically silently no-op'ing model-config caching for
# Claude Code users routed to anthropic/bedrock-claude deployments.
# See customer report: bedrock/us.anthropic.claude-sonnet-4-5 with
# cache_control_injection_points produced cache_creation_input_tokens=0
# on /v1/messages while working on /v1/chat/completions.
if "cache_control_injection_points" in kwargs:
_system = anthropic_messages_optional_request_params.get("system")
_tools = anthropic_messages_optional_request_params.get("tools")
# Mutates kwargs in place (pops the key) so it does not leak into the
# upstream provider request as an unknown field.
(
messages,
new_system,
new_tools,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=messages,
system=_system,
tools=_tools,
non_default_params=kwargs,
)
if new_system is not None:
anthropic_messages_optional_request_params["system"] = new_system
if new_tools is not None:
anthropic_messages_optional_request_params["tools"] = new_tools
return base_llm_http_handler.anthropic_messages_handler(
model=model,
messages=messages,

View file

@ -22,7 +22,35 @@ class CacheControlToolConfigInjectionPoint(TypedDict):
location: Literal["tool_config"]
class CacheControlSystemInjectionPoint(TypedDict, total=False):
"""Type for system-level injection points (Anthropic Messages API).
Anthropic's /v1/messages keeps the system prompt as a top-level ``system``
parameter — it is NOT a message in the ``messages`` array. Use this
injection-point shape to cache the system prompt (the common big win for
Claude Code: long system prompt cached across turns).
"""
location: Literal["system"]
control: Optional[ChatCompletionCachedContent]
class CacheControlToolsInjectionPoint(TypedDict, total=False):
"""Type for tools-level injection points (Anthropic Messages API).
Marks the last tool in the ``tools`` array with ``cache_control`` so the
entire tool list participates in the prompt cache. Anthropic only honors
``cache_control`` on the final tool entry; the marker covers all tools
that precede it.
"""
location: Literal["tools"]
control: Optional[ChatCompletionCachedContent]
CacheControlInjectionPoint = Union[
CacheControlMessageInjectionPoint,
CacheControlToolConfigInjectionPoint,
CacheControlSystemInjectionPoint,
CacheControlToolsInjectionPoint,
]

View file

@ -0,0 +1,582 @@
"""Tests for ``cache_control_injection_points`` on the Anthropic /v1/messages
path.
The OpenAI-shaped /v1/chat/completions path picks the injection points up via
``litellm_logging_obj.async_get_chat_completion_prompt`` inside
``litellm.completion``. The native Anthropic Messages path (used for
``anthropic/*`` and ``bedrock/*claude*`` deployments) historically did not, so
model-config caching silently no-op'd for Claude Code customers.
These tests pin three things:
1. The hook supports the three /v1/messages-shaped locations:
``message``, ``system``, ``tools``.
2. The hook writes ``cache_control`` at the *block* level (Anthropic
rejects message-level cache_control on /v1/messages).
3. The wiring inside ``anthropic_messages_handler`` actually forwards
the mutations to ``base_llm_http_handler.anthropic_messages_handler``.
"""
import os
import sys
from unittest.mock import MagicMock, patch
sys.path.insert(0, os.path.abspath("../.."))
from litellm.integrations.anthropic_cache_control_hook import AnthropicCacheControlHook
# ---------------------------------------------------------------------------
# Unit tests for AnthropicCacheControlHook.apply_to_anthropic_messages_request
# ---------------------------------------------------------------------------
class TestApplyToAnthropicMessagesRequest:
"""Direct unit tests for the new Anthropic-Messages-format entrypoint."""
def test_should_noop_when_no_injection_points(self):
messages = [{"role": "user", "content": "hi"}]
system = "you are helpful"
tools = [{"name": "t", "input_schema": {"type": "object"}}]
(
new_messages,
new_system,
new_tools,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=messages,
system=system,
tools=tools,
non_default_params={"foo": "bar"},
)
assert new_messages == messages
assert new_system == system
assert new_tools == tools
def test_should_pop_injection_points_from_non_default_params(self):
non_default_params = {
"cache_control_injection_points": [{"location": "system"}],
"other": "value",
}
AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=[{"role": "user", "content": "hi"}],
system="sys",
tools=None,
non_default_params=non_default_params,
)
assert "cache_control_injection_points" not in non_default_params
assert non_default_params == {"other": "value"}
def test_should_apply_cache_control_to_string_system_prompt(self):
"""``location: "system"`` upgrades a string system prompt into a
single-block list so ``cache_control`` has a valid place to live."""
(
_,
new_system,
_,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=[{"role": "user", "content": "hi"}],
system="you are a helpful assistant",
tools=None,
non_default_params={
"cache_control_injection_points": [{"location": "system"}]
},
)
assert new_system == [
{
"type": "text",
"text": "you are a helpful assistant",
"cache_control": {"type": "ephemeral"},
}
]
def test_should_apply_cache_control_to_list_system_prompt_last_block(self):
"""Multi-block system prompts get ``cache_control`` on the final
block only, per Anthropic spec — preceding blocks are covered."""
system = [
{"type": "text", "text": "first block"},
{"type": "text", "text": "second block"},
]
(
_,
new_system,
_,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=[{"role": "user", "content": "hi"}],
system=system,
tools=None,
non_default_params={
"cache_control_injection_points": [{"location": "system"}]
},
)
assert "cache_control" not in new_system[0]
assert new_system[1]["cache_control"] == {"type": "ephemeral"}
def test_should_apply_cache_control_to_last_tool(self):
tools = [
{"name": "tool_a", "input_schema": {"type": "object"}},
{"name": "tool_b", "input_schema": {"type": "object"}},
{"name": "tool_c", "input_schema": {"type": "object"}},
]
(
_,
_,
new_tools,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=[{"role": "user", "content": "hi"}],
system=None,
tools=tools,
non_default_params={
"cache_control_injection_points": [{"location": "tools"}]
},
)
assert "cache_control" not in new_tools[0]
assert "cache_control" not in new_tools[1]
assert new_tools[2]["cache_control"] == {"type": "ephemeral"}
def test_should_promote_string_content_to_block_list_for_message_injection(self):
"""Anthropic /v1/messages rejects message-level ``cache_control`` —
when content is a string the hook must wrap it into a single-block
list so the marker is valid."""
messages = [
{"role": "user", "content": "first turn"},
{"role": "assistant", "content": "first response"},
{"role": "user", "content": "second turn"},
]
(
new_messages,
_,
_,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=messages,
system=None,
tools=None,
non_default_params={
"cache_control_injection_points": [{"location": "message", "index": -1}]
},
)
# Last message converted from string → list-of-blocks with cache_control
assert new_messages[-1] == {
"role": "user",
"content": [
{
"type": "text",
"text": "second turn",
"cache_control": {"type": "ephemeral"},
}
],
}
# Earlier messages untouched
assert new_messages[0]["content"] == "first turn"
# Caller's messages not mutated
assert messages[-1]["content"] == "second turn"
def test_should_apply_cache_control_to_last_block_of_list_content_message(self):
messages = [
{
"role": "user",
"content": [
{"type": "text", "text": "page 1"},
{"type": "text", "text": "page 2"},
{"type": "text", "text": "page 3"},
],
}
]
(
new_messages,
_,
_,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=messages,
system=None,
tools=None,
non_default_params={
"cache_control_injection_points": [
{"location": "message", "role": "user"}
]
},
)
content = new_messages[0]["content"]
assert "cache_control" not in content[0]
assert "cache_control" not in content[1]
assert content[2]["cache_control"] == {"type": "ephemeral"}
def test_should_auto_translate_role_system_to_location_system_when_no_system_message(
self,
):
"""Customer-reported config shape: ``location: message, role: system``.
On /v1/chat/completions this targets the system message inside
``messages`` (where OpenAI puts it). On /v1/messages the system
prompt is a top-level parameter, NOT a message — so the config
previously silently no-op'd, which is exactly the customer's repro:
``cache_creation_input_tokens=0`` despite a valid-looking config.
The hook auto-translates ``role: system`` → ``location: system``
when (a) no message in the array has ``role: system`` and (b) a
top-level system prompt IS present. This makes the customer's exact
existing config produce cache hits on /v1/messages with no config
edit required.
"""
messages = [{"role": "user", "content": "hi"}]
(
new_messages,
new_system,
_,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=messages,
system="instructions",
tools=None,
non_default_params={
"cache_control_injection_points": [
{"location": "message", "role": "system"}
]
},
)
assert new_messages == messages
assert new_system == [
{
"type": "text",
"text": "instructions",
"cache_control": {"type": "ephemeral"},
}
]
def test_should_not_auto_translate_when_system_role_message_exists(self):
"""If the caller actually has a system-role message in the array,
respect that and target it — do not steal the marker for the
top-level ``system`` param. This matches /v1/chat/completions
semantics so users with mixed-shape requests aren't surprised."""
messages = [
{"role": "system", "content": "in-array system message"},
{"role": "user", "content": "hi"},
]
(
new_messages,
new_system,
_,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=messages,
system="top-level system",
tools=None,
non_default_params={
"cache_control_injection_points": [
{"location": "message", "role": "system"}
]
},
)
# System-role message in array was targeted → upgraded to block list
assert new_messages[0]["content"][0]["cache_control"] == {"type": "ephemeral"}
# Top-level system left alone
assert new_system == "top-level system"
def test_should_not_auto_translate_role_system_when_no_system_prompt_either(self):
"""If neither a system-role message nor a top-level system prompt
exists, the injection point has nowhere to apply. The hook must
not raise and must not mutate anything."""
messages = [{"role": "user", "content": "hi"}]
(
new_messages,
new_system,
_,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=messages,
system=None,
tools=None,
non_default_params={
"cache_control_injection_points": [
{"location": "message", "role": "system"}
]
},
)
assert new_messages == messages
assert new_system is None
def test_should_handle_multiple_injection_points_in_one_call(self):
"""Realistic Claude Code config: cache the system prompt, the tool
list, and the last user message all at once."""
(
new_messages,
new_system,
new_tools,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=[
{"role": "user", "content": "first"},
{"role": "assistant", "content": "ack"},
{"role": "user", "content": "second"},
],
system="long system prompt",
tools=[{"name": "search", "input_schema": {"type": "object"}}],
non_default_params={
"cache_control_injection_points": [
{"location": "system"},
{"location": "tools"},
{"location": "message", "index": -1},
]
},
)
assert new_system[0]["cache_control"] == {"type": "ephemeral"}
assert new_tools[0]["cache_control"] == {"type": "ephemeral"}
assert new_messages[-1]["content"][0]["cache_control"] == {"type": "ephemeral"}
def test_should_respect_explicit_control_value(self):
(
_,
new_system,
_,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=[{"role": "user", "content": "hi"}],
system="instructions",
tools=None,
non_default_params={
"cache_control_injection_points": [
{
"location": "system",
"control": {"type": "ephemeral", "ttl": "1h"},
}
]
},
)
assert new_system[0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"}
def test_should_noop_location_system_when_no_system_prompt(self):
"""No system prompt → ``location: "system"`` logs a warning and
passes ``system=None`` through unchanged. It must not raise."""
(
_,
new_system,
_,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=[{"role": "user", "content": "hi"}],
system=None,
tools=None,
non_default_params={
"cache_control_injection_points": [{"location": "system"}]
},
)
assert new_system is None
def test_should_noop_location_tools_when_no_tools(self):
(
_,
_,
new_tools,
) = AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=[{"role": "user", "content": "hi"}],
system=None,
tools=None,
non_default_params={
"cache_control_injection_points": [{"location": "tools"}]
},
)
assert new_tools is None
def test_should_warn_on_out_of_bounds_message_index(self):
with patch(
"litellm.integrations.anthropic_cache_control_hook.verbose_logger"
) as mock_logger:
AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=[{"role": "user", "content": "hi"}],
system=None,
tools=None,
non_default_params={
"cache_control_injection_points": [
{"location": "message", "index": 99}
]
},
)
mock_logger.warning.assert_called_once()
assert "out of bounds" in mock_logger.warning.call_args[0][0]
def test_should_not_mutate_caller_messages_system_or_tools(self):
messages = [{"role": "user", "content": "hi"}]
system = [{"type": "text", "text": "instr"}]
tools = [{"name": "t", "input_schema": {"type": "object"}}]
AnthropicCacheControlHook.apply_to_anthropic_messages_request(
messages=messages,
system=system,
tools=tools,
non_default_params={
"cache_control_injection_points": [
{"location": "message", "index": 0},
{"location": "system"},
{"location": "tools"},
]
},
)
assert messages == [{"role": "user", "content": "hi"}]
assert system == [{"type": "text", "text": "instr"}]
assert tools == [{"name": "t", "input_schema": {"type": "object"}}]
# ---------------------------------------------------------------------------
# End-to-end wiring through anthropic_messages_handler
# ---------------------------------------------------------------------------
def _async_return(value):
async def _coro():
return value
return _coro()
class TestWiringIntoAnthropicMessagesHandler:
"""Verify that the dispatch layer actually invokes the hook and forwards
the mutated values to ``base_llm_http_handler.anthropic_messages_handler``.
Without this wiring the hook is dead code on /v1/messages."""
def _run_handler_and_capture(self, **call_kwargs):
"""Invoke ``anthropic_messages_handler`` with the native Anthropic
provider config branch and capture the outbound payload."""
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
anthropic_messages_handler,
)
captured = {}
def fake_base_handler(**kwargs):
captured["messages"] = kwargs.get("messages")
captured["anthropic_messages_optional_request_params"] = kwargs.get(
"anthropic_messages_optional_request_params"
)
captured["kwargs"] = kwargs.get("kwargs")
return MagicMock()
with patch(
"litellm.llms.anthropic.experimental_pass_through.messages.handler.base_llm_http_handler"
) as mock_handler:
mock_handler.anthropic_messages_handler = fake_base_handler
try:
anthropic_messages_handler(**call_kwargs)
except (ValueError, TypeError, AttributeError):
pass
return captured
def test_should_inject_cache_control_into_system_on_bedrock_messages_path(self):
"""Repro of customer report: bedrock/us.anthropic.claude-sonnet-4-5
+ ``cache_control_injection_points`` + Claude Code /v1/messages.
Before this fix the outbound system block had no cache_control,
producing cache_creation_input_tokens=0."""
captured = self._run_handler_and_capture(
max_tokens=1024,
messages=[{"role": "user", "content": "hello"}],
model="bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0",
system="A very long Claude Code system prompt over 1024 tokens...",
custom_llm_provider="bedrock",
api_key="fake",
cache_control_injection_points=[{"location": "system"}],
)
optional = captured["anthropic_messages_optional_request_params"]
assert isinstance(optional["system"], list), (
"system should have been upgraded from str to a block list so "
"cache_control has a valid place to live"
)
assert optional["system"][0]["cache_control"] == {"type": "ephemeral"}
# Param must not leak to upstream
assert "cache_control_injection_points" not in optional
assert "cache_control_injection_points" not in (captured["kwargs"] or {})
def test_should_inject_cache_control_into_tools_on_messages_path(self):
captured = self._run_handler_and_capture(
max_tokens=1024,
messages=[{"role": "user", "content": "hello"}],
model="bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0",
tools=[
{"name": "a", "input_schema": {"type": "object"}},
{"name": "b", "input_schema": {"type": "object"}},
],
custom_llm_provider="bedrock",
api_key="fake",
cache_control_injection_points=[{"location": "tools"}],
)
optional = captured["anthropic_messages_optional_request_params"]
assert "cache_control" not in optional["tools"][0]
assert optional["tools"][1]["cache_control"] == {"type": "ephemeral"}
def test_should_inject_cache_control_into_last_message_block_on_messages_path(
self,
):
captured = self._run_handler_and_capture(
max_tokens=1024,
messages=[
{"role": "user", "content": "first"},
{"role": "assistant", "content": "ack"},
{"role": "user", "content": "second"},
],
model="bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0",
custom_llm_provider="bedrock",
api_key="fake",
cache_control_injection_points=[{"location": "message", "index": -1}],
)
last_message = captured["messages"][-1]
assert isinstance(last_message["content"], list)
assert last_message["content"][0]["cache_control"] == {"type": "ephemeral"}
def test_should_handle_customer_exact_repro_config(self):
"""End-to-end repro of the customer report:
model: bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0
cache_control_injection_points:
- location: message
role: system
Before this fix the outbound system payload had no cache marker on
/v1/messages (Claude Code path), producing
``cache_creation_input_tokens=0``. After this fix the customer's
unmodified config produces a system block with cache_control.
"""
captured = self._run_handler_and_capture(
max_tokens=1024,
messages=[{"role": "user", "content": "hello"}],
model="bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0",
system="A long Claude Code system prompt (>1024 tokens).",
custom_llm_provider="bedrock",
api_key="fake",
cache_control_injection_points=[{"location": "message", "role": "system"}],
)
optional = captured["anthropic_messages_optional_request_params"]
assert isinstance(optional["system"], list)
assert optional["system"][0]["cache_control"] == {"type": "ephemeral"}
def test_should_be_noop_when_injection_points_absent(self):
"""Regression guard: without the param, request must flow through
unchanged. No silent upgrades to block-list shape."""
captured = self._run_handler_and_capture(
max_tokens=1024,
messages=[{"role": "user", "content": "hello"}],
model="bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0",
system="instructions",
custom_llm_provider="bedrock",
api_key="fake",
)
assert captured["messages"] == [{"role": "user", "content": "hello"}]
optional = captured["anthropic_messages_optional_request_params"]
assert optional.get("system") == "instructions"