feat(anthropic): support top-level cache_control for automatic prompt caching (#22442)

This commit is contained in:
Giulio Leone 2026-03-05 17:34:56 +01:00 • committed by GitHub
parent 5183a6e850
commit 6b7d767637
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 90 additions and 8 deletions

View file

@ -194,6 +194,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
"web_search_options",
"speed",
"context_management",
"cache_control",
]
if (
@ -1031,6 +1032,9 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
elif param == "speed" and isinstance(value, str):
# Pass through Anthropic-specific speed parameter for fast mode
optional_params["speed"] = value
elif param == "cache_control" and isinstance(value, dict):
# Pass through top-level cache_control for automatic prompt caching
optional_params["cache_control"] = value
## handle thinking tokens
self.update_optional_params_with_thinking_tokens(

View file

@ -145,8 +145,12 @@ def _safe_get_request_headers(request: Optional[Request]) -> dict:
return {}
state = getattr(request, "state", None)
cached = getattr(state, "_cached_headers", None)
if cached is not None:
if isinstance(cached, dict):
return cached
if cached is not None:
verbose_proxy_logger.debug(
"Unexpected cached request headers type - {}".format(type(cached))
)
try:
headers = dict(request.headers)
except Exception as e:
@ -516,4 +520,3 @@ def _add_vector_store_id_from_path(request_data: dict, request: Request) -> None
verbose_proxy_logger.debug(
f"populate_request_with_path_params: No vector_store_id present in path={path}"
)

View file

@ -799,7 +799,7 @@ class LiteLLMProxyRequestSetup:
Add team-based callbacks from the config
"""
team_config = proxy_config.load_team_config(team_id=team_id)
if len(team_config.keys()) == 0:
if not isinstance(team_config, dict) or len(team_config) == 0:
return None
callback_vars_dict = {**team_config.get("callback_vars", team_config)}

View file

@ -363,6 +363,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False):
output_format: Optional[AnthropicOutputSchema] # Structured outputs support
speed: Optional[str] # Fast mode support for Opus models
output_config: Optional[AnthropicOutputConfig] # Configuration for Claude's output behavior
cache_control: Optional[Dict[str, Any]] # Automatic prompt caching
class AnthropicMessagesRequest(AnthropicMessagesRequestOptionalParams, total=False):

View file

@ -1,4 +1,3 @@
import json
import os
import sys
@ -13,7 +12,7 @@ from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
from litellm.types.utils import PromptTokensDetailsWrapper, ServerToolUse
from litellm.types.utils import ServerToolUse
def test_response_format_transformation_unit_test():
@ -1964,7 +1963,7 @@ def test_calculate_usage_completion_tokens_details_always_populated():
# completion_tokens_details should NOT be None
assert usage.completion_tokens_details is not None
assert usage.completion_tokens_details.reasoning_tokens is 0
assert usage.completion_tokens_details.reasoning_tokens == 0
assert usage.completion_tokens_details.text_tokens == 248
assert usage.completion_tokens == 248
assert usage.prompt_tokens == 37
@ -2861,6 +2860,83 @@ def test_map_openai_params_with_context_management():
assert result["context_management"] == non_default_params_anthropic["context_management"]
def test_cache_control_in_supported_params():
"""
Test that cache_control is listed as a supported OpenAI param for Anthropic.
"""
config = AnthropicConfig()
params = config.get_supported_openai_params(model="claude-sonnet-4-20250514")
assert "cache_control" in params
def test_map_openai_params_with_cache_control():
"""
Test that map_openai_params correctly passes through top-level cache_control
for Anthropic's automatic prompt caching.
"""
config = AnthropicConfig()
non_default_params = {
"cache_control": {"type": "ephemeral"}
}
optional_params = {}
result = config.map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model="claude-sonnet-4-20250514",
drop_params=False,
)
assert "cache_control" in result
assert result["cache_control"] == {"type": "ephemeral"}
def test_map_openai_params_cache_control_ignored_when_not_dict():
"""
Test that cache_control is ignored when it is not a dict.
"""
config = AnthropicConfig()
non_default_params = {
"cache_control": "ephemeral"
}
optional_params = {}
result = config.map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model="claude-sonnet-4-20250514",
drop_params=False,
)
assert "cache_control" not in result
def test_transform_request_includes_cache_control():
"""
Test that transform_request includes top-level cache_control in the request body.
"""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
optional_params = {
"max_tokens": 100,
"cache_control": {"type": "ephemeral"},
}
result = config.transform_request(
model="claude-sonnet-4-20250514",
messages=messages,
optional_params=optional_params,
litellm_params={},
headers={},
)
assert "cache_control" in result
assert result["cache_control"] == {"type": "ephemeral"}
def test_compaction_block_empty_list_not_added():
"""
Test that empty compaction_blocks list is not added to provider_specific_fields.
@ -2975,7 +3051,6 @@ def test_fast_mode_cost_calculation():
Test that fast mode applies the 'fast' multiplier from provider_specific_entry
on top of the base model cost (1.1x for claude-opus-4-6).
"""
from unittest.mock import MagicMock, patch
from litellm.llms.anthropic.cost_calculation import cost_per_token
from litellm.types.utils import Usage
@ -3015,7 +3090,6 @@ def test_fast_mode_with_inference_geo():
Test that fast mode + inference_geo both apply their multipliers from
provider_specific_entry (1.1 * 1.1 = 1.21x for claude-opus-4-6).
"""
from unittest.mock import patch
from litellm.llms.anthropic.cost_calculation import cost_per_token
from litellm.types.utils import Usage