From 6b7d767637b126c162b4238f108139a90dd034da Mon Sep 17 00:00:00 2001 From: Giulio Leone Date: Thu, 5 Mar 2026 17:34:56 +0100 Subject: [PATCH] feat(anthropic): support top-level cache_control for automatic prompt caching (#22442) --- litellm/llms/anthropic/chat/transformation.py | 4 + .../proxy/common_utils/http_parsing_utils.py | 7 +- litellm/proxy/litellm_pre_call_utils.py | 2 +- litellm/types/llms/anthropic.py | 1 + .../test_anthropic_chat_transformation.py | 84 +++++++++++++++++-- 5 files changed, 90 insertions(+), 8 deletions(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index b9d07d7c544..5227d369027 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -194,6 +194,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): "web_search_options", "speed", "context_management", + "cache_control", ] if ( @@ -1031,6 +1032,9 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): elif param == "speed" and isinstance(value, str): # Pass through Anthropic-specific speed parameter for fast mode optional_params["speed"] = value + elif param == "cache_control" and isinstance(value, dict): + # Pass through top-level cache_control for automatic prompt caching + optional_params["cache_control"] = value ## handle thinking tokens self.update_optional_params_with_thinking_tokens( diff --git a/litellm/proxy/common_utils/http_parsing_utils.py b/litellm/proxy/common_utils/http_parsing_utils.py index 04d46ecaeb8..dc7b25ea092 100644 --- a/litellm/proxy/common_utils/http_parsing_utils.py +++ b/litellm/proxy/common_utils/http_parsing_utils.py @@ -145,8 +145,12 @@ def _safe_get_request_headers(request: Optional[Request]) -> dict: return {} state = getattr(request, "state", None) cached = getattr(state, "_cached_headers", None) - if cached is not None: + if isinstance(cached, dict): return cached + if cached is not None: + verbose_proxy_logger.debug( + "Unexpected cached request headers type - {}".format(type(cached)) + ) try: headers = dict(request.headers) except Exception as e: @@ -516,4 +520,3 @@ def _add_vector_store_id_from_path(request_data: dict, request: Request) -> None verbose_proxy_logger.debug( f"populate_request_with_path_params: No vector_store_id present in path={path}" ) - diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index 32eab99fb99..2b6723a6bba 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -799,7 +799,7 @@ class LiteLLMProxyRequestSetup: Add team-based callbacks from the config """ team_config = proxy_config.load_team_config(team_id=team_id) - if len(team_config.keys()) == 0: + if not isinstance(team_config, dict) or len(team_config) == 0: return None callback_vars_dict = {**team_config.get("callback_vars", team_config)} diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py index cef9c450423..5b8044911e5 100644 --- a/litellm/types/llms/anthropic.py +++ b/litellm/types/llms/anthropic.py @@ -363,6 +363,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False): output_format: Optional[AnthropicOutputSchema] # Structured outputs support speed: Optional[str] # Fast mode support for Opus models output_config: Optional[AnthropicOutputConfig] # Configuration for Claude's output behavior + cache_control: Optional[Dict[str, Any]] # Automatic prompt caching class AnthropicMessagesRequest(AnthropicMessagesRequestOptionalParams, total=False): diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py index 0970405ee9b..b540b0d952d 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -1,4 +1,3 @@ -import json import os import sys @@ -13,7 +12,7 @@ from litellm.llms.anthropic.chat.transformation import AnthropicConfig from litellm.llms.anthropic.experimental_pass_through.messages.transformation import ( AnthropicMessagesConfig, ) -from litellm.types.utils import PromptTokensDetailsWrapper, ServerToolUse +from litellm.types.utils import ServerToolUse def test_response_format_transformation_unit_test(): @@ -1964,7 +1963,7 @@ def test_calculate_usage_completion_tokens_details_always_populated(): # completion_tokens_details should NOT be None assert usage.completion_tokens_details is not None - assert usage.completion_tokens_details.reasoning_tokens is 0 + assert usage.completion_tokens_details.reasoning_tokens == 0 assert usage.completion_tokens_details.text_tokens == 248 assert usage.completion_tokens == 248 assert usage.prompt_tokens == 37 @@ -2861,6 +2860,83 @@ def test_map_openai_params_with_context_management(): assert result["context_management"] == non_default_params_anthropic["context_management"] +def test_cache_control_in_supported_params(): + """ + Test that cache_control is listed as a supported OpenAI param for Anthropic. + """ + config = AnthropicConfig() + params = config.get_supported_openai_params(model="claude-sonnet-4-20250514") + assert "cache_control" in params + + +def test_map_openai_params_with_cache_control(): + """ + Test that map_openai_params correctly passes through top-level cache_control + for Anthropic's automatic prompt caching. + """ + config = AnthropicConfig() + + non_default_params = { + "cache_control": {"type": "ephemeral"} + } + optional_params = {} + + result = config.map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model="claude-sonnet-4-20250514", + drop_params=False, + ) + + assert "cache_control" in result + assert result["cache_control"] == {"type": "ephemeral"} + + +def test_map_openai_params_cache_control_ignored_when_not_dict(): + """ + Test that cache_control is ignored when it is not a dict. + """ + config = AnthropicConfig() + + non_default_params = { + "cache_control": "ephemeral" + } + optional_params = {} + + result = config.map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model="claude-sonnet-4-20250514", + drop_params=False, + ) + + assert "cache_control" not in result + + +def test_transform_request_includes_cache_control(): + """ + Test that transform_request includes top-level cache_control in the request body. + """ + config = AnthropicConfig() + + messages = [{"role": "user", "content": "Hello"}] + optional_params = { + "max_tokens": 100, + "cache_control": {"type": "ephemeral"}, + } + + result = config.transform_request( + model="claude-sonnet-4-20250514", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert "cache_control" in result + assert result["cache_control"] == {"type": "ephemeral"} + + def test_compaction_block_empty_list_not_added(): """ Test that empty compaction_blocks list is not added to provider_specific_fields. @@ -2975,7 +3051,6 @@ def test_fast_mode_cost_calculation(): Test that fast mode applies the 'fast' multiplier from provider_specific_entry on top of the base model cost (1.1x for claude-opus-4-6). """ - from unittest.mock import MagicMock, patch from litellm.llms.anthropic.cost_calculation import cost_per_token from litellm.types.utils import Usage @@ -3015,7 +3090,6 @@ def test_fast_mode_with_inference_geo(): Test that fast mode + inference_geo both apply their multipliers from provider_specific_entry (1.1 * 1.1 = 1.21x for claude-opus-4-6). """ - from unittest.mock import patch from litellm.llms.anthropic.cost_calculation import cost_per_token from litellm.types.utils import Usage