mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
feat(anthropic): support top-level cache_control for automatic prompt caching (#22442)
This commit is contained in:
parent
5183a6e850
commit
6b7d767637
5 changed files with 90 additions and 8 deletions
|
|
@ -194,6 +194,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
"web_search_options",
|
||||
"speed",
|
||||
"context_management",
|
||||
"cache_control",
|
||||
]
|
||||
|
||||
if (
|
||||
|
|
@ -1031,6 +1032,9 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
elif param == "speed" and isinstance(value, str):
|
||||
# Pass through Anthropic-specific speed parameter for fast mode
|
||||
optional_params["speed"] = value
|
||||
elif param == "cache_control" and isinstance(value, dict):
|
||||
# Pass through top-level cache_control for automatic prompt caching
|
||||
optional_params["cache_control"] = value
|
||||
|
||||
## handle thinking tokens
|
||||
self.update_optional_params_with_thinking_tokens(
|
||||
|
|
|
|||
|
|
@ -145,8 +145,12 @@ def _safe_get_request_headers(request: Optional[Request]) -> dict:
|
|||
return {}
|
||||
state = getattr(request, "state", None)
|
||||
cached = getattr(state, "_cached_headers", None)
|
||||
if cached is not None:
|
||||
if isinstance(cached, dict):
|
||||
return cached
|
||||
if cached is not None:
|
||||
verbose_proxy_logger.debug(
|
||||
"Unexpected cached request headers type - {}".format(type(cached))
|
||||
)
|
||||
try:
|
||||
headers = dict(request.headers)
|
||||
except Exception as e:
|
||||
|
|
@ -516,4 +520,3 @@ def _add_vector_store_id_from_path(request_data: dict, request: Request) -> None
|
|||
verbose_proxy_logger.debug(
|
||||
f"populate_request_with_path_params: No vector_store_id present in path={path}"
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -799,7 +799,7 @@ class LiteLLMProxyRequestSetup:
|
|||
Add team-based callbacks from the config
|
||||
"""
|
||||
team_config = proxy_config.load_team_config(team_id=team_id)
|
||||
if len(team_config.keys()) == 0:
|
||||
if not isinstance(team_config, dict) or len(team_config) == 0:
|
||||
return None
|
||||
|
||||
callback_vars_dict = {**team_config.get("callback_vars", team_config)}
|
||||
|
|
|
|||
|
|
@ -363,6 +363,7 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False):
|
|||
output_format: Optional[AnthropicOutputSchema] # Structured outputs support
|
||||
speed: Optional[str] # Fast mode support for Opus models
|
||||
output_config: Optional[AnthropicOutputConfig] # Configuration for Claude's output behavior
|
||||
cache_control: Optional[Dict[str, Any]] # Automatic prompt caching
|
||||
|
||||
|
||||
class AnthropicMessagesRequest(AnthropicMessagesRequestOptionalParams, total=False):
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
|
|
@ -13,7 +12,7 @@ from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
|||
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
|
||||
AnthropicMessagesConfig,
|
||||
)
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, ServerToolUse
|
||||
from litellm.types.utils import ServerToolUse
|
||||
|
||||
|
||||
def test_response_format_transformation_unit_test():
|
||||
|
|
@ -1964,7 +1963,7 @@ def test_calculate_usage_completion_tokens_details_always_populated():
|
|||
|
||||
# completion_tokens_details should NOT be None
|
||||
assert usage.completion_tokens_details is not None
|
||||
assert usage.completion_tokens_details.reasoning_tokens is 0
|
||||
assert usage.completion_tokens_details.reasoning_tokens == 0
|
||||
assert usage.completion_tokens_details.text_tokens == 248
|
||||
assert usage.completion_tokens == 248
|
||||
assert usage.prompt_tokens == 37
|
||||
|
|
@ -2861,6 +2860,83 @@ def test_map_openai_params_with_context_management():
|
|||
assert result["context_management"] == non_default_params_anthropic["context_management"]
|
||||
|
||||
|
||||
def test_cache_control_in_supported_params():
|
||||
"""
|
||||
Test that cache_control is listed as a supported OpenAI param for Anthropic.
|
||||
"""
|
||||
config = AnthropicConfig()
|
||||
params = config.get_supported_openai_params(model="claude-sonnet-4-20250514")
|
||||
assert "cache_control" in params
|
||||
|
||||
|
||||
def test_map_openai_params_with_cache_control():
|
||||
"""
|
||||
Test that map_openai_params correctly passes through top-level cache_control
|
||||
for Anthropic's automatic prompt caching.
|
||||
"""
|
||||
config = AnthropicConfig()
|
||||
|
||||
non_default_params = {
|
||||
"cache_control": {"type": "ephemeral"}
|
||||
}
|
||||
optional_params = {}
|
||||
|
||||
result = config.map_openai_params(
|
||||
non_default_params=non_default_params,
|
||||
optional_params=optional_params,
|
||||
model="claude-sonnet-4-20250514",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
assert "cache_control" in result
|
||||
assert result["cache_control"] == {"type": "ephemeral"}
|
||||
|
||||
|
||||
def test_map_openai_params_cache_control_ignored_when_not_dict():
|
||||
"""
|
||||
Test that cache_control is ignored when it is not a dict.
|
||||
"""
|
||||
config = AnthropicConfig()
|
||||
|
||||
non_default_params = {
|
||||
"cache_control": "ephemeral"
|
||||
}
|
||||
optional_params = {}
|
||||
|
||||
result = config.map_openai_params(
|
||||
non_default_params=non_default_params,
|
||||
optional_params=optional_params,
|
||||
model="claude-sonnet-4-20250514",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
assert "cache_control" not in result
|
||||
|
||||
|
||||
def test_transform_request_includes_cache_control():
|
||||
"""
|
||||
Test that transform_request includes top-level cache_control in the request body.
|
||||
"""
|
||||
config = AnthropicConfig()
|
||||
|
||||
messages = [{"role": "user", "content": "Hello"}]
|
||||
optional_params = {
|
||||
"max_tokens": 100,
|
||||
"cache_control": {"type": "ephemeral"},
|
||||
}
|
||||
|
||||
result = config.transform_request(
|
||||
model="claude-sonnet-4-20250514",
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert "cache_control" in result
|
||||
assert result["cache_control"] == {"type": "ephemeral"}
|
||||
|
||||
|
||||
def test_compaction_block_empty_list_not_added():
|
||||
"""
|
||||
Test that empty compaction_blocks list is not added to provider_specific_fields.
|
||||
|
|
@ -2975,7 +3051,6 @@ def test_fast_mode_cost_calculation():
|
|||
Test that fast mode applies the 'fast' multiplier from provider_specific_entry
|
||||
on top of the base model cost (1.1x for claude-opus-4-6).
|
||||
"""
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from litellm.llms.anthropic.cost_calculation import cost_per_token
|
||||
from litellm.types.utils import Usage
|
||||
|
|
@ -3015,7 +3090,6 @@ def test_fast_mode_with_inference_geo():
|
|||
Test that fast mode + inference_geo both apply their multipliers from
|
||||
provider_specific_entry (1.1 * 1.1 = 1.21x for claude-opus-4-6).
|
||||
"""
|
||||
from unittest.mock import patch
|
||||
|
||||
from litellm.llms.anthropic.cost_calculation import cost_per_token
|
||||
from litellm.types.utils import Usage
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue