Merge pull request #39355 from BerriAI/litellm_fix_messages_passthrough_cache_control_ttl

fix(messages): drop cache_control ttl on non-Anthropic /v1/messages passthrough
This commit is contained in:
Mateo Wang 2026-09-02 16:47:26 -07:00 committed by GitHub
commit 4ffe536a69
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
6 changed files with 394 additions and 51 deletions

View file

@ -1411,6 +1411,97 @@ def flatten_unencrypted_web_search_results_in_anthropic_messages( # mutable-ok:
return [_flatten_web_search_results_in_message(m) for m in messages] # mutable-ok: JSON wire format
def _normalized_cache_control(cache_control: object) -> dict[str, str] | None: # mutable-ok: JSON wire format
if not isinstance(cache_control, Mapping):
return None
cache_type: Final = cache_control.get("type")
return {"type": cache_type if isinstance(cache_type, str) else "ephemeral"} # mutable-ok: JSON wire format
def _with_portable_cache_control(block: Mapping[str, object]) -> dict[str, object]: # mutable-ok: JSON wire format
if "cache_control" not in block:
return dict(block) # mutable-ok: JSON wire format
normalized: Final = _normalized_cache_control(block["cache_control"])
rest: Final = {key: value for key, value in block.items() if key != "cache_control"} # mutable-ok: JSON wire format
return rest if normalized is None else {**rest, "cache_control": normalized} # mutable-ok: JSON wire format
def _with_portable_cache_control_in_blocks(blocks: object) -> object:
if isinstance(blocks, str) or not isinstance(blocks, Sequence):
return blocks
return [ # mutable-ok: JSON wire format
_with_portable_cache_control(block) if isinstance(block, Mapping) else block for block in blocks
]
def _with_portable_cache_control_in_content_block(block: object) -> object:
if not isinstance(block, Mapping):
return block
portable: Final = _with_portable_cache_control(block)
if portable.get("type") != "tool_result" or "content" not in portable:
return portable
return { # mutable-ok: JSON wire format
**portable,
"content": _with_portable_cache_control_in_blocks(portable["content"]),
}
def _with_portable_cache_control_in_message(message: object) -> object:
if not isinstance(message, Mapping) or "content" not in message:
return message
content: Final = message["content"]
if isinstance(content, str) or not isinstance(content, Sequence):
return message
return { # mutable-ok: JSON wire format
**message,
"content": [ # mutable-ok: JSON wire format
_with_portable_cache_control_in_content_block(block) for block in content
],
}
def _with_portable_cache_control_in_messages(messages: object) -> object:
if isinstance(messages, str) or not isinstance(messages, Sequence):
return messages
return [ # mutable-ok: JSON wire format
_with_portable_cache_control_in_message(message) for message in messages
]
def _with_portable_cache_control_in_scoped_value(key: str, value: object) -> object:
match key:
case "system" | "tools":
return _with_portable_cache_control_in_blocks(value)
case "messages":
return _with_portable_cache_control_in_messages(value)
case _:
return value
def normalize_cache_control_in_anthropic_payload(
payload: Mapping[str, object],
) -> dict[str, object]: # mutable-ok: JSON wire format
"""
Return a copy of an Anthropic /v1/messages payload with every
``cache_control`` entry reduced to ``{"type": <its type, or "ephemeral">}``
at the places the Messages API defines it: the request itself, system
blocks, tools, message content blocks, and ``tool_result`` content blocks.
Application data such as ``tool_use.input`` and tool ``input_schema`` is
never touched, even when it happens to contain a ``cache_control`` key.
Anthropic itself accepts prompt-caching extensions such as ``ttl``, but
strict non-Anthropic implementations of the Messages API validate the field
literally and reject the whole request (``cache_control.ttl: 1h is not
supported``, ``cache_control.type is required``), which 400s clients like
Claude Code that send cache hints. Non-dict ``cache_control`` values are
dropped entirely. The caller's payload is never mutated.
"""
portable: Final = _with_portable_cache_control(payload)
return { # mutable-ok: JSON wire format
key: _with_portable_cache_control_in_scoped_value(key, value) for key, value in portable.items()
}
def process_anthropic_headers(headers: httpx.Headers | dict) -> dict:
openai_headers: Final = {}
if "anthropic-ratelimit-requests-limit" in headers:

View file

@ -99,6 +99,10 @@ def _deployment_passes_through_anthropic_messages(model_info: object) -> bool:
return isinstance(supported_endpoints, (list, tuple)) and "/v1/messages" in supported_endpoints
def _deployment_supports_cache_control_ttl(model_info: object) -> bool:
return isinstance(model_info, dict) and model_info.get("cache_control_ttl") is True
####### ENVIRONMENT VARIABLES ###################
# Initialize any necessary instances or variables here
base_llm_http_handler = BaseLLMHTTPHandler()
@ -568,7 +572,9 @@ def anthropic_messages_handler(
OpenAILikeAnthropicMessagesConfig,
)
anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig()
anthropic_messages_provider_config = OpenAILikeAnthropicMessagesConfig(
cache_control_ttl=_deployment_supports_cache_control_ttl(kwargs.get("model_info")),
)
if anthropic_messages_provider_config is None:
# Route to Responses API for OpenAI / Azure, chat/completions for everything else.
if _should_route_to_responses_api(custom_llm_provider, original_model, model):

View file

@ -54,7 +54,10 @@ That's it! The provider will be automatically loaded and available.
"constraints": {
"temperature_max": 1.0,
"temperature_min": 0.0,
"temperature_min_with_n_gt_1": 0.3
"temperature_min_with_n_gt_1": 0.3,
// /v1/messages providers only: keep Anthropic cache_control extensions
// such as ttl instead of stripping them down to {"type": ...}
"cache_control_ttl": true
},
// Optional: Special handling flags

View file

@ -1,11 +1,13 @@
from typing import Any, Final
import litellm
from litellm.llms.anthropic.common_utils import normalize_cache_control_in_anthropic_payload
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
from litellm.llms.openai_like.json_loader import SimpleProviderConfig
from litellm.secret_managers.main import get_secret_str
from litellm.types.router import GenericLiteLLMParams
DEFAULT_ANTHROPIC_API_VERSION: Final = "2023-06-01"
@ -19,10 +21,17 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig):
``"/v1/messages"``. The inbound Anthropic payload (system, cache_control,
thinking, tools, ...) is forwarded essentially unchanged to
``{api_base}/v1/messages``, so Anthropic-only features that the
Anthropic->OpenAI translation would otherwise drop are preserved. Response
parsing and streaming are inherited from the native Anthropic config.
Anthropic->OpenAI translation would otherwise drop are preserved. The one
exception is ``cache_control``, whose Anthropic-only extensions (``ttl``)
are stripped unless the deployment opts in with
``model_info.cache_control_ttl: true``. Response parsing and streaming are
inherited from the native Anthropic config.
"""
def __init__(self, cache_control_ttl: bool = False) -> None:
super().__init__()
self._cache_control_ttl: Final = cache_control_ttl
def validate_anthropic_messages_environment(
self,
headers: dict[str, str],
@ -53,6 +62,35 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig):
def should_filter_anthropic_beta_headers(self) -> bool:
return False
def supports_cache_control_ttl(self) -> bool:
return self._cache_control_ttl
def transform_anthropic_messages_request(
self,
model: str,
messages: list[dict], # mutable-ok: matches dict-typed base signature
anthropic_messages_optional_request_params: dict, # mutable-ok: matches dict-typed base signature
litellm_params: GenericLiteLLMParams,
headers: dict, # mutable-ok: matches dict-typed base signature
) -> dict: # mutable-ok: matches dict-typed base signature
"""
Anthropic ignores prompt-caching hints it cannot honor, but strict
non-Anthropic implementations of the Messages API 400 the whole request
on Anthropic-only ``cache_control`` extensions (``cache_control.ttl: 1h
is not supported``), so unless the provider declares ttl support the
hints are reduced to their portable ``{"type": ...}`` core.
"""
request: Final = super().transform_anthropic_messages_request(
model=model,
messages=messages,
anthropic_messages_optional_request_params=anthropic_messages_optional_request_params,
litellm_params=litellm_params,
headers=headers,
)
if self.supports_cache_control_ttl():
return request
return normalize_cache_control_in_anthropic_payload(request)
def get_complete_url(
self,
api_base: str | None,
@ -81,7 +119,7 @@ class JSONProviderAnthropicMessagesConfig(OpenAILikeAnthropicMessagesConfig):
"""
def __init__(self, provider: SimpleProviderConfig):
super().__init__()
super().__init__(cache_control_ttl=bool(provider.constraints.get("cache_control_ttl")))
self._provider = provider
@property

View file

@ -296,21 +296,15 @@ async def test_bedrock_converse_budget_tokens_preserved():
mock_acompletion.assert_called_once()
call_kwargs = mock_acompletion.call_args.kwargs
print(
"acompletion call kwargs: ", json.dumps(call_kwargs, indent=4, default=str)
)
print("acompletion call kwargs: ", json.dumps(call_kwargs, indent=4, default=str))
# Verify thinking parameter is passed through with budget_tokens preserved
thinking_param = call_kwargs.get("thinking")
assert (
thinking_param is not None
), "thinking parameter should be passed to acompletion"
assert (
thinking_param.get("type") == "enabled"
), "thinking.type should be 'enabled'"
assert (
thinking_param.get("budget_tokens") == 1024
), f"thinking.budget_tokens should be 1024, but got {thinking_param.get('budget_tokens')}"
assert thinking_param is not None, "thinking parameter should be passed to acompletion"
assert thinking_param.get("type") == "enabled", "thinking.type should be 'enabled'"
assert thinking_param.get("budget_tokens") == 1024, (
f"thinking.budget_tokens should be 1024, but got {thinking_param.get('budget_tokens')}"
)
def test_openai_model_with_thinking_converts_to_reasoning():
@ -342,23 +336,18 @@ def test_openai_model_with_thinking_converts_to_reasoning():
call_kwargs = mock_responses.call_args.kwargs
# Verify reasoning is set (converted from thinking)
assert (
"reasoning" in call_kwargs
), "reasoning should be passed to litellm.responses"
assert "reasoning" in call_kwargs, "reasoning should be passed to litellm.responses"
# budget_tokens=1024 -> effort="low" (at the LOW budget threshold)
# reasoning_auto_summary is False by default, so no summary key
expected_reasoning = {"effort": "low"}
assert call_kwargs["reasoning"] == expected_reasoning, (
f"reasoning should be {expected_reasoning} for budget_tokens=1024, "
f"got {call_kwargs.get('reasoning')}"
f"reasoning should be {expected_reasoning} for budget_tokens=1024, got {call_kwargs.get('reasoning')}"
)
assert "summary" not in call_kwargs["reasoning"]
# Verify thinking is NOT passed directly to the Responses API
assert (
"thinking" not in call_kwargs
), "thinking should NOT be passed directly to litellm.responses"
assert "thinking" not in call_kwargs, "thinking should NOT be passed directly to litellm.responses"
class TestThinkingParameterTransformation:
@ -411,9 +400,7 @@ class TestThinkingParameterTransformation:
thinking=thinking,
model="openai/gpt-5.2",
)
assert result == {
"reasoning_effort": {"effort": "high", "summary": "detailed"}
}
assert result == {"reasoning_effort": {"effort": "high", "summary": "detailed"}}
finally:
litellm.reasoning_auto_summary = original
@ -611,9 +598,9 @@ class TestThinkingSummaryPreservation:
mock_responses.assert_called_once()
call_kwargs = mock_responses.call_args.kwargs
reasoning = call_kwargs["reasoning"]
assert (
reasoning["summary"] == "concise"
), f"Expected summary='concise', got summary='{reasoning.get('summary')}'"
assert reasoning["summary"] == "concise", (
f"Expected summary='concise', got summary='{reasoning.get('summary')}'"
)
def test_responses_adapter_preserves_summary(self):
"""translate_thinking_to_reasoning should include summary when user provides it."""
@ -622,9 +609,7 @@ class TestThinkingSummaryPreservation:
)
thinking = {"type": "enabled", "budget_tokens": 5000, "summary": "concise"}
result = LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning(
thinking
)
result = LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning(thinking)
assert result == {"effort": "high", "summary": "concise"}
def test_responses_adapter_no_summary_by_default(self):
@ -638,11 +623,7 @@ class TestThinkingSummaryPreservation:
try:
litellm.reasoning_auto_summary = False
thinking = {"type": "enabled", "budget_tokens": 5000}
result = (
LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning(
thinking
)
)
result = LiteLLMAnthropicToResponsesAPIAdapter.translate_thinking_to_reasoning(thinking)
assert result == {"effort": "high"}
assert result is not None and "summary" not in result
finally:
@ -659,9 +640,7 @@ class TestThinkingSummaryPreservation:
thinking=thinking,
model="openai/gpt-5.2",
)
assert result == {
"reasoning_effort": {"effort": "high", "summary": "concise"}
}
assert result == {"reasoning_effort": {"effort": "high", "summary": "concise"}}
def test_translate_thinking_for_model_disabled_stays_plain_string_when_auto_summary_enabled(self):
"""Disabled thinking must stay a plain string even when reasoning_auto_summary is on."""
@ -807,9 +786,7 @@ def test_presanitized_flag_not_leaked_to_provider_params():
def fake_base_handler(*args, **kwargs):
captured.update(kwargs)
captured["optional"] = kwargs.get(
"anthropic_messages_optional_request_params", {}
)
captured["optional"] = kwargs.get("anthropic_messages_optional_request_params", {})
return "stub"
with patch.object(
@ -974,6 +951,38 @@ def test_gate_passthrough_skipped_when_only_chat_completions_supported(monkeypat
assert "config" not in captured
@pytest.mark.parametrize(
"model_info, expected_ttl_support",
[
({"supported_endpoints": ["/v1/messages"]}, False),
({"supported_endpoints": ["/v1/messages"], "cache_control_ttl": True}, True),
({"supported_endpoints": ["/v1/messages"], "cache_control_ttl": "yes"}, False),
],
)
def test_gate_passthrough_forwards_cache_control_ttl_only_when_deployment_opts_in(
monkeypatch, model_info, expected_ttl_support
):
"""The passthrough config strips cache_control.ttl unless the deployment sets
model_info.cache_control_ttl to exactly true."""
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
anthropic_messages_handler,
)
captured, _ = _gate_stubs(monkeypatch)
result = anthropic_messages_handler(
max_tokens=100,
messages=[{"role": "user", "content": "Hello"}],
model="openai/some-model",
api_key="sk-test",
api_base="https://host/v1",
model_info=model_info,
)
assert result == "native-passthrough"
assert captured["config"].supports_cache_control_ttl() is expected_ttl_support
def test_first_party_claude_4_8_plus_cost_map_entries_carry_mid_conversation_system_flag():
"""Regional and provider-prefixed Claude 4.8+/5 entries carry
``supports_mid_conversation_system``, but the bare first-party keys
@ -987,9 +996,7 @@ def test_first_party_claude_4_8_plus_cost_map_entries_carry_mid_conversation_sys
import litellm
cost_map_path = os.path.join(
os.path.dirname(litellm.__file__), "model_prices_and_context_window_backup.json"
)
cost_map_path = os.path.join(os.path.dirname(litellm.__file__), "model_prices_and_context_window_backup.json")
with open(cost_map_path) as f:
cost_map = json.load(f)
rules = cost_map["fallback_generalizations"]["rules"]
@ -1028,9 +1035,7 @@ def test_first_party_claude_4_8_plus_cost_map_entries_carry_mid_conversation_sys
("perplexity/sonar", "sonar", "https://api.perplexity.ai/chat/completions"),
],
)
async def test_messages_strips_provider_prefix_exactly_once(
requested_model, expected_wire_model, expected_url
):
async def test_messages_strips_provider_prefix_exactly_once(requested_model, expected_wire_model, expected_url):
"""
BerriAI/litellm#37716: only the leading provider segment may be stripped on the way upstream.

View file

@ -318,3 +318,203 @@ def test_json_provider_messages_config_probes_capabilities_under_provider_slug()
)
assert JSONProviderAnthropicMessagesConfig(provider).custom_llm_provider == "exampleprovider"
assert OpenAILikeAnthropicMessagesConfig().custom_llm_provider == "anthropic"
def _cache_control_request_params() -> tuple[list, dict]:
messages = [
{
"role": "user",
"content": [
{
"type": "text",
"text": "write a regex for a US phone number",
"cache_control": {"type": "ephemeral", "ttl": "1h"},
}
],
}
]
optional_params = {
"max_tokens": 256,
"system": [
{
"type": "text",
"text": "You are Claude Code.",
"cache_control": {"type": "ephemeral", "ttl": "5m"},
}
],
"tools": [
{
"name": "lookup",
"input_schema": {"type": "object"},
"cache_control": {"type": "ephemeral", "ttl": "1h"},
}
],
}
return messages, optional_params
def test_request_strips_cache_control_ttl_everywhere(config):
"""Regression: Claude Code always sends ``cache_control: {type: ephemeral,
ttl: 1h}``, and strict non-Anthropic /v1/messages validators 400 the whole
request on the ttl extension (``cache_control.ttl: 1h is not supported``)."""
messages, optional_params = _cache_control_request_params()
payload = config.transform_anthropic_messages_request(
model="some-model",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert payload["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral"}
assert payload["system"][0]["cache_control"] == {"type": "ephemeral"}
assert payload["tools"][0]["cache_control"] == {"type": "ephemeral"}
assert messages[0]["content"][0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"}
def test_request_defaults_missing_cache_control_type_and_drops_non_dict(config):
payload = config.transform_anthropic_messages_request(
model="some-model",
messages=[
{
"role": "user",
"content": [
{"type": "text", "text": "a", "cache_control": {"ttl": "1h"}},
{"type": "text", "text": "b", "cache_control": None},
],
}
],
anthropic_messages_optional_request_params={"max_tokens": 64},
litellm_params=GenericLiteLLMParams(),
headers={},
)
blocks = payload["messages"][0]["content"]
assert blocks[0]["cache_control"] == {"type": "ephemeral"}
assert "cache_control" not in blocks[1]
def test_native_anthropic_config_keeps_cache_control_ttl():
"""Anthropic itself accepts ttl, so the normalization must stay scoped to
the OpenAI-like passthrough and never reach the native Anthropic path."""
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
messages, optional_params = _cache_control_request_params()
payload = AnthropicMessagesConfig().transform_anthropic_messages_request(
model="claude-sonnet-4-20250514",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert payload["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"}
assert payload["system"][0]["cache_control"] == {"type": "ephemeral", "ttl": "5m"}
def test_deployment_opt_in_keeps_cache_control_ttl():
config = OpenAILikeAnthropicMessagesConfig(cache_control_ttl=True)
payload = config.transform_anthropic_messages_request(
model="some-model",
messages=[
{
"role": "user",
"content": [{"type": "text", "text": "hi", "cache_control": {"type": "ephemeral", "ttl": "1h"}}],
}
],
anthropic_messages_optional_request_params={"max_tokens": 16},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert payload["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"}
def test_json_provider_constraint_opts_into_cache_control_ttl():
from litellm.llms.openai_like.json_loader import SimpleProviderConfig
from litellm.llms.openai_like.messages.transformation import (
JSONProviderAnthropicMessagesConfig,
)
base_data = {"base_url": "https://api.example.com/v1", "api_key_env": "EXAMPLE_API_KEY"}
strict = JSONProviderAnthropicMessagesConfig(SimpleProviderConfig(slug="strictprov", data=base_data))
lenient = JSONProviderAnthropicMessagesConfig(
SimpleProviderConfig(slug="lenientprov", data={**base_data, "constraints": {"cache_control_ttl": True}})
)
def transform(provider_config):
messages, optional_params = _cache_control_request_params()
return provider_config.transform_anthropic_messages_request(
model="some-model",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert transform(strict)["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral"}
assert transform(lenient)["messages"][0]["content"][0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"}
def test_request_strips_ttl_only_where_the_messages_api_defines_cache_control(config):
"""Regression: the sanitizer must only touch ``cache_control`` where the
Messages API defines it (request, system, tools, content blocks, tool_result
content), never application data such as ``tool_use.input`` or a tool's
``input_schema`` that happens to contain a ``cache_control`` key."""
tool_input = {"cache_control": {"type": "ephemeral", "ttl": "1h"}, "query": "x"}
input_schema = {
"type": "object",
"properties": {"cache_control": {"type": "string", "ttl": "1h"}},
}
messages = [
{
"role": "assistant",
"content": [{"type": "tool_use", "id": "toolu_1", "name": "lookup", "input": tool_input}],
},
{
"role": "user",
"content": [
{
"type": "tool_result",
"tool_use_id": "toolu_1",
"cache_control": {"type": "ephemeral", "ttl": "1h"},
"content": [
{"type": "text", "text": "result", "cache_control": {"type": "ephemeral", "ttl": "1h"}}
],
},
{"type": "text", "text": "plain string content stays", "cache_control": {"ttl": "1h"}},
],
},
{"role": "user", "content": "a plain string message"},
]
optional_params = {
"max_tokens": 64,
"cache_control": {"type": "ephemeral", "ttl": "1h"},
"tools": [
{
"name": "lookup",
"input_schema": input_schema,
"cache_control": {"type": "ephemeral", "ttl": "1h"},
}
],
}
payload = config.transform_anthropic_messages_request(
model="some-model",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert payload["cache_control"] == {"type": "ephemeral"}
assert payload["tools"][0]["cache_control"] == {"type": "ephemeral"}
assert payload["tools"][0]["input_schema"] == input_schema
assert payload["messages"][0]["content"][0]["input"] == tool_input
tool_result = payload["messages"][1]["content"][0]
assert tool_result["cache_control"] == {"type": "ephemeral"}
assert tool_result["content"][0]["cache_control"] == {"type": "ephemeral"}
assert payload["messages"][1]["content"][1]["cache_control"] == {"type": "ephemeral"}
assert payload["messages"][2] == {"role": "user", "content": "a plain string message"}