fix(anthropic): request encrypted reasoning only where the Responses provider returns it

The bridge now asks for reasoning.encrypted_content whenever the provider's
Responses config lists include, independent of the client's thinking block,
and leaves it out for providers such as Perplexity that reject the param.
Bridge-tagged blocks are stripped on the chat adapter path too, so a mid
session model switch to Gemini or Bedrock no longer forwards them as real
signatures, a bare prefix counts as bridge-tagged, and non-mapping messages
pass through the strip untouched.
This commit is contained in:
mateo-berri 2026-09-09 14:11:43 -07:00
parent 751431dc3f
commit f49ebc93ea
10 changed files with 126 additions and 9 deletions

View file

@ -1853,22 +1853,30 @@ def encrypted_reasoning_signature(encrypted_content: str) -> str:
return f"{ENCRYPTED_REASONING_SIGNATURE_PREFIX}{encrypted_content}"
def _carries_encrypted_reasoning(signature: object) -> bool:
return isinstance(signature, str) and signature.startswith(ENCRYPTED_REASONING_SIGNATURE_PREFIX)
def encrypted_content_from_signature(signature: object) -> str | None:
if not isinstance(signature, str) or not signature.startswith(ENCRYPTED_REASONING_SIGNATURE_PREFIX):
if not isinstance(signature, str) or not _carries_encrypted_reasoning(signature):
return None
return signature.removeprefix(ENCRYPTED_REASONING_SIGNATURE_PREFIX) or None
def _encrypted_content_of_block(block: Mapping[str, object]) -> str | None:
def _encrypted_reasoning_field(block: Mapping[str, object]) -> object:
match block.get("type"):
case "thinking":
return encrypted_content_from_signature(block.get("signature"))
return block.get("signature")
case "redacted_thinking":
return encrypted_content_from_signature(block.get("data"))
return block.get("data")
case _:
return None
def _encrypted_content_of_block(block: Mapping[str, object]) -> str | None:
return encrypted_content_from_signature(_encrypted_reasoning_field(block))
def is_encrypted_reasoning_block(block: object) -> bool:
"""A thinking or redacted_thinking block carrying Responses API encrypted reasoning.
@ -1878,7 +1886,7 @@ def is_encrypted_reasoning_block(block: object) -> bool:
if not isinstance(block, Mapping):
return False
mapping: Final = cast(Mapping[str, object], block) # cast-ok: narrowed by isinstance
return _encrypted_content_of_block(mapping) is not None
return _carries_encrypted_reasoning(_encrypted_reasoning_field(mapping))
def _reasoning_replay_group_key(indexed_block: tuple[int, Mapping[str, object]]) -> str:

View file

@ -1203,6 +1203,8 @@ def strip_thinking_blocks_from_anthropic_messages(messages: list[Any]) -> list[A
def _without_encrypted_reasoning_blocks(message: dict) -> dict | None: # mutable-ok: Anthropic message payload shape
if not isinstance(message, Mapping):
return message
content: Final = message.get("content")
if not isinstance(content, list):
return message

View file

@ -113,6 +113,7 @@ from litellm.litellm_core_utils.reasoning_effort_utils import (
from litellm.llms.anthropic.common_utils import (
is_empty_unsigned_thinking_block,
normalize_anthropic_tool_use_id,
strip_encrypted_reasoning_blocks_from_anthropic_messages,
)
from litellm.llms.anthropic.experimental_pass_through.context_management import (
PolyfillResult,
@ -417,7 +418,8 @@ class LiteLLMAnthropicMessagesAdapter:
model: str | None = None,
) -> list:
new_messages: Final[list[AllMessageValues]] = []
for m in messages:
replayable_messages: Final = strip_encrypted_reasoning_blocks_from_anthropic_messages(messages)
for m in replayable_messages:
user_message: ChatCompletionUserMessage | None = None
tool_message_list: list[ChatCompletionToolMessage] = []
new_user_content_list: list[ChatCompletionTextObject | ChatCompletionImageObject] = []

View file

@ -19,6 +19,7 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicMessagesResponse,
)
from litellm.types.llms.openai import ResponsesAPIResponse
from litellm.utils import ProviderConfigManager
from ..utils import litellm_logging_obj_from_kwargs, local_model_name
from .streaming_iterator import AnthropicResponsesStreamWrapper
@ -34,6 +35,15 @@ def _forwarded_kwargs(extra_kwargs: Mapping[str, object] | None) -> Mapping[str,
return extra_kwargs or {}
def _provider_returns_encrypted_reasoning(model: str, custom_llm_provider: object) -> bool:
provider: Final = (
custom_llm_provider if isinstance(custom_llm_provider, str) else litellm.get_llm_provider(model=model)[1]
)
provider_model: Final = local_model_name(model, provider)
responses_config: Final = ProviderConfigManager.get_provider_responses_api_config(provider, provider_model)
return responses_config is not None and "include" in responses_config.get_supported_openai_params(provider_model)
def _build_responses_kwargs(
*,
max_tokens: int,
@ -85,8 +95,13 @@ def _build_responses_kwargs(
request_data["output_format"] = output_format
anthropic_request: Final = AnthropicMessagesRequest(**request_data)
responses_kwargs: Final = _ADAPTER.translate_request(anthropic_request)
forwarded_kwargs: Final = _forwarded_kwargs(extra_kwargs)
responses_kwargs: Final = _ADAPTER.translate_request(
anthropic_request,
include_encrypted_reasoning=_provider_returns_encrypted_reasoning(
model, forwarded_kwargs.get("custom_llm_provider")
),
)
# Normalize reasoning effort based on model capabilities
# (e.g. "max" → "xhigh"/"high", "minimal" → "low" if unsupported)

View file

@ -505,10 +505,16 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
def translate_request(
self,
anthropic_request: AnthropicMessagesRequest,
include_encrypted_reasoning: bool = True,
) -> dict[str, Any]:
"""
Translate a full Anthropic /v1/messages request dict to
litellm.responses() / litellm.aresponses() kwargs.
``include_encrypted_reasoning`` asks the provider for ``reasoning.encrypted_content``
on every call, so a reasoning model's items can be replayed intact next turn even
when the client sent no ``thinking`` block; pass False for a provider whose
Responses API rejects ``include``.
"""
model: Final[str] = anthropic_request["model"]
messages_list: Final = cast(
@ -538,6 +544,8 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
"model": model,
"input": input_items,
}
if include_encrypted_reasoning:
responses_kwargs["include"] = [RESPONSES_INCLUDE_ENCRYPTED_REASONING] # mutable-ok: API request payload
if system and not developer_parts:
if isinstance(system, str):
@ -582,7 +590,6 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
)
if reasoning:
responses_kwargs["reasoning"] = reasoning
responses_kwargs["include"] = [RESPONSES_INCLUDE_ENCRYPTED_REASONING] # mutable-ok: json list
# output_format / output_config.format -> text format
# output_format: {"type": "json_schema", "schema": {...}}

View file

@ -7,6 +7,7 @@ import pytest
from litellm.litellm_core_utils.prompt_templates.common_utils import (
ENCRYPTED_REASONING_SIGNATURE_PREFIX,
TOOL_RESULT_IMAGE_BOUNDARY,
TOOL_RESULT_IMAGE_PLACEHOLDER,
add_system_prompt_to_messages,
@ -1610,6 +1611,8 @@ class TestEncryptedReasoningReplay:
[
({"type": "thinking", "thinking": "x", "signature": encrypted_reasoning_signature("g")}, True),
({"type": "redacted_thinking", "data": encrypted_reasoning_signature("g")}, True),
({"type": "thinking", "thinking": "x", "signature": ENCRYPTED_REASONING_SIGNATURE_PREFIX}, True),
({"type": "redacted_thinking", "data": ENCRYPTED_REASONING_SIGNATURE_PREFIX}, True),
({"type": "thinking", "thinking": "x", "signature": "ErcBCkgIValid"}, False),
({"type": "redacted_thinking", "data": "EmwKAhgBEgy"}, False),
({"type": "text", "text": encrypted_reasoning_signature("g")}, False),

View file

@ -9,6 +9,7 @@ import litellm
from litellm.litellm_core_utils.prompt_templates.common_utils import (
TOOL_RESULT_IMAGE_PLACEHOLDER,
encrypted_reasoning_signature,
)
from litellm.litellm_core_utils.prompt_templates.factory import (
THOUGHT_SIGNATURE_SEPARATOR,
@ -407,6 +408,43 @@ def test_translate_anthropic_messages_to_openai_thinking_blocks():
assert result[1]["tool_calls"][0]["id"] == "toolu_01234"
def test_translate_anthropic_messages_to_openai_drops_bridge_encrypted_reasoning_blocks():
"""A session that moves from an OpenAI reasoning model to a chat provider replays reasoning only OpenAI can read.
Gemini rejects the whole request when such a block reaches it as a thought_signature, so the
adapter drops those blocks and keeps the provider-signed ones.
"""
anthropic_messages = [
AnthropicMessagesUserMessageParam(
role="user",
content=[{"type": "text", "text": "Who drinks water?"}],
),
AnthopicMessagesAssistantMessageParam(
role="assistant",
content=[
{"type": "thinking", "thinking": "plan", "signature": encrypted_reasoning_signature("gAAAA_1")},
{"type": "redacted_thinking", "data": encrypted_reasoning_signature("gAAAA_2")},
{"type": "text", "text": "The Norwegian."},
],
),
AnthopicMessagesAssistantMessageParam(
role="assistant",
content=[
{"type": "thinking", "thinking": "native", "signature": "EqQBCkYIAxgCIkA_signed"},
{"type": "text", "text": "Still the Norwegian."},
],
),
]
result = LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(messages=anthropic_messages)
assert [m["role"] for m in result] == ["user", "assistant", "assistant"]
assert not result[1].get("thinking_blocks")
assert result[1]["content"] == "The Norwegian."
assert [b["signature"] for b in result[2]["thinking_blocks"]] == ["EqQBCkYIAxgCIkA_signed"]
def test_translate_anthropic_messages_to_openai_sets_reasoning_content():
"""Reasoning-aware chat providers read reasoning_content, so thinking text must land there.

View file

@ -54,6 +54,29 @@ def test_build_responses_kwargs_prefers_explicit_prompt_cache_key_over_derived()
assert responses_kwargs["prompt_cache_key"] == "explicit-key"
def test_build_responses_kwargs_asks_openai_for_encrypted_reasoning_without_thinking():
responses_kwargs = _build_responses_kwargs(
max_tokens=1024,
messages=MESSAGES,
model="openai/gpt-5.6-luna",
extra_kwargs={"custom_llm_provider": "openai"},
)
assert responses_kwargs["include"] == ["reasoning.encrypted_content"]
assert "reasoning" not in responses_kwargs
def test_build_responses_kwargs_skips_include_for_a_responses_provider_that_rejects_it():
responses_kwargs = _build_responses_kwargs(
max_tokens=1024,
messages=MESSAGES,
model="perplexity/sonar",
thinking={"type": "enabled", "budget_tokens": 4096},
extra_kwargs={"custom_llm_provider": "perplexity"},
)
assert "include" not in responses_kwargs
assert "reasoning" in responses_kwargs
def test_build_responses_kwargs_without_metadata_sets_no_prompt_cache_key():
responses_kwargs = _build_responses_kwargs(
max_tokens=1024,

View file

@ -1162,7 +1162,6 @@ class TestTranslateRequestBroaderCoverage:
req = _make_request(thinking={"type": "disabled"})
kwargs = _ADAPTER.translate_request(req)
assert "reasoning" not in kwargs
assert "include" not in kwargs
def test_thinking_asks_for_the_encrypted_reasoning(self):
"""The documented way to get reasoning that survives store=false is to ask for it."""
@ -1170,6 +1169,17 @@ class TestTranslateRequestBroaderCoverage:
kwargs = _ADAPTER.translate_request(req)
assert kwargs["include"] == ["reasoning.encrypted_content"]
def test_encrypted_reasoning_is_asked_for_without_a_thinking_block(self):
"""A reasoning model reasons whether or not the client sent `thinking`, so the replay needs it either way."""
kwargs = _ADAPTER.translate_request(_make_request())
assert kwargs["include"] == ["reasoning.encrypted_content"]
def test_encrypted_reasoning_is_not_asked_for_when_the_provider_rejects_include(self):
req = _make_request(thinking={"type": "enabled", "budget_tokens": 12000})
kwargs = _ADAPTER.translate_request(req, include_encrypted_reasoning=False)
assert kwargs["reasoning"] == {"effort": "high"}
assert "include" not in kwargs
def test_metadata_user_id_mapped_to_user(self):
req = _make_request(metadata={"user_id": "user-42"})
kwargs = _ADAPTER.translate_request(req)

View file

@ -1600,6 +1600,15 @@ class TestAnthropicThinkingSignatureSelfHeal:
assert len(msgs[1]["content"]) == 2
assert len(msgs[2]["content"]) == 4
def test_strip_encrypted_reasoning_leaves_malformed_messages_for_the_provider_to_reject(self):
"""A bare string in messages must reach Anthropic as a 400, not die in the stripper as a 500."""
from litellm.llms.anthropic.common_utils import (
strip_encrypted_reasoning_blocks_from_anthropic_messages,
)
msgs = ["hi", {"role": "user", "content": "hello"}]
assert strip_encrypted_reasoning_blocks_from_anthropic_messages(msgs) == msgs
def test_strip_empty_text_blocks_treats_null_text_as_empty(self):
from litellm.llms.anthropic.common_utils import (
strip_empty_content_blocks_from_anthropic_messages,