mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(anthropic): request encrypted reasoning only where the Responses provider returns it
The bridge now asks for reasoning.encrypted_content whenever the provider's Responses config lists include, independent of the client's thinking block, and leaves it out for providers such as Perplexity that reject the param. Bridge-tagged blocks are stripped on the chat adapter path too, so a mid session model switch to Gemini or Bedrock no longer forwards them as real signatures, a bare prefix counts as bridge-tagged, and non-mapping messages pass through the strip untouched.
This commit is contained in:
parent
751431dc3f
commit
f49ebc93ea
10 changed files with 126 additions and 9 deletions
|
|
@ -1853,22 +1853,30 @@ def encrypted_reasoning_signature(encrypted_content: str) -> str:
|
|||
return f"{ENCRYPTED_REASONING_SIGNATURE_PREFIX}{encrypted_content}"
|
||||
|
||||
|
||||
def _carries_encrypted_reasoning(signature: object) -> bool:
|
||||
return isinstance(signature, str) and signature.startswith(ENCRYPTED_REASONING_SIGNATURE_PREFIX)
|
||||
|
||||
|
||||
def encrypted_content_from_signature(signature: object) -> str | None:
|
||||
if not isinstance(signature, str) or not signature.startswith(ENCRYPTED_REASONING_SIGNATURE_PREFIX):
|
||||
if not isinstance(signature, str) or not _carries_encrypted_reasoning(signature):
|
||||
return None
|
||||
return signature.removeprefix(ENCRYPTED_REASONING_SIGNATURE_PREFIX) or None
|
||||
|
||||
|
||||
def _encrypted_content_of_block(block: Mapping[str, object]) -> str | None:
|
||||
def _encrypted_reasoning_field(block: Mapping[str, object]) -> object:
|
||||
match block.get("type"):
|
||||
case "thinking":
|
||||
return encrypted_content_from_signature(block.get("signature"))
|
||||
return block.get("signature")
|
||||
case "redacted_thinking":
|
||||
return encrypted_content_from_signature(block.get("data"))
|
||||
return block.get("data")
|
||||
case _:
|
||||
return None
|
||||
|
||||
|
||||
def _encrypted_content_of_block(block: Mapping[str, object]) -> str | None:
|
||||
return encrypted_content_from_signature(_encrypted_reasoning_field(block))
|
||||
|
||||
|
||||
def is_encrypted_reasoning_block(block: object) -> bool:
|
||||
"""A thinking or redacted_thinking block carrying Responses API encrypted reasoning.
|
||||
|
||||
|
|
@ -1878,7 +1886,7 @@ def is_encrypted_reasoning_block(block: object) -> bool:
|
|||
if not isinstance(block, Mapping):
|
||||
return False
|
||||
mapping: Final = cast(Mapping[str, object], block) # cast-ok: narrowed by isinstance
|
||||
return _encrypted_content_of_block(mapping) is not None
|
||||
return _carries_encrypted_reasoning(_encrypted_reasoning_field(mapping))
|
||||
|
||||
|
||||
def _reasoning_replay_group_key(indexed_block: tuple[int, Mapping[str, object]]) -> str:
|
||||
|
|
|
|||
|
|
@ -1203,6 +1203,8 @@ def strip_thinking_blocks_from_anthropic_messages(messages: list[Any]) -> list[A
|
|||
|
||||
|
||||
def _without_encrypted_reasoning_blocks(message: dict) -> dict | None: # mutable-ok: Anthropic message payload shape
|
||||
if not isinstance(message, Mapping):
|
||||
return message
|
||||
content: Final = message.get("content")
|
||||
if not isinstance(content, list):
|
||||
return message
|
||||
|
|
|
|||
|
|
@ -113,6 +113,7 @@ from litellm.litellm_core_utils.reasoning_effort_utils import (
|
|||
from litellm.llms.anthropic.common_utils import (
|
||||
is_empty_unsigned_thinking_block,
|
||||
normalize_anthropic_tool_use_id,
|
||||
strip_encrypted_reasoning_blocks_from_anthropic_messages,
|
||||
)
|
||||
from litellm.llms.anthropic.experimental_pass_through.context_management import (
|
||||
PolyfillResult,
|
||||
|
|
@ -417,7 +418,8 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
model: str | None = None,
|
||||
) -> list:
|
||||
new_messages: Final[list[AllMessageValues]] = []
|
||||
for m in messages:
|
||||
replayable_messages: Final = strip_encrypted_reasoning_blocks_from_anthropic_messages(messages)
|
||||
for m in replayable_messages:
|
||||
user_message: ChatCompletionUserMessage | None = None
|
||||
tool_message_list: list[ChatCompletionToolMessage] = []
|
||||
new_user_content_list: list[ChatCompletionTextObject | ChatCompletionImageObject] = []
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
|
|||
AnthropicMessagesResponse,
|
||||
)
|
||||
from litellm.types.llms.openai import ResponsesAPIResponse
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
from ..utils import litellm_logging_obj_from_kwargs, local_model_name
|
||||
from .streaming_iterator import AnthropicResponsesStreamWrapper
|
||||
|
|
@ -34,6 +35,15 @@ def _forwarded_kwargs(extra_kwargs: Mapping[str, object] | None) -> Mapping[str,
|
|||
return extra_kwargs or {}
|
||||
|
||||
|
||||
def _provider_returns_encrypted_reasoning(model: str, custom_llm_provider: object) -> bool:
|
||||
provider: Final = (
|
||||
custom_llm_provider if isinstance(custom_llm_provider, str) else litellm.get_llm_provider(model=model)[1]
|
||||
)
|
||||
provider_model: Final = local_model_name(model, provider)
|
||||
responses_config: Final = ProviderConfigManager.get_provider_responses_api_config(provider, provider_model)
|
||||
return responses_config is not None and "include" in responses_config.get_supported_openai_params(provider_model)
|
||||
|
||||
|
||||
def _build_responses_kwargs(
|
||||
*,
|
||||
max_tokens: int,
|
||||
|
|
@ -85,8 +95,13 @@ def _build_responses_kwargs(
|
|||
request_data["output_format"] = output_format
|
||||
|
||||
anthropic_request: Final = AnthropicMessagesRequest(**request_data)
|
||||
responses_kwargs: Final = _ADAPTER.translate_request(anthropic_request)
|
||||
forwarded_kwargs: Final = _forwarded_kwargs(extra_kwargs)
|
||||
responses_kwargs: Final = _ADAPTER.translate_request(
|
||||
anthropic_request,
|
||||
include_encrypted_reasoning=_provider_returns_encrypted_reasoning(
|
||||
model, forwarded_kwargs.get("custom_llm_provider")
|
||||
),
|
||||
)
|
||||
|
||||
# Normalize reasoning effort based on model capabilities
|
||||
# (e.g. "max" → "xhigh"/"high", "minimal" → "low" if unsupported)
|
||||
|
|
|
|||
|
|
@ -505,10 +505,16 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
def translate_request(
|
||||
self,
|
||||
anthropic_request: AnthropicMessagesRequest,
|
||||
include_encrypted_reasoning: bool = True,
|
||||
) -> dict[str, Any]:
|
||||
"""
|
||||
Translate a full Anthropic /v1/messages request dict to
|
||||
litellm.responses() / litellm.aresponses() kwargs.
|
||||
|
||||
``include_encrypted_reasoning`` asks the provider for ``reasoning.encrypted_content``
|
||||
on every call, so a reasoning model's items can be replayed intact next turn even
|
||||
when the client sent no ``thinking`` block; pass False for a provider whose
|
||||
Responses API rejects ``include``.
|
||||
"""
|
||||
model: Final[str] = anthropic_request["model"]
|
||||
messages_list: Final = cast(
|
||||
|
|
@ -538,6 +544,8 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
"model": model,
|
||||
"input": input_items,
|
||||
}
|
||||
if include_encrypted_reasoning:
|
||||
responses_kwargs["include"] = [RESPONSES_INCLUDE_ENCRYPTED_REASONING] # mutable-ok: API request payload
|
||||
|
||||
if system and not developer_parts:
|
||||
if isinstance(system, str):
|
||||
|
|
@ -582,7 +590,6 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
)
|
||||
if reasoning:
|
||||
responses_kwargs["reasoning"] = reasoning
|
||||
responses_kwargs["include"] = [RESPONSES_INCLUDE_ENCRYPTED_REASONING] # mutable-ok: json list
|
||||
|
||||
# output_format / output_config.format -> text format
|
||||
# output_format: {"type": "json_schema", "schema": {...}}
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ import pytest
|
|||
|
||||
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
ENCRYPTED_REASONING_SIGNATURE_PREFIX,
|
||||
TOOL_RESULT_IMAGE_BOUNDARY,
|
||||
TOOL_RESULT_IMAGE_PLACEHOLDER,
|
||||
add_system_prompt_to_messages,
|
||||
|
|
@ -1610,6 +1611,8 @@ class TestEncryptedReasoningReplay:
|
|||
[
|
||||
({"type": "thinking", "thinking": "x", "signature": encrypted_reasoning_signature("g")}, True),
|
||||
({"type": "redacted_thinking", "data": encrypted_reasoning_signature("g")}, True),
|
||||
({"type": "thinking", "thinking": "x", "signature": ENCRYPTED_REASONING_SIGNATURE_PREFIX}, True),
|
||||
({"type": "redacted_thinking", "data": ENCRYPTED_REASONING_SIGNATURE_PREFIX}, True),
|
||||
({"type": "thinking", "thinking": "x", "signature": "ErcBCkgIValid"}, False),
|
||||
({"type": "redacted_thinking", "data": "EmwKAhgBEgy"}, False),
|
||||
({"type": "text", "text": encrypted_reasoning_signature("g")}, False),
|
||||
|
|
|
|||
|
|
@ -9,6 +9,7 @@ import litellm
|
|||
|
||||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
TOOL_RESULT_IMAGE_PLACEHOLDER,
|
||||
encrypted_reasoning_signature,
|
||||
)
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
THOUGHT_SIGNATURE_SEPARATOR,
|
||||
|
|
@ -407,6 +408,43 @@ def test_translate_anthropic_messages_to_openai_thinking_blocks():
|
|||
assert result[1]["tool_calls"][0]["id"] == "toolu_01234"
|
||||
|
||||
|
||||
def test_translate_anthropic_messages_to_openai_drops_bridge_encrypted_reasoning_blocks():
|
||||
"""A session that moves from an OpenAI reasoning model to a chat provider replays reasoning only OpenAI can read.
|
||||
|
||||
Gemini rejects the whole request when such a block reaches it as a thought_signature, so the
|
||||
adapter drops those blocks and keeps the provider-signed ones.
|
||||
"""
|
||||
|
||||
anthropic_messages = [
|
||||
AnthropicMessagesUserMessageParam(
|
||||
role="user",
|
||||
content=[{"type": "text", "text": "Who drinks water?"}],
|
||||
),
|
||||
AnthopicMessagesAssistantMessageParam(
|
||||
role="assistant",
|
||||
content=[
|
||||
{"type": "thinking", "thinking": "plan", "signature": encrypted_reasoning_signature("gAAAA_1")},
|
||||
{"type": "redacted_thinking", "data": encrypted_reasoning_signature("gAAAA_2")},
|
||||
{"type": "text", "text": "The Norwegian."},
|
||||
],
|
||||
),
|
||||
AnthopicMessagesAssistantMessageParam(
|
||||
role="assistant",
|
||||
content=[
|
||||
{"type": "thinking", "thinking": "native", "signature": "EqQBCkYIAxgCIkA_signed"},
|
||||
{"type": "text", "text": "Still the Norwegian."},
|
||||
],
|
||||
),
|
||||
]
|
||||
|
||||
result = LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(messages=anthropic_messages)
|
||||
|
||||
assert [m["role"] for m in result] == ["user", "assistant", "assistant"]
|
||||
assert not result[1].get("thinking_blocks")
|
||||
assert result[1]["content"] == "The Norwegian."
|
||||
assert [b["signature"] for b in result[2]["thinking_blocks"]] == ["EqQBCkYIAxgCIkA_signed"]
|
||||
|
||||
|
||||
def test_translate_anthropic_messages_to_openai_sets_reasoning_content():
|
||||
"""Reasoning-aware chat providers read reasoning_content, so thinking text must land there.
|
||||
|
||||
|
|
|
|||
|
|
@ -54,6 +54,29 @@ def test_build_responses_kwargs_prefers_explicit_prompt_cache_key_over_derived()
|
|||
assert responses_kwargs["prompt_cache_key"] == "explicit-key"
|
||||
|
||||
|
||||
def test_build_responses_kwargs_asks_openai_for_encrypted_reasoning_without_thinking():
|
||||
responses_kwargs = _build_responses_kwargs(
|
||||
max_tokens=1024,
|
||||
messages=MESSAGES,
|
||||
model="openai/gpt-5.6-luna",
|
||||
extra_kwargs={"custom_llm_provider": "openai"},
|
||||
)
|
||||
assert responses_kwargs["include"] == ["reasoning.encrypted_content"]
|
||||
assert "reasoning" not in responses_kwargs
|
||||
|
||||
|
||||
def test_build_responses_kwargs_skips_include_for_a_responses_provider_that_rejects_it():
|
||||
responses_kwargs = _build_responses_kwargs(
|
||||
max_tokens=1024,
|
||||
messages=MESSAGES,
|
||||
model="perplexity/sonar",
|
||||
thinking={"type": "enabled", "budget_tokens": 4096},
|
||||
extra_kwargs={"custom_llm_provider": "perplexity"},
|
||||
)
|
||||
assert "include" not in responses_kwargs
|
||||
assert "reasoning" in responses_kwargs
|
||||
|
||||
|
||||
def test_build_responses_kwargs_without_metadata_sets_no_prompt_cache_key():
|
||||
responses_kwargs = _build_responses_kwargs(
|
||||
max_tokens=1024,
|
||||
|
|
|
|||
|
|
@ -1162,7 +1162,6 @@ class TestTranslateRequestBroaderCoverage:
|
|||
req = _make_request(thinking={"type": "disabled"})
|
||||
kwargs = _ADAPTER.translate_request(req)
|
||||
assert "reasoning" not in kwargs
|
||||
assert "include" not in kwargs
|
||||
|
||||
def test_thinking_asks_for_the_encrypted_reasoning(self):
|
||||
"""The documented way to get reasoning that survives store=false is to ask for it."""
|
||||
|
|
@ -1170,6 +1169,17 @@ class TestTranslateRequestBroaderCoverage:
|
|||
kwargs = _ADAPTER.translate_request(req)
|
||||
assert kwargs["include"] == ["reasoning.encrypted_content"]
|
||||
|
||||
def test_encrypted_reasoning_is_asked_for_without_a_thinking_block(self):
|
||||
"""A reasoning model reasons whether or not the client sent `thinking`, so the replay needs it either way."""
|
||||
kwargs = _ADAPTER.translate_request(_make_request())
|
||||
assert kwargs["include"] == ["reasoning.encrypted_content"]
|
||||
|
||||
def test_encrypted_reasoning_is_not_asked_for_when_the_provider_rejects_include(self):
|
||||
req = _make_request(thinking={"type": "enabled", "budget_tokens": 12000})
|
||||
kwargs = _ADAPTER.translate_request(req, include_encrypted_reasoning=False)
|
||||
assert kwargs["reasoning"] == {"effort": "high"}
|
||||
assert "include" not in kwargs
|
||||
|
||||
def test_metadata_user_id_mapped_to_user(self):
|
||||
req = _make_request(metadata={"user_id": "user-42"})
|
||||
kwargs = _ADAPTER.translate_request(req)
|
||||
|
|
|
|||
|
|
@ -1600,6 +1600,15 @@ class TestAnthropicThinkingSignatureSelfHeal:
|
|||
assert len(msgs[1]["content"]) == 2
|
||||
assert len(msgs[2]["content"]) == 4
|
||||
|
||||
def test_strip_encrypted_reasoning_leaves_malformed_messages_for_the_provider_to_reject(self):
|
||||
"""A bare string in messages must reach Anthropic as a 400, not die in the stripper as a 500."""
|
||||
from litellm.llms.anthropic.common_utils import (
|
||||
strip_encrypted_reasoning_blocks_from_anthropic_messages,
|
||||
)
|
||||
|
||||
msgs = ["hi", {"role": "user", "content": "hello"}]
|
||||
assert strip_encrypted_reasoning_blocks_from_anthropic_messages(msgs) == msgs
|
||||
|
||||
def test_strip_empty_text_blocks_treats_null_text_as_empty(self):
|
||||
from litellm.llms.anthropic.common_utils import (
|
||||
strip_empty_content_blocks_from_anthropic_messages,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue