fix(anthropic): replay OpenAI encrypted reasoning byte for byte behind /v1/messages

Reasoning items the Responses API returned for a /v1/messages turn were rebuilt
from their summary text on every replay, so the prompt the model saw changed
between turns and the prompt cache never matched. The bridge now asks for
reasoning.encrypted_content, carries it in the thinking signature (or as a
redacted_thinking block when there is no summary), and replays it verbatim as
the reasoning item's encrypted_content. Anthropic replay paths drop those
tagged blocks so a cross-model resume never forwards OpenAI bytes to Anthropic
This commit is contained in:
mateo-berri 2026-09-08 20:27:14 -07:00
parent 075655c7ee
commit 2bca7ff673
11 changed files with 535 additions and 66 deletions

View file

@ -25,7 +25,7 @@ import litellm
from litellm import ModelResponse
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.prompt_templates.common_utils import (
responses_reasoning_item_from_thinking_blocks,
responses_reasoning_items_from_thinking_blocks,
)
from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator
from litellm.llms.base_llm.bridges.completion_transformation import (
@ -129,8 +129,8 @@ def _reasoning_input_items(msg: "AllMessageValues") -> list[dict[str, object]]:
return stored
raw_blocks: Final = msg.get("thinking_blocks") or ()
blocks: Final = cast("Iterable[ChatCompletionThinkingBlock]", raw_blocks) # cast-ok: untyped client json
from_thinking: Final = responses_reasoning_item_from_thinking_blocks(blocks)
return [] if from_thinking is None else [dict(from_thinking)] # mutable-ok: API message payload
replayed: Final = responses_reasoning_items_from_thinking_blocks(blocks)
return [dict(item) for item in replayed] # mutable-ok: API message payload
def _build_reasoning_item(

View file

@ -1823,14 +1823,11 @@ def _extract_reasoning_content(message: dict) -> tuple[str | None, str | None]:
return None, message_content
def _readable_thinking_text(
block: ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock,
) -> str:
def _readable_thinking_text(block: Mapping[str, object]) -> str:
"""The text a chat model can read back, empty for redacted blocks and malformed ones."""
if block.get("type") != "thinking":
return ""
thinking: Final = cast(ChatCompletionThinkingBlock, block).get("thinking") # cast-ok: narrowed by the type tag
return str(thinking or "")
return str(block.get("thinking") or "")
def reasoning_content_from_thinking_blocks(
@ -1843,24 +1840,83 @@ def reasoning_content_from_thinking_blocks(
return "\n".join(text for block in thinking_blocks if (text := _readable_thinking_text(block)))
def responses_reasoning_item_from_thinking_blocks(
thinking_blocks: Iterable[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock],
) -> ChatCompletionReasoningItem | None:
"""Build a Responses API `reasoning` input item from Anthropic thinking blocks.
ENCRYPTED_REASONING_SIGNATURE_PREFIX: Final = "litellm_encrypted_reasoning:"
The item carries no `id`: the Responses API rejects an empty one and 404s on any id it
did not mint itself, while an item without an id is always accepted.
def encrypted_reasoning_signature(encrypted_content: str) -> str:
"""The opaque value a Responses API reasoning item's `encrypted_content` travels in.
Anthropic clients echo a thinking block's `signature` and a redacted block's `data`
back verbatim, so either field can carry the encrypted reasoning across turns; the
prefix tells the two apart from a signature Anthropic minted.
"""
return f"{ENCRYPTED_REASONING_SIGNATURE_PREFIX}{encrypted_content}"
def encrypted_content_from_signature(signature: object) -> str | None:
if not isinstance(signature, str) or not signature.startswith(ENCRYPTED_REASONING_SIGNATURE_PREFIX):
return None
return signature.removeprefix(ENCRYPTED_REASONING_SIGNATURE_PREFIX) or None
def _encrypted_content_of_block(block: Mapping[str, object]) -> str | None:
match block.get("type"):
case "thinking":
return encrypted_content_from_signature(block.get("signature"))
case "redacted_thinking":
return encrypted_content_from_signature(block.get("data"))
case _:
return None
def is_encrypted_reasoning_block(block: object) -> bool:
"""A thinking or redacted_thinking block carrying Responses API encrypted reasoning.
Only the Responses API that minted the content can read it back, so an Anthropic
backend has to drop such a block rather than fail signature verification on it.
"""
if not isinstance(block, Mapping):
return False
mapping: Final = cast(Mapping[str, object], block) # cast-ok: narrowed by isinstance
return _encrypted_content_of_block(mapping) is not None
def _reasoning_replay_group_key(indexed_block: tuple[int, Mapping[str, object]]) -> str:
index, block = indexed_block
return f"encrypted:{index}" if is_encrypted_reasoning_block(block) else "summary"
def _reasoning_item_from_block_group(group: tuple[Mapping[str, object], ...]) -> ChatCompletionReasoningItem | None:
summary: Final[list[ChatCompletionReasoningSummaryTextBlock]] = [ # mutable-ok: API message payload
ChatCompletionReasoningSummaryTextBlock(type="summary_text", text=text)
for block in thinking_blocks
for block in group
if (text := _readable_thinking_text(block))
]
encrypted_content: Final = _encrypted_content_of_block(group[0])
if encrypted_content is not None:
return ChatCompletionReasoningItem(type="reasoning", summary=summary, encrypted_content=encrypted_content)
if not summary:
return None
return ChatCompletionReasoningItem(type="reasoning", summary=summary)
def responses_reasoning_items_from_thinking_blocks(
thinking_blocks: Iterable[Mapping[str, object]],
) -> tuple[ChatCompletionReasoningItem, ...]:
"""Build Responses API `reasoning` input items from Anthropic thinking blocks.
A block carrying encrypted reasoning replays the item it came from byte for byte;
a run of plain thinking blocks collapses into one summary-only item. No item carries
an `id`: the Responses API 404s on any id it did not mint itself and rejects an empty
one, while an item without an id is always accepted.
"""
return tuple(
item
for _, group in groupby(enumerate(thinking_blocks), key=_reasoning_replay_group_key)
if (item := _reasoning_item_from_block_group(tuple(block for _, block in group))) is not None
)
def _parse_content_for_reasoning(
message_text: str | None,
) -> tuple[str | None, str | None]:

View file

@ -46,6 +46,7 @@ from litellm.types.utils import GenericImageParsingChunk
from .common_utils import (
convert_content_list_to_str,
infer_content_type_from_url_and_content,
is_encrypted_reasoning_block,
is_non_content_values_set,
parse_tool_call_arguments,
)
@ -2299,13 +2300,16 @@ def sanitize_messages_for_tool_calling(
def _is_unsignable_thinking_block(block: object) -> bool:
"""A `thinking` block that Anthropic cannot accept on input.
"""A thinking block that Anthropic cannot accept on input.
Anthropic verifies the thinking signature cryptographically, so a block whose
signature is null, empty, or missing (e.g. from an open-source reasoning model)
is rejected with a 400 and must be dropped rather than blanked or repaired.
`redacted_thinking` blocks carry no signature and are always kept.
is rejected with a 400 and must be dropped rather than blanked or repaired, and
so is a block whose signature or data carries another provider's encrypted
reasoning. A `redacted_thinking` block Anthropic minted is always kept.
"""
if is_encrypted_reasoning_block(block):
return True
if not isinstance(block, dict) or block.get("type") != "thinking":
return False
signature: Final = block.get("signature")

View file

@ -21,6 +21,7 @@ from litellm.constants import (
)
from litellm.litellm_core_utils.prompt_templates.common_utils import (
get_file_ids_from_messages,
is_encrypted_reasoning_block,
)
from litellm.litellm_core_utils.prompt_templates.factory import (
THOUGHT_SIGNATURE_SEPARATOR,
@ -1235,8 +1236,10 @@ def strip_empty_content_blocks_from_anthropic_messages(
on the unified ``/v1/messages`` path. ``/v1/chat/completions`` already
handles this in ``anthropic_messages_pt``; this helper provides the
equivalent guarantee for the native Anthropic Messages path.
``redacted_thinking`` blocks are never touched: they carry opaque
``data`` instead of thinking text.
A thinking or ``redacted_thinking`` block whose signature or data carries
another provider's encrypted reasoning (a turn served by the Responses API
bridge) is dropped too, since Anthropic cannot verify it; every other
``redacted_thinking`` block is left alone.
Messages whose content is a list and becomes empty after stripping are
omitted, matching :func:`strip_thinking_blocks_from_anthropic_messages`.
@ -1249,7 +1252,11 @@ def strip_empty_content_blocks_from_anthropic_messages(
out.append(m)
continue
content = m["content"]
filtered = [b for b in content if not _is_empty_text_block(b) and not is_empty_thinking_block(b)]
filtered = [ # mutable-ok: rebuilt message content list
b
for b in content
if not _is_empty_text_block(b) and not is_empty_thinking_block(b) and not is_encrypted_reasoning_block(b)
]
if len(filtered) == len(content):
out.append(m)
elif filtered:

View file

@ -9,13 +9,19 @@ from typing import TYPE_CHECKING, Any, Final
from litellm import verbose_logger
from litellm._uuid import uuid
from litellm.litellm_core_utils.prompt_templates.common_utils import (
encrypted_reasoning_signature,
)
from litellm.llms.anthropic.experimental_pass_through.messages.utils import (
refusal_stop_details,
responses_output_refusal_text,
)
from litellm.types.llms.anthropic_messages.anthropic_response import AnthropicUsage
from .transformation import LiteLLMAnthropicToResponsesAPIAdapter
from .transformation import (
REASONING_SUMMARY_PART_SEPARATOR,
LiteLLMAnthropicToResponsesAPIAdapter,
)
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObject
@ -29,9 +35,10 @@ class AnthropicResponsesStreamWrapper:
response.created -> message_start
response.output_item.added -> content_block_start (if message/function_call)
response.output_text.delta -> content_block_delta (text_delta)
response.reasoning_summary_part.added -> content_block_delta (thinking_delta separator)
response.reasoning_summary_text.delta -> content_block_delta (thinking_delta)
response.function_call_arguments.delta -> content_block_delta (input_json_delta)
response.output_item.done -> content_block_stop
response.output_item.done -> content_block_delta (signature_delta) + content_block_stop
response.completed -> message_delta + message_stop
"""
@ -94,6 +101,38 @@ class AnthropicResponsesStreamWrapper:
)
return block_idx
@staticmethod
def _field(source: object, name: str) -> object:
return source.get(name) if isinstance(source, dict) else getattr(source, name, None)
def _close_reasoning_item(self, item: object, item_id: str | None) -> None:
block_idx: Final = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index
encrypted_content: Final = self._field(item, "encrypted_content")
signature: Final = (
encrypted_reasoning_signature(encrypted_content)
if isinstance(encrypted_content, str) and encrypted_content
else None
)
if block_idx < 0 and signature is None:
return
if block_idx < 0:
redacted_idx: Final = self._open_block(
item_id,
{"type": "redacted_thinking", "data": signature}, # mutable-ok: API message payload
)
stop: Final = {"type": "content_block_stop", "index": redacted_idx} # mutable-ok: API message payload
self._chunk_queue.append(stop)
return
if signature is not None:
self._chunk_queue.append(
{ # mutable-ok: API message payload
"type": "content_block_delta",
"index": block_idx,
"delta": {"type": "signature_delta", "signature": signature}, # mutable-ok: API message payload
}
)
self._chunk_queue.append({"type": "content_block_stop", "index": block_idx}) # mutable-ok: API message payload
def _process_event(self, event: object) -> None:
"""Convert one Responses API event into zero or more Anthropic chunks queued for emission."""
event_type = getattr(event, "type", None)
@ -175,6 +214,26 @@ class AnthropicResponsesStreamWrapper:
)
return
if event_type == "response.reasoning_summary_part.added":
part_item_id: Final = self._field(event, "item_id")
summary_index: Final = self._field(event, "summary_index")
part_block_idx: Final = (
self._item_id_to_block_index.get(part_item_id, -1) if isinstance(part_item_id, str) else -1
)
if part_block_idx < 0 or not isinstance(summary_index, int) or summary_index == 0:
return
self._chunk_queue.append(
{ # mutable-ok: API message payload
"type": "content_block_delta",
"index": part_block_idx,
"delta": { # mutable-ok: API message payload
"type": "thinking_delta",
"thinking": REASONING_SUMMARY_PART_SEPARATOR,
},
}
)
return
# ---- reasoning summary text delta ----
if event_type == "response.reasoning_summary_text.delta":
item_id = getattr(event, "item_id", None) or (event.get("item_id") if isinstance(event, dict) else None)
@ -220,6 +279,9 @@ class AnthropicResponsesStreamWrapper:
item_id = (
getattr(item, "id", None) or (item.get("id") if isinstance(item, dict) else None) if item else None
)
if self._field(item, "type") == "reasoning":
self._close_reasoning_item(item, item_id)
return
block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index
if block_idx < 0:
return

View file

@ -13,7 +13,8 @@ from typing import Any, Final, cast
from litellm.litellm_core_utils.prompt_templates.common_utils import (
TOOL_RESULT_IMAGE_BOUNDARY,
TOOL_RESULT_IMAGE_PLACEHOLDER,
responses_reasoning_item_from_thinking_blocks,
encrypted_reasoning_signature,
responses_reasoning_items_from_thinking_blocks,
with_prompt_cache_breakpoint,
)
from litellm.litellm_core_utils.reasoning_effort_utils import (
@ -33,6 +34,7 @@ from litellm.types.llms.anthropic import (
AnthropicFinishReason,
AnthropicMessagesRequest,
AnthropicMessagesToolChoice,
AnthropicResponseContentBlockRedactedThinking,
AnthropicResponseContentBlockText,
AnthropicResponseContentBlockThinking,
AnthropicResponseContentBlockToolUse,
@ -43,11 +45,13 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
AnthropicUsage,
)
from litellm.types.llms.openai import (
ChatCompletionThinkingBlock,
ResponseAPIUsage,
ResponsesAPIResponse,
)
REASONING_SUMMARY_PART_SEPARATOR: Final = "\n\n"
RESPONSES_INCLUDE_ENCRYPTED_REASONING: Final = "reasoning.encrypted_content"
class LiteLLMAnthropicToResponsesAPIAdapter:
"""
@ -163,49 +167,55 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
return str(getattr(part, "text", None) or "")
@classmethod
def _thinking_blocks_from_reasoning_item(
def _thinking_block_from_reasoning_item(
cls,
summary: Iterable[object],
) -> tuple[dict[str, Any], ...]: # mutable-ok: API message payload
"""Anthropic thinking blocks for one Responses reasoning item.
encrypted_content: object,
) -> dict[str, Any] | None: # mutable-ok: API message payload
"""The one Anthropic block for a Responses reasoning item.
The signature stays empty: only Anthropic can sign a thinking block, and a stand-in
value would be replayed as a real one and rejected by every backend that verifies it.
The item's encrypted reasoning rides the block's opaque field (`signature`, or
`data` when there is no summary text) so the client echoes it back and the next
turn replays the very item OpenAI produced; without it the signature stays empty,
since only Anthropic can sign a thinking block.
"""
return tuple(
AnthropicResponseContentBlockThinking(
type="thinking",
thinking=text,
signature=None,
).model_dump()
for part in summary
if (text := cls._summary_part_text(part))
text: Final = REASONING_SUMMARY_PART_SEPARATOR.join(
part_text for part in summary if (part_text := cls._summary_part_text(part))
)
if not isinstance(encrypted_content, str) or not encrypted_content:
if not text:
return None
return AnthropicResponseContentBlockThinking(type="thinking", thinking=text, signature=None).model_dump()
signature: Final = encrypted_reasoning_signature(encrypted_content)
if not text:
return AnthropicResponseContentBlockRedactedThinking(type="redacted_thinking", data=signature).model_dump()
return AnthropicResponseContentBlockThinking(type="thinking", thinking=text, signature=signature).model_dump()
@staticmethod
def _assistant_block_group_key(indexed_block: tuple[int, Mapping[str, object]]) -> str:
"""Group a run of consecutive thinking blocks together; keep every other block alone."""
index, block = indexed_block
return "thinking" if block.get("type") == "thinking" else f"block:{index}"
return "thinking" if block.get("type") in ("thinking", "redacted_thinking") else f"block:{index}"
@classmethod
def _assistant_group_to_input_item(
def _assistant_group_to_input_items(
cls, group: tuple[Mapping[str, object], ...]
) -> dict[str, Any] | None: # mutable-ok: API message payload
) -> tuple[dict[str, Any], ...]: # mutable-ok: API message payload
first: Final = group[0]
btype: Final = first.get("type")
if btype == "thinking":
blocks: Final = cast(tuple[ChatCompletionThinkingBlock, ...], group) # cast-ok: untrusted client payload
reasoning_item: Final = responses_reasoning_item_from_thinking_blocks(blocks)
return None if reasoning_item is None else dict(reasoning_item) # mutable-ok: API message payload
if btype in ("thinking", "redacted_thinking"):
replayed: Final = responses_reasoning_items_from_thinking_blocks(group)
return tuple(dict(item) for item in replayed) # mutable-ok: API message payload
if btype == "tool_use":
return { # mutable-ok: API message payload
"type": "function_call",
"call_id": first.get("id", ""),
"name": first.get("name", ""),
"arguments": json.dumps(first.get("input", {})), # mutable-ok: API message payload
}
return None
return (
{ # mutable-ok: API message payload
"type": "function_call",
"call_id": first.get("id", ""),
"name": first.get("name", ""),
"arguments": json.dumps(first.get("input", {})), # mutable-ok: API message payload
},
)
return ()
def translate_messages_to_responses_input(
self,
@ -362,7 +372,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
input_items.extend(
item
for _, group in groupby(enumerate(blocks), key=self._assistant_block_group_key)
if (item := self._assistant_group_to_input_item(tuple(block for _, block in group))) is not None
for item in self._assistant_group_to_input_items(tuple(block for _, block in group))
)
asst_parts: list[dict[str, Any]] = [ # mutable-ok: API message payload
{"type": "output_text", "text": block.get("text", "")} # mutable-ok: API message payload
@ -572,6 +582,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
)
if reasoning:
responses_kwargs["reasoning"] = reasoning
responses_kwargs["include"] = [RESPONSES_INCLUDE_ENCRYPTED_REASONING] # mutable-ok: json list
# output_format / output_config.format -> text format
# output_format: {"type": "json_schema", "schema": {...}}
@ -634,7 +645,9 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
for item in response.output:
if isinstance(item, ResponseReasoningItem):
content.extend(self._thinking_blocks_from_reasoning_item(item.summary))
reasoning_block = self._thinking_block_from_reasoning_item(item.summary, item.encrypted_content)
if reasoning_block is not None:
content.append(reasoning_block)
elif isinstance(item, ResponseOutputMessage):
for part in item.content:
@ -684,11 +697,12 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
).model_dump()
)
elif item_type == "reasoning":
content.extend(
self._thinking_blocks_from_reasoning_item(
cast(Iterable[object], item.get("summary") or ()), # cast-ok: untyped provider json
)
reasoning_block = self._thinking_block_from_reasoning_item(
cast(Iterable[object], item.get("summary") or ()), # cast-ok: untyped provider json
item.get("encrypted_content"),
)
if reasoning_block is not None:
content.append(reasoning_block)
elif item_type == "function_call":
try:
input_data = json.loads(item.get("arguments", "{}"))

View file

@ -10,10 +10,14 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import (
TOOL_RESULT_IMAGE_BOUNDARY,
TOOL_RESULT_IMAGE_PLACEHOLDER,
add_system_prompt_to_messages,
encrypted_content_from_signature,
encrypted_reasoning_signature,
get_file_ids_from_messages,
get_format_from_file_id,
handle_any_messages_to_chat_completion_str_messages_conversion,
hoist_images_from_tool_messages,
is_encrypted_reasoning_block,
responses_reasoning_items_from_thinking_blocks,
split_concatenated_json_objects,
update_messages_with_model_file_ids,
)
@ -1554,3 +1558,63 @@ class TestRequestContainsImageContent:
for _ in range(50):
nested = {"type": "tool_result", "content": [nested]}
assert request_contains_image_content([{"role": "user", "content": [nested]}]) is False
class TestEncryptedReasoningReplay:
"""Regression for https://github.com/BerriAI/litellm/issues/40288."""
def test_signature_round_trips_the_encrypted_content(self):
assert encrypted_content_from_signature(encrypted_reasoning_signature("gAAAA_bytes")) == "gAAAA_bytes"
@pytest.mark.parametrize("signature", [None, "", "ErcBCkgIValidAnthropicSignature", "litellm_encrypted_reasoning:", 7])
def test_anything_else_is_not_encrypted_content(self, signature):
assert encrypted_content_from_signature(signature) is None
def test_encrypted_thinking_block_replays_its_own_item(self):
items = responses_reasoning_items_from_thinking_blocks(
[{"type": "thinking", "thinking": "Plan.", "signature": encrypted_reasoning_signature("gAAAA_1")}]
)
assert items == (
{"type": "reasoning", "summary": [{"type": "summary_text", "text": "Plan."}], "encrypted_content": "gAAAA_1"},
)
def test_encrypted_redacted_block_replays_with_an_empty_summary(self):
items = responses_reasoning_items_from_thinking_blocks(
[{"type": "redacted_thinking", "data": encrypted_reasoning_signature("gAAAA_1")}]
)
assert items == ({"type": "reasoning", "summary": [], "encrypted_content": "gAAAA_1"},)
def test_plain_blocks_collapse_into_one_summary_item_around_encrypted_ones(self):
items = responses_reasoning_items_from_thinking_blocks(
[
{"type": "thinking", "thinking": "A.", "signature": None},
{"type": "thinking", "thinking": "B.", "signature": ""},
{"type": "thinking", "thinking": "C.", "signature": encrypted_reasoning_signature("gAAAA_c")},
{"type": "redacted_thinking", "data": "anthropic-minted-opaque-data"},
{"type": "thinking", "thinking": "D."},
]
)
assert items == (
{"type": "reasoning", "summary": [{"type": "summary_text", "text": "A."}, {"type": "summary_text", "text": "B."}]},
{"type": "reasoning", "summary": [{"type": "summary_text", "text": "C."}], "encrypted_content": "gAAAA_c"},
{"type": "reasoning", "summary": [{"type": "summary_text", "text": "D."}]},
)
assert all("id" not in item for item in items)
def test_blocks_without_text_or_encrypted_content_produce_nothing(self):
assert responses_reasoning_items_from_thinking_blocks([{"type": "thinking", "thinking": ""}]) == ()
assert responses_reasoning_items_from_thinking_blocks([]) == ()
@pytest.mark.parametrize(
("block", "expected"),
[
({"type": "thinking", "thinking": "x", "signature": encrypted_reasoning_signature("g")}, True),
({"type": "redacted_thinking", "data": encrypted_reasoning_signature("g")}, True),
({"type": "thinking", "thinking": "x", "signature": "ErcBCkgIValid"}, False),
({"type": "redacted_thinking", "data": "EmwKAhgBEgy"}, False),
({"type": "text", "text": encrypted_reasoning_signature("g")}, False),
("not a block", False),
],
)
def test_is_encrypted_reasoning_block(self, block, expected):
assert is_encrypted_reasoning_block(block) is expected

View file

@ -191,8 +191,16 @@ def test_bedrock_converse_assistant_with_empty_thinking_block_and_tool_calls():
{"type": "thinking", "thinking": "oss reasoning", "signature": None},
{"type": "thinking", "thinking": "oss reasoning", "signature": ""},
{"type": "thinking", "thinking": "oss reasoning"},
{"type": "thinking", "thinking": "openai reasoning", "signature": "litellm_encrypted_reasoning:gAAAA"},
{"type": "redacted_thinking", "data": "litellm_encrypted_reasoning:gAAAA"},
],
ids=[
"null_signature",
"empty_signature",
"missing_signature",
"encrypted_reasoning_signature",
"encrypted_reasoning_redacted_data",
],
ids=["null_signature", "empty_signature", "missing_signature"],
)
def test_anthropic_messages_pt_drops_unsignable_thinking_block(thinking_block):
"""Open-source reasoning models (DeepSeek-R1, Qwen, etc.) emit thinking blocks
@ -219,7 +227,7 @@ def test_anthropic_messages_pt_drops_unsignable_thinking_block(thinking_block):
assistant = next(m for m in result if m["role"] == "assistant")
content = assistant["content"]
assert all(
block.get("type") != "thinking" for block in content
block.get("type") not in ("thinking", "redacted_thinking") for block in content
), f"unsignable thinking block must be dropped, got {content!r}"
assert any(
block.get("type") == "text" and block.get("text") == "2+2 equals 4."

View file

@ -10,6 +10,9 @@ from types import SimpleNamespace
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../..")))
from litellm.litellm_core_utils.prompt_templates.common_utils import (
encrypted_reasoning_signature,
)
from litellm.llms.anthropic.experimental_pass_through.responses_adapters.streaming_iterator import (
AnthropicResponsesStreamWrapper,
)
@ -114,7 +117,7 @@ class TestReasoningItemWithoutSummaryText:
"""
@staticmethod
def _gpt_turn(reasoning_summary_deltas: list) -> list:
def _gpt_turn(reasoning_summary_deltas: list, encrypted_content: str | None = None) -> list:
return [
{"type": "response.created"},
{"type": "response.output_item.added", "item": {"type": "reasoning", "id": "rs_1"}},
@ -122,7 +125,10 @@ class TestReasoningItemWithoutSummaryText:
{"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": delta}
for delta in reasoning_summary_deltas
),
{"type": "response.output_item.done", "item": {"type": "reasoning", "id": "rs_1"}},
{
"type": "response.output_item.done",
"item": {"type": "reasoning", "id": "rs_1", "encrypted_content": encrypted_content},
},
{"type": "response.output_item.added", "item": {"type": "message", "id": "msg_1"}},
{"type": "response.output_text.delta", "item_id": "msg_1", "delta": "Hello"},
{"type": "response.output_item.done", "item": {"type": "message", "id": "msg_1"}},
@ -171,6 +177,70 @@ class TestReasoningItemWithoutSummaryText:
assert not [c for c in chunks if c.get("delta", {}).get("type") == "signature_delta"]
_ENCRYPTED_REASONING = "gAAAAABp_encrypted_reasoning_bytes_only_openai_can_read"
class TestEncryptedReasoningIsStreamedForReplay:
"""Regression for https://github.com/BerriAI/litellm/issues/40288.
The client echoes a thinking block's signature (or a redacted block's data) back on the
next turn, so the item's ``encrypted_content`` has to reach it through one of those.
"""
def test_encrypted_content_is_streamed_as_the_signature_before_the_block_closes(self):
chunks = _drain_async(
TestReasoningItemWithoutSummaryText._gpt_turn(
reasoning_summary_deltas=["Weighing options"], encrypted_content=_ENCRYPTED_REASONING
)
)
assert [(c["type"], c.get("index"), c.get("delta", {}).get("type")) for c in chunks[1:5]] == [
("content_block_start", 0, None),
("content_block_delta", 0, "thinking_delta"),
("content_block_delta", 0, "signature_delta"),
("content_block_stop", 0, None),
]
assert chunks[3]["delta"]["signature"] == encrypted_reasoning_signature(_ENCRYPTED_REASONING)
def test_reasoning_without_summary_streams_a_redacted_thinking_block(self):
chunks = _drain_async(
TestReasoningItemWithoutSummaryText._gpt_turn(
reasoning_summary_deltas=[], encrypted_content=_ENCRYPTED_REASONING
)
)
assert [(c["type"], c.get("index")) for c in chunks[1:]] == [
("content_block_start", 0),
("content_block_stop", 0),
("content_block_start", 1),
("content_block_delta", 1),
("content_block_stop", 1),
]
assert chunks[1]["content_block"] == {
"type": "redacted_thinking",
"data": encrypted_reasoning_signature(_ENCRYPTED_REASONING),
}
def test_summary_parts_are_separated_inside_the_one_thinking_block(self):
"""Two summary parts read as two paragraphs, not as one run-on sentence."""
events = [
{"type": "response.created"},
{"type": "response.output_item.added", "item": {"type": "reasoning", "id": "rs_1"}},
{"type": "response.reasoning_summary_part.added", "item_id": "rs_1", "summary_index": 0},
{"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "First."},
{"type": "response.reasoning_summary_part.added", "item_id": "rs_1", "summary_index": 1},
{"type": "response.reasoning_summary_text.delta", "item_id": "rs_1", "delta": "Second."},
{"type": "response.output_item.done", "item": {"type": "reasoning", "id": "rs_1"}},
]
chunks = _process_all(events)
thinking = "".join(
c["delta"]["thinking"] for c in chunks if c.get("delta", {}).get("type") == "thinking_delta"
)
assert thinking == "First.\n\nSecond."
assert [c["type"] for c in chunks].count("content_block_start") == 1
class TestToolUseBlockClosedExactlyOnce:
"""Regression for https://github.com/BerriAI/litellm/issues/37273.

View file

@ -19,6 +19,7 @@ from litellm.constants import (
from litellm.litellm_core_utils.prompt_templates.common_utils import (
TOOL_RESULT_IMAGE_BOUNDARY,
TOOL_RESULT_IMAGE_PLACEHOLDER,
encrypted_reasoning_signature,
)
from litellm.llms.anthropic.experimental_pass_through.responses_adapters.transformation import (
LiteLLMAnthropicToResponsesAPIAdapter,
@ -566,6 +567,66 @@ class TestTranslateMessagesToResponsesInput:
result = _translate_messages(messages)
assert "id" not in result[0]
def test_thinking_block_with_encrypted_signature_replays_the_encrypted_content(self):
"""Regression for https://github.com/BerriAI/litellm/issues/40288 (inbound fault site)."""
messages = [
{
"role": "assistant",
"content": [
{
"type": "thinking",
"thinking": "Private reasoning.",
"signature": encrypted_reasoning_signature("gAAAA_turn_one"),
}
],
}
]
result = _translate_messages(messages)
assert result == [
{
"type": "reasoning",
"summary": [{"type": "summary_text", "text": "Private reasoning."}],
"encrypted_content": "gAAAA_turn_one",
}
]
def test_redacted_thinking_with_encrypted_data_replays_the_encrypted_content(self):
messages = [
{
"role": "assistant",
"content": [{"type": "redacted_thinking", "data": encrypted_reasoning_signature("gAAAA_turn_one")}],
}
]
result = _translate_messages(messages)
assert result == [{"type": "reasoning", "summary": [], "encrypted_content": "gAAAA_turn_one"}]
def test_each_encrypted_thinking_block_stays_its_own_reasoning_item(self):
"""Two upstream items must not be merged into one, or the encrypted content of one is lost."""
messages = [
{
"role": "assistant",
"content": [
{"type": "thinking", "thinking": "First.", "signature": encrypted_reasoning_signature("gAAAA_1")},
{"type": "thinking", "thinking": "Second.", "signature": encrypted_reasoning_signature("gAAAA_2")},
],
}
]
result = _translate_messages(messages)
assert [item["encrypted_content"] for item in result] == ["gAAAA_1", "gAAAA_2"]
def test_anthropic_signed_thinking_block_replays_as_a_summary_only_item(self):
"""A real Anthropic signature is opaque here, so it never masquerades as encrypted content."""
messages = [
{
"role": "assistant",
"content": [{"type": "thinking", "thinking": "Private reasoning.", "signature": "ErcBCkgIValid"}],
}
]
result = _translate_messages(messages)
assert result == [
{"type": "reasoning", "summary": [{"type": "summary_text", "text": "Private reasoning."}]}
]
def test_consecutive_thinking_blocks_become_one_reasoning_item(self):
"""Summary parts of one upstream reasoning item are regrouped into that item."""
messages = [
@ -1101,6 +1162,13 @@ class TestTranslateRequestBroaderCoverage:
req = _make_request(thinking={"type": "disabled"})
kwargs = _ADAPTER.translate_request(req)
assert "reasoning" not in kwargs
assert "include" not in kwargs
def test_thinking_asks_for_the_encrypted_reasoning(self):
"""The documented way to get reasoning that survives store=false is to ask for it."""
req = _make_request(thinking={"type": "enabled", "budget_tokens": 12000})
kwargs = _ADAPTER.translate_request(req)
assert kwargs["include"] == ["reasoning.encrypted_content"]
def test_metadata_user_id_mapped_to_user(self):
req = _make_request(metadata={"user_id": "user-42"})
@ -1229,7 +1297,9 @@ def _make_function_call_item(call_id: str, name: str, arguments: str) -> MagicMo
return item
def _make_reasoning_item(summaries: List[str], item_id: str = "rs_test_1") -> MagicMock:
def _make_reasoning_item(
summaries: List[str], item_id: str = "rs_test_1", encrypted_content: str | None = None
) -> MagicMock:
"""Build a mock ResponseReasoningItem."""
from openai.types.responses import ResponseReasoningItem # type: ignore[import]
@ -1242,9 +1312,13 @@ def _make_reasoning_item(summaries: List[str], item_id: str = "rs_test_1") -> Ma
item = MagicMock(spec=ResponseReasoningItem)
item.id = item_id
item.summary = summary_mocks
item.encrypted_content = encrypted_content
return item
_ENCRYPTED_REASONING = "gAAAAABp_encrypted_reasoning_bytes_only_openai_can_read"
class TestTranslateResponse:
"""Responses API -> AnthropicMessagesResponse conversion."""
@ -1369,7 +1443,81 @@ class TestTranslateResponse:
reasoning = _make_reasoning_item(["Part one.", "Part two."], item_id="rs_abc123")
response = _make_mock_response(output=[reasoning])
result: Any = _ADAPTER.translate_response(response)
assert [block["signature"] for block in result["content"]] == [None, None]
assert [block["signature"] for block in result["content"]] == [None]
assert "rs_abc123" not in json.dumps(result["content"])
def test_summary_parts_join_into_one_thinking_block(self):
"""One reasoning item is one block, so its signature is echoed back exactly once."""
reasoning = _make_reasoning_item(["Part one.", "Part two."])
response = _make_mock_response(output=[reasoning])
result: Any = _ADAPTER.translate_response(response)
assert [block["thinking"] for block in result["content"]] == ["Part one.\n\nPart two."]
def test_encrypted_content_rides_the_thinking_signature(self):
"""Regression for https://github.com/BerriAI/litellm/issues/40288 (outbound fault site)."""
reasoning = _make_reasoning_item(["Part one."], item_id="rs_abc123", encrypted_content=_ENCRYPTED_REASONING)
response = _make_mock_response(output=[reasoning])
result: Any = _ADAPTER.translate_response(response)
assert result["content"] == [
{
"type": "thinking",
"thinking": "Part one.",
"signature": encrypted_reasoning_signature(_ENCRYPTED_REASONING),
}
]
def test_reasoning_without_summary_becomes_redacted_thinking(self):
"""With summaries off the encrypted reasoning still has to reach the client to be replayed."""
reasoning = _make_reasoning_item([], encrypted_content=_ENCRYPTED_REASONING)
response = _make_mock_response(output=[reasoning])
result: Any = _ADAPTER.translate_response(response)
assert result["content"] == [
{"type": "redacted_thinking", "data": encrypted_reasoning_signature(_ENCRYPTED_REASONING)}
]
def test_dict_reasoning_item_carries_its_encrypted_content(self):
response = _make_mock_response(
output=[
{
"type": "reasoning",
"id": "rs_dict_1",
"encrypted_content": _ENCRYPTED_REASONING,
"summary": [{"type": "summary_text", "text": "Weighing the options."}],
}
]
)
result: Any = _ADAPTER.translate_response(response)
assert result["content"][0]["signature"] == encrypted_reasoning_signature(_ENCRYPTED_REASONING)
def test_reasoning_item_round_trip_is_byte_stable(self):
"""Regression for https://github.com/BerriAI/litellm/issues/40288.
The reasoning item the next turn replays must be the one OpenAI produced, with its
encrypted reasoning intact, and identical on every later turn so the prompt cache
prefix keeps matching.
"""
reasoning = _make_reasoning_item(["Part one.", "Part two."], encrypted_content=_ENCRYPTED_REASONING)
turn: Any = _ADAPTER.translate_response(_make_mock_response(output=[reasoning]))
history = [{"role": "assistant", "content": turn["content"]}]
replayed_items = [_translate_messages(history) for _ in range(2)]
assert replayed_items[0] == replayed_items[1]
assert replayed_items[0] == [
{
"type": "reasoning",
"summary": [{"type": "summary_text", "text": "Part one.\n\nPart two."}],
"encrypted_content": _ENCRYPTED_REASONING,
}
]
def test_redacted_reasoning_round_trip_replays_the_encrypted_content(self):
reasoning = _make_reasoning_item([], encrypted_content=_ENCRYPTED_REASONING)
turn: Any = _ADAPTER.translate_response(_make_mock_response(output=[reasoning]))
replayed = _translate_messages([{"role": "assistant", "content": turn["content"]}])
assert replayed == [{"type": "reasoning", "summary": [], "encrypted_content": _ENCRYPTED_REASONING}]
def test_dict_reasoning_item_becomes_thinking_block(self):
"""A reasoning item arriving as a plain dict is kept, not dropped."""
@ -1385,14 +1533,26 @@ class TestTranslateResponse:
result: Any = _ADAPTER.translate_response(response)
assert result["content"] == [{"type": "thinking", "thinking": "Weighing the options.", "signature": None}]
def test_thinking_blocks_are_dropped_when_replayed_to_anthropic(self):
@pytest.mark.parametrize(
("summaries", "encrypted_content"),
[
(["Part one."], None),
(["Part one."], _ENCRYPTED_REASONING),
([], _ENCRYPTED_REASONING),
],
ids=["unsigned_thinking", "encrypted_thinking", "encrypted_redacted_thinking"],
)
def test_thinking_blocks_are_dropped_when_replayed_to_anthropic(self, summaries, encrypted_content):
"""Replaying this turn to an Anthropic model must not send a signature it cannot verify."""
from litellm.litellm_core_utils.prompt_templates.factory import (
_drop_unsignable_thinking_blocks,
)
response = _make_mock_response(output=[_make_reasoning_item(["Part one."], item_id="rs_abc123")])
response = _make_mock_response(
output=[_make_reasoning_item(summaries, item_id="rs_abc123", encrypted_content=encrypted_content)]
)
result: Any = _ADAPTER.translate_response(response)
assert len(result["content"]) == 1
assert _drop_unsignable_thinking_blocks(result["content"]) == []
def test_usage_mapped_correctly(self):

View file

@ -1541,6 +1541,30 @@ class TestAnthropicThinkingSignatureSelfHeal:
out = strip_empty_content_blocks_from_anthropic_messages(msgs)
assert [b["type"] for b in out[0]["content"]] == ["thinking"]
def test_strip_drops_encrypted_reasoning_blocks_from_the_responses_bridge(self):
"""A session resumed on an Anthropic model replays reasoning only OpenAI can verify."""
from litellm.litellm_core_utils.prompt_templates.common_utils import (
encrypted_reasoning_signature,
)
from litellm.llms.anthropic.common_utils import (
strip_empty_content_blocks_from_anthropic_messages,
)
msgs = [
{
"role": "assistant",
"content": [
{"type": "thinking", "thinking": "plan", "signature": encrypted_reasoning_signature("gAAAA_1")},
{"type": "redacted_thinking", "data": encrypted_reasoning_signature("gAAAA_2")},
{"type": "redacted_thinking", "data": "EmwKAhgBEgy_anthropic_minted"},
{"type": "text", "text": "The answer."},
],
}
]
out = strip_empty_content_blocks_from_anthropic_messages(msgs)
assert [b["type"] for b in out[0]["content"]] == ["redacted_thinking", "text"]
assert len(msgs[0]["content"]) == 4
def test_strip_empty_text_blocks_treats_null_text_as_empty(self):
from litellm.llms.anthropic.common_utils import (
strip_empty_content_blocks_from_anthropic_messages,