From 32bf1aba2958a8601d78ba7f86eb97d58a1adf85 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 22 Aug 2026 14:27:20 -0700 Subject: [PATCH] fix(anthropic): stop signing replayed thinking blocks and strip reasoning_content A reasoning item id is not an Anthropic signature. Passing it off as one got the block replayed to Anthropic and Bedrock as if it were real, and every backend that verifies signatures rejected the turn. Thinking blocks now come back unsigned, and the streaming path no longer emits a signature_delta for them. Azure AI Foundry, Fireworks, and vLLM reject unknown message fields, so they now strip reasoning_content alongside thinking_blocks the way Mistral already did. The thinking-block helpers take ChatCompletionThinkingBlock and ChatCompletionRedactedThinkingBlock instead of loose mappings. --- .../transformation.py | 8 +-- .../prompt_templates/common_utils.py | 34 ++++++++----- .../responses_adapters/streaming_iterator.py | 11 ---- .../responses_adapters/transformation.py | 25 ++++++---- litellm/llms/azure_ai/chat/transformation.py | 10 +++- .../llms/fireworks_ai/chat/transformation.py | 1 + .../llms/hosted_vllm/chat/transformation.py | 5 +- ...t_responses_adapters_streaming_iterator.py | 10 +--- .../test_responses_adapters_transformation.py | 50 +++++++++++++------ .../chat/test_azure_ai_transformation.py | 2 + .../test_fireworks_ai_chat_transformation.py | 2 + .../test_hosted_vllm_chat_transformation.py | 2 + 12 files changed, 96 insertions(+), 64 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index ea54bd83e12..3e55c3c637e 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -97,10 +97,10 @@ def _reasoning_input_items(msg: "AllMessageValues") -> list[dict[str, object]]: stored: Final = [_reasoning_item_to_response_input(r_item) for r_item in _get_reasoning_items(msg)] if stored: return stored - from_thinking: Final = responses_reasoning_item_from_thinking_blocks( - cast(Iterable[Mapping[str, Any]], msg.get("thinking_blocks") or ()) # cast-ok: untyped client thinking blocks - ) - return [from_thinking] if from_thinking is not None else [] + raw_blocks: Final = msg.get("thinking_blocks") or () + blocks: Final = cast("Iterable[ChatCompletionThinkingBlock]", raw_blocks) # cast-ok: untyped client json + from_thinking: Final = responses_reasoning_item_from_thinking_blocks(blocks) + return [] if from_thinking is None else [dict(from_thinking)] def _build_reasoning_item( diff --git a/litellm/litellm_core_utils/prompt_templates/common_utils.py b/litellm/litellm_core_utils/prompt_templates/common_utils.py index 4211fbe610a..e0ba2e13c6e 100644 --- a/litellm/litellm_core_utils/prompt_templates/common_utils.py +++ b/litellm/litellm_core_utils/prompt_templates/common_utils.py @@ -28,8 +28,12 @@ from litellm.types.llms.openai import ( ChatCompletionAssistantMessage, ChatCompletionFileObject, ChatCompletionImageObject, + ChatCompletionReasoningItem, + ChatCompletionReasoningSummaryTextBlock, + ChatCompletionRedactedThinkingBlock, ChatCompletionResponseMessage, ChatCompletionTextObject, + ChatCompletionThinkingBlock, ChatCompletionToolParam, ChatCompletionUserMessage, ) @@ -1549,36 +1553,42 @@ def _extract_reasoning_content(message: dict) -> tuple[str | None, str | None]: return None, message_content +def _readable_thinking_text( + block: ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock, +) -> str: + """The text a chat model can read back, empty for redacted blocks and malformed ones.""" + if block.get("type") != "thinking": + return "" + thinking: Final = cast(ChatCompletionThinkingBlock, block).get("thinking") # cast-ok: narrowed by the type tag + return str(thinking or "") + + def reasoning_content_from_thinking_blocks( - thinking_blocks: Iterable[Mapping[str, Any]], + thinking_blocks: Iterable[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock], ) -> str: """Flatten Anthropic thinking blocks into the `reasoning_content` string chat models expect. Redacted blocks carry no readable text, so they contribute nothing. """ - return "\n".join( - text - for block in thinking_blocks - if block.get("type") == "thinking" and (text := str(block.get("thinking") or "")) - ) + return "\n".join(text for block in thinking_blocks if (text := _readable_thinking_text(block))) def responses_reasoning_item_from_thinking_blocks( - thinking_blocks: Iterable[Mapping[str, Any]], -) -> dict[str, Any] | None: # mutable-ok: API message payload + thinking_blocks: Iterable[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock], +) -> ChatCompletionReasoningItem | None: """Build a Responses API `reasoning` input item from Anthropic thinking blocks. The item carries no `id`: the Responses API rejects an empty one and 404s on any id it did not mint itself, while an item without an id is always accepted. """ - summary: Final[list[dict[str, Any]]] = [ # mutable-ok: API message payload - {"type": "summary_text", "text": text} # mutable-ok: API message payload + summary: Final[list[ChatCompletionReasoningSummaryTextBlock]] = [ # mutable-ok: API message payload + ChatCompletionReasoningSummaryTextBlock(type="summary_text", text=text) for block in thinking_blocks - if block.get("type") == "thinking" and (text := str(block.get("thinking") or "")) + if (text := _readable_thinking_text(block)) ] if not summary: return None - return {"type": "reasoning", "summary": summary} # mutable-ok: API message payload + return ChatCompletionReasoningItem(type="reasoning", summary=summary) def _parse_content_for_reasoning( diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index 286edb24b9b..5577d4a9c2d 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -189,17 +189,6 @@ class AnthropicResponsesStreamWrapper: block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index if block_idx < 0: return - done_item_type: Final = ( - getattr(item, "type", None) or (item.get("type") if isinstance(item, dict) else None) if item else None - ) - if done_item_type == "reasoning" and item_id: - self._chunk_queue.append( - { - "type": "content_block_delta", - "index": block_idx, - "delta": {"type": "signature_delta", "signature": item_id}, - } - ) self._chunk_queue.append( { "type": "content_block_stop", diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py index b1061f86d0b..1a6b0498a52 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py @@ -38,7 +38,11 @@ from litellm.types.llms.anthropic_messages.anthropic_response import ( AnthropicMessagesResponse, AnthropicUsage, ) -from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse +from litellm.types.llms.openai import ( + ChatCompletionThinkingBlock, + ResponseAPIUsage, + ResponsesAPIResponse, +) class LiteLLMAnthropicToResponsesAPIAdapter: @@ -112,19 +116,18 @@ class LiteLLMAnthropicToResponsesAPIAdapter: @classmethod def _thinking_blocks_from_reasoning_item( cls, - item_id: str | None, summary: Iterable[object], ) -> tuple[dict[str, Any], ...]: # mutable-ok: API message payload """Anthropic thinking blocks for one Responses reasoning item. - The reasoning item id rides along as the signature so that a follow-up turn can - regroup the summary parts into the single reasoning item they came from. + The signature stays empty: only Anthropic can sign a thinking block, and a stand-in + value would be replayed as a real one and rejected by every backend that verifies it. """ return tuple( AnthropicResponseContentBlockThinking( type="thinking", thinking=text, - signature=item_id or None, + signature=None, ).model_dump() for part in summary if (text := cls._summary_part_text(part)) @@ -132,10 +135,9 @@ class LiteLLMAnthropicToResponsesAPIAdapter: @staticmethod def _assistant_block_group_key(indexed_block: tuple[int, Mapping[str, Any]]) -> str: - """Group consecutive thinking blocks sharing a signature; keep every other block alone.""" + """Group a run of consecutive thinking blocks together; keep every other block alone.""" index, block = indexed_block - signature: Final = block.get("signature") or "" - return f"thinking:{signature}" if block.get("type") == "thinking" else f"block:{index}" + return "thinking" if block.get("type") == "thinking" else f"block:{index}" @classmethod def _assistant_group_to_input_item( @@ -144,7 +146,9 @@ class LiteLLMAnthropicToResponsesAPIAdapter: first: Final = group[0] btype: Final = first.get("type") if btype == "thinking": - return responses_reasoning_item_from_thinking_blocks(group) + blocks: Final = cast(tuple[ChatCompletionThinkingBlock, ...], group) # cast-ok: untrusted client payload + reasoning_item: Final = responses_reasoning_item_from_thinking_blocks(blocks) + return None if reasoning_item is None else dict(reasoning_item) if btype == "tool_use": return { # mutable-ok: API message payload "type": "function_call", @@ -559,7 +563,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter: for item in response.output: if isinstance(item, ResponseReasoningItem): - content.extend(self._thinking_blocks_from_reasoning_item(item.id, item.summary)) + content.extend(self._thinking_blocks_from_reasoning_item(item.summary)) elif isinstance(item, ResponseOutputMessage): for part in item.content: @@ -594,7 +598,6 @@ class LiteLLMAnthropicToResponsesAPIAdapter: elif item_type == "reasoning": content.extend( self._thinking_blocks_from_reasoning_item( - item.get("id"), cast(Iterable[object], item.get("summary") or ()), # cast-ok: untyped provider json ) ) diff --git a/litellm/llms/azure_ai/chat/transformation.py b/litellm/llms/azure_ai/chat/transformation.py index bc8ea31ea8c..9e7161120cc 100644 --- a/litellm/llms/azure_ai/chat/transformation.py +++ b/litellm/llms/azure_ai/chat/transformation.py @@ -30,7 +30,12 @@ class AzureFoundryErrorStrings(str, enum.Enum): SET_EXTRA_PARAMETERS_TO_PASS_THROUGH = "Set extra-parameters to 'pass-through'" -NON_OPENAI_SPEC_MESSAGE_FIELDS: Final = ("thinking_blocks", "provider_specific_fields", "cache_control") +NON_OPENAI_SPEC_MESSAGE_FIELDS: Final = ( + "thinking_blocks", + "reasoning_content", + "provider_specific_fields", + "cache_control", +) class AzureAIStudioConfig(OpenAIConfig): @@ -173,7 +178,8 @@ class AzureAIStudioConfig(OpenAIConfig): """ - Azure AI Studio doesn't support content as a list. This handles: 1. Strips message fields that are not part of the OpenAI chat-completions - schema (thinking_blocks, provider_specific_fields, cache_control). + schema (thinking_blocks, reasoning_content, provider_specific_fields, + cache_control). Azure AI Foundry backends set additionalProperties=false and reject these with "Extra inputs are not permitted", which breaks multi-turn Anthropic-format clients that echo thinking blocks back as history. diff --git a/litellm/llms/fireworks_ai/chat/transformation.py b/litellm/llms/fireworks_ai/chat/transformation.py index e64237da978..4e9731ef485 100644 --- a/litellm/llms/fireworks_ai/chat/transformation.py +++ b/litellm/llms/fireworks_ai/chat/transformation.py @@ -504,6 +504,7 @@ class FireworksAIConfig(FireworksAIMixin, OpenAIGPTConfig): m = cast(dict, message) m.pop("provider_specific_fields", None) m.pop("thinking_blocks", None) + m.pop("reasoning_content", None) return messages diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index 46a2320b655..29dc485732f 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -164,12 +164,13 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): """ Support translating: - video files from file_id or file_data to video_url - - thinking_blocks on assistant messages are removed, and content lists - are converted to strings for vLLM compatibility + - thinking_blocks and reasoning_content on assistant messages are removed, + and content lists are converted to strings for vLLM compatibility """ for message in messages: if message["role"] == "assistant": message.pop("thinking_blocks", None) + message.pop("reasoning_content", None) existing_content = message.get("content") if isinstance(existing_content, list): text_parts = [] diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py index 12a6ab4324f..aebbed88c70 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py @@ -139,7 +139,6 @@ class TestReasoningItemWithoutSummaryText: ("content_block_start", 0), ("content_block_delta", 0), ("content_block_delta", 0), - ("content_block_delta", 0), ("content_block_stop", 0), ("content_block_start", 1), ("content_block_delta", 1), @@ -148,15 +147,10 @@ class TestReasoningItemWithoutSummaryText: assert chunks[1]["content_block"] == {"type": "thinking", "thinking": "", "signature": ""} assert "".join(c["delta"]["thinking"] for c in chunks[2:4]) == "Weighing options" - def test_reasoning_item_id_is_streamed_as_the_thinking_signature(self): + def test_the_reasoning_item_id_is_never_streamed_as_a_signature(self): + """A stand-in signature would be replayed as a real one, so none is ever sent.""" chunks = _drain_async(self._gpt_turn(reasoning_summary_deltas=["Weighing options"])) - signature_deltas = [c for c in chunks if c.get("delta", {}).get("type") == "signature_delta"] - assert [(c["index"], c["delta"]["signature"]) for c in signature_deltas] == [(0, "rs_1")] - - def test_no_signature_delta_without_a_thinking_block(self): - chunks = _drain_async(self._gpt_turn(reasoning_summary_deltas=[])) - assert not [c for c in chunks if c.get("delta", {}).get("type") == "signature_delta"] diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_transformation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_transformation.py index 6479c43ee7a..790bcd269e0 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_transformation.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_transformation.py @@ -513,15 +513,14 @@ class TestTranslateMessagesToResponsesInput: result = _translate_messages(messages) assert "id" not in result[0] - def test_thinking_blocks_sharing_a_signature_become_one_reasoning_item(self): - """Summary parts of one upstream reasoning item are regrouped by their signature.""" + def test_consecutive_thinking_blocks_become_one_reasoning_item(self): + """Summary parts of one upstream reasoning item are regrouped into that item.""" messages = [ { "role": "assistant", "content": [ - {"type": "thinking", "thinking": "First part.", "signature": "rs_abc123"}, - {"type": "thinking", "thinking": "Second part.", "signature": "rs_abc123"}, - {"type": "thinking", "thinking": "A later item.", "signature": "rs_def456"}, + {"type": "thinking", "thinking": "First part."}, + {"type": "thinking", "thinking": "Second part."}, ], } ] @@ -533,13 +532,26 @@ class TestTranslateMessagesToResponsesInput: {"type": "summary_text", "text": "First part."}, {"type": "summary_text", "text": "Second part."}, ], - }, - { - "type": "reasoning", - "summary": [{"type": "summary_text", "text": "A later item."}], - }, + } ] + def test_a_tool_call_splits_the_reasoning_items_around_it(self): + """Thinking on either side of a tool call belongs to two different reasoning items.""" + messages = [ + { + "role": "assistant", + "content": [ + {"type": "thinking", "thinking": "Before the call."}, + {"type": "tool_use", "id": "call_1", "name": "get_weather", "input": {"city": "Denver"}}, + {"type": "thinking", "thinking": "After the call."}, + ], + } + ] + result = _translate_messages(messages) + assert [item["type"] for item in result] == ["reasoning", "function_call", "reasoning"] + assert result[0]["summary"] == [{"type": "summary_text", "text": "Before the call."}] + assert result[2]["summary"] == [{"type": "summary_text", "text": "After the call."}] + def test_thinking_and_text_stay_separate(self): """The visible answer stays the only thing in the assistant message.""" messages = [ @@ -1237,12 +1249,12 @@ class TestTranslateResponse: result: Any = _ADAPTER.translate_response(response) assert result["content"] == [] - def test_reasoning_item_id_becomes_thinking_signature(self): - """The reasoning item id rides back as the signature so the next turn can regroup it.""" + def test_reasoning_item_id_never_becomes_a_thinking_signature(self): + """Only Anthropic can sign a thinking block, so a stand-in signature is never invented.""" reasoning = _make_reasoning_item(["Part one.", "Part two."], item_id="rs_abc123") response = _make_mock_response(output=[reasoning]) result: Any = _ADAPTER.translate_response(response) - assert [block["signature"] for block in result["content"]] == ["rs_abc123", "rs_abc123"] + assert [block["signature"] for block in result["content"]] == [None, None] def test_dict_reasoning_item_becomes_thinking_block(self): """A reasoning item arriving as a plain dict is kept, not dropped.""" @@ -1257,9 +1269,19 @@ class TestTranslateResponse: ) result: Any = _ADAPTER.translate_response(response) assert result["content"] == [ - {"type": "thinking", "thinking": "Weighing the options.", "signature": "rs_dict_1"} + {"type": "thinking", "thinking": "Weighing the options.", "signature": None} ] + def test_thinking_blocks_are_dropped_when_replayed_to_anthropic(self): + """Replaying this turn to an Anthropic model must not send a signature it cannot verify.""" + from litellm.litellm_core_utils.prompt_templates.factory import ( + _drop_unsignable_thinking_blocks, + ) + + response = _make_mock_response(output=[_make_reasoning_item(["Part one."], item_id="rs_abc123")]) + result: Any = _ADAPTER.translate_response(response) + assert _drop_unsignable_thinking_blocks(result["content"]) == [] + def test_usage_mapped_correctly(self): """Input/output tokens from ResponseAPIUsage are mapped to AnthropicUsage.""" response = _make_mock_response( diff --git a/tests/test_litellm/llms/azure_ai/chat/test_azure_ai_transformation.py b/tests/test_litellm/llms/azure_ai/chat/test_azure_ai_transformation.py index 0fd9a381a5a..d4fcbc823a6 100644 --- a/tests/test_litellm/llms/azure_ai/chat/test_azure_ai_transformation.py +++ b/tests/test_litellm/llms/azure_ai/chat/test_azure_ai_transformation.py @@ -300,6 +300,7 @@ def test_azure_ai_strips_non_openai_spec_message_fields(): "cache_control": {"type": "ephemeral"}, } ], + "reasoning_content": "The user wants me to read a file.", "provider_specific_fields": {"thought_signature": "sig-top"}, "tool_calls": [ { @@ -327,6 +328,7 @@ def test_azure_ai_strips_non_openai_spec_message_fields(): transformed_messages = request["messages"] assert not _find_key_anywhere(transformed_messages, "thinking_blocks") + assert not _find_key_anywhere(transformed_messages, "reasoning_content") assert not _find_key_anywhere(transformed_messages, "provider_specific_fields") assert not _find_key_anywhere(transformed_messages, "cache_control") diff --git a/tests/test_litellm/llms/fireworks_ai/chat/test_fireworks_ai_chat_transformation.py b/tests/test_litellm/llms/fireworks_ai/chat/test_fireworks_ai_chat_transformation.py index e728fc4bc40..63a749dab84 100644 --- a/tests/test_litellm/llms/fireworks_ai/chat/test_fireworks_ai_chat_transformation.py +++ b/tests/test_litellm/llms/fireworks_ai/chat/test_fireworks_ai_chat_transformation.py @@ -473,12 +473,14 @@ def test_transform_messages_helper_strips_thinking_blocks(): "thinking_blocks": [ {"type": "thinking", "thinking": "internal", "signature": ""} ], + "reasoning_content": "internal", }, ] out = config._transform_messages_helper( messages, model="accounts/fireworks/models/glm-5p1", litellm_params={} ) assert "thinking_blocks" not in out[1] + assert "reasoning_content" not in out[1] assert out[1]["content"] == "I can help." diff --git a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py index e316cd14dd4..82b05601a85 100644 --- a/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py +++ b/tests/test_litellm/llms/hosted_vllm/chat/test_hosted_vllm_chat_transformation.py @@ -200,6 +200,7 @@ def test_hosted_vllm_thinking_blocks_prepended_to_assistant_content(): "signature": "abc123", } ], + "reasoning_content": "Let me reason about this...", }, { "role": "user", @@ -218,6 +219,7 @@ def test_hosted_vllm_thinking_blocks_prepended_to_assistant_content(): assert isinstance(assistant_msg["content"], str) assert assistant_msg["content"] == "Here is my answer." assert "thinking_blocks" not in assistant_msg + assert "reasoning_content" not in assistant_msg def test_hosted_vllm_thinking_blocks_with_list_content():