mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(anthropic): stop signing replayed thinking blocks and strip reasoning_content
A reasoning item id is not an Anthropic signature. Passing it off as one got the block replayed to Anthropic and Bedrock as if it were real, and every backend that verifies signatures rejected the turn. Thinking blocks now come back unsigned, and the streaming path no longer emits a signature_delta for them. Azure AI Foundry, Fireworks, and vLLM reject unknown message fields, so they now strip reasoning_content alongside thinking_blocks the way Mistral already did. The thinking-block helpers take ChatCompletionThinkingBlock and ChatCompletionRedactedThinkingBlock instead of loose mappings.
This commit is contained in:
parent
766f72f1d7
commit
32bf1aba29
12 changed files with 96 additions and 64 deletions
|
|
@ -97,10 +97,10 @@ def _reasoning_input_items(msg: "AllMessageValues") -> list[dict[str, object]]:
|
|||
stored: Final = [_reasoning_item_to_response_input(r_item) for r_item in _get_reasoning_items(msg)]
|
||||
if stored:
|
||||
return stored
|
||||
from_thinking: Final = responses_reasoning_item_from_thinking_blocks(
|
||||
cast(Iterable[Mapping[str, Any]], msg.get("thinking_blocks") or ()) # cast-ok: untyped client thinking blocks
|
||||
)
|
||||
return [from_thinking] if from_thinking is not None else []
|
||||
raw_blocks: Final = msg.get("thinking_blocks") or ()
|
||||
blocks: Final = cast("Iterable[ChatCompletionThinkingBlock]", raw_blocks) # cast-ok: untyped client json
|
||||
from_thinking: Final = responses_reasoning_item_from_thinking_blocks(blocks)
|
||||
return [] if from_thinking is None else [dict(from_thinking)]
|
||||
|
||||
|
||||
def _build_reasoning_item(
|
||||
|
|
|
|||
|
|
@ -28,8 +28,12 @@ from litellm.types.llms.openai import (
|
|||
ChatCompletionAssistantMessage,
|
||||
ChatCompletionFileObject,
|
||||
ChatCompletionImageObject,
|
||||
ChatCompletionReasoningItem,
|
||||
ChatCompletionReasoningSummaryTextBlock,
|
||||
ChatCompletionRedactedThinkingBlock,
|
||||
ChatCompletionResponseMessage,
|
||||
ChatCompletionTextObject,
|
||||
ChatCompletionThinkingBlock,
|
||||
ChatCompletionToolParam,
|
||||
ChatCompletionUserMessage,
|
||||
)
|
||||
|
|
@ -1549,36 +1553,42 @@ def _extract_reasoning_content(message: dict) -> tuple[str | None, str | None]:
|
|||
return None, message_content
|
||||
|
||||
|
||||
def _readable_thinking_text(
|
||||
block: ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock,
|
||||
) -> str:
|
||||
"""The text a chat model can read back, empty for redacted blocks and malformed ones."""
|
||||
if block.get("type") != "thinking":
|
||||
return ""
|
||||
thinking: Final = cast(ChatCompletionThinkingBlock, block).get("thinking") # cast-ok: narrowed by the type tag
|
||||
return str(thinking or "")
|
||||
|
||||
|
||||
def reasoning_content_from_thinking_blocks(
|
||||
thinking_blocks: Iterable[Mapping[str, Any]],
|
||||
thinking_blocks: Iterable[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock],
|
||||
) -> str:
|
||||
"""Flatten Anthropic thinking blocks into the `reasoning_content` string chat models expect.
|
||||
|
||||
Redacted blocks carry no readable text, so they contribute nothing.
|
||||
"""
|
||||
return "\n".join(
|
||||
text
|
||||
for block in thinking_blocks
|
||||
if block.get("type") == "thinking" and (text := str(block.get("thinking") or ""))
|
||||
)
|
||||
return "\n".join(text for block in thinking_blocks if (text := _readable_thinking_text(block)))
|
||||
|
||||
|
||||
def responses_reasoning_item_from_thinking_blocks(
|
||||
thinking_blocks: Iterable[Mapping[str, Any]],
|
||||
) -> dict[str, Any] | None: # mutable-ok: API message payload
|
||||
thinking_blocks: Iterable[ChatCompletionThinkingBlock | ChatCompletionRedactedThinkingBlock],
|
||||
) -> ChatCompletionReasoningItem | None:
|
||||
"""Build a Responses API `reasoning` input item from Anthropic thinking blocks.
|
||||
|
||||
The item carries no `id`: the Responses API rejects an empty one and 404s on any id it
|
||||
did not mint itself, while an item without an id is always accepted.
|
||||
"""
|
||||
summary: Final[list[dict[str, Any]]] = [ # mutable-ok: API message payload
|
||||
{"type": "summary_text", "text": text} # mutable-ok: API message payload
|
||||
summary: Final[list[ChatCompletionReasoningSummaryTextBlock]] = [ # mutable-ok: API message payload
|
||||
ChatCompletionReasoningSummaryTextBlock(type="summary_text", text=text)
|
||||
for block in thinking_blocks
|
||||
if block.get("type") == "thinking" and (text := str(block.get("thinking") or ""))
|
||||
if (text := _readable_thinking_text(block))
|
||||
]
|
||||
if not summary:
|
||||
return None
|
||||
return {"type": "reasoning", "summary": summary} # mutable-ok: API message payload
|
||||
return ChatCompletionReasoningItem(type="reasoning", summary=summary)
|
||||
|
||||
|
||||
def _parse_content_for_reasoning(
|
||||
|
|
|
|||
|
|
@ -189,17 +189,6 @@ class AnthropicResponsesStreamWrapper:
|
|||
block_idx = self._item_id_to_block_index.get(item_id, -1) if item_id else self._current_block_index
|
||||
if block_idx < 0:
|
||||
return
|
||||
done_item_type: Final = (
|
||||
getattr(item, "type", None) or (item.get("type") if isinstance(item, dict) else None) if item else None
|
||||
)
|
||||
if done_item_type == "reasoning" and item_id:
|
||||
self._chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_delta",
|
||||
"index": block_idx,
|
||||
"delta": {"type": "signature_delta", "signature": item_id},
|
||||
}
|
||||
)
|
||||
self._chunk_queue.append(
|
||||
{
|
||||
"type": "content_block_stop",
|
||||
|
|
|
|||
|
|
@ -38,7 +38,11 @@ from litellm.types.llms.anthropic_messages.anthropic_response import (
|
|||
AnthropicMessagesResponse,
|
||||
AnthropicUsage,
|
||||
)
|
||||
from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse
|
||||
from litellm.types.llms.openai import (
|
||||
ChatCompletionThinkingBlock,
|
||||
ResponseAPIUsage,
|
||||
ResponsesAPIResponse,
|
||||
)
|
||||
|
||||
|
||||
class LiteLLMAnthropicToResponsesAPIAdapter:
|
||||
|
|
@ -112,19 +116,18 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
@classmethod
|
||||
def _thinking_blocks_from_reasoning_item(
|
||||
cls,
|
||||
item_id: str | None,
|
||||
summary: Iterable[object],
|
||||
) -> tuple[dict[str, Any], ...]: # mutable-ok: API message payload
|
||||
"""Anthropic thinking blocks for one Responses reasoning item.
|
||||
|
||||
The reasoning item id rides along as the signature so that a follow-up turn can
|
||||
regroup the summary parts into the single reasoning item they came from.
|
||||
The signature stays empty: only Anthropic can sign a thinking block, and a stand-in
|
||||
value would be replayed as a real one and rejected by every backend that verifies it.
|
||||
"""
|
||||
return tuple(
|
||||
AnthropicResponseContentBlockThinking(
|
||||
type="thinking",
|
||||
thinking=text,
|
||||
signature=item_id or None,
|
||||
signature=None,
|
||||
).model_dump()
|
||||
for part in summary
|
||||
if (text := cls._summary_part_text(part))
|
||||
|
|
@ -132,10 +135,9 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
|
||||
@staticmethod
|
||||
def _assistant_block_group_key(indexed_block: tuple[int, Mapping[str, Any]]) -> str:
|
||||
"""Group consecutive thinking blocks sharing a signature; keep every other block alone."""
|
||||
"""Group a run of consecutive thinking blocks together; keep every other block alone."""
|
||||
index, block = indexed_block
|
||||
signature: Final = block.get("signature") or ""
|
||||
return f"thinking:{signature}" if block.get("type") == "thinking" else f"block:{index}"
|
||||
return "thinking" if block.get("type") == "thinking" else f"block:{index}"
|
||||
|
||||
@classmethod
|
||||
def _assistant_group_to_input_item(
|
||||
|
|
@ -144,7 +146,9 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
first: Final = group[0]
|
||||
btype: Final = first.get("type")
|
||||
if btype == "thinking":
|
||||
return responses_reasoning_item_from_thinking_blocks(group)
|
||||
blocks: Final = cast(tuple[ChatCompletionThinkingBlock, ...], group) # cast-ok: untrusted client payload
|
||||
reasoning_item: Final = responses_reasoning_item_from_thinking_blocks(blocks)
|
||||
return None if reasoning_item is None else dict(reasoning_item)
|
||||
if btype == "tool_use":
|
||||
return { # mutable-ok: API message payload
|
||||
"type": "function_call",
|
||||
|
|
@ -559,7 +563,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
|
||||
for item in response.output:
|
||||
if isinstance(item, ResponseReasoningItem):
|
||||
content.extend(self._thinking_blocks_from_reasoning_item(item.id, item.summary))
|
||||
content.extend(self._thinking_blocks_from_reasoning_item(item.summary))
|
||||
|
||||
elif isinstance(item, ResponseOutputMessage):
|
||||
for part in item.content:
|
||||
|
|
@ -594,7 +598,6 @@ class LiteLLMAnthropicToResponsesAPIAdapter:
|
|||
elif item_type == "reasoning":
|
||||
content.extend(
|
||||
self._thinking_blocks_from_reasoning_item(
|
||||
item.get("id"),
|
||||
cast(Iterable[object], item.get("summary") or ()), # cast-ok: untyped provider json
|
||||
)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -30,7 +30,12 @@ class AzureFoundryErrorStrings(str, enum.Enum):
|
|||
SET_EXTRA_PARAMETERS_TO_PASS_THROUGH = "Set extra-parameters to 'pass-through'"
|
||||
|
||||
|
||||
NON_OPENAI_SPEC_MESSAGE_FIELDS: Final = ("thinking_blocks", "provider_specific_fields", "cache_control")
|
||||
NON_OPENAI_SPEC_MESSAGE_FIELDS: Final = (
|
||||
"thinking_blocks",
|
||||
"reasoning_content",
|
||||
"provider_specific_fields",
|
||||
"cache_control",
|
||||
)
|
||||
|
||||
|
||||
class AzureAIStudioConfig(OpenAIConfig):
|
||||
|
|
@ -173,7 +178,8 @@ class AzureAIStudioConfig(OpenAIConfig):
|
|||
"""
|
||||
- Azure AI Studio doesn't support content as a list. This handles:
|
||||
1. Strips message fields that are not part of the OpenAI chat-completions
|
||||
schema (thinking_blocks, provider_specific_fields, cache_control).
|
||||
schema (thinking_blocks, reasoning_content, provider_specific_fields,
|
||||
cache_control).
|
||||
Azure AI Foundry backends set additionalProperties=false and reject
|
||||
these with "Extra inputs are not permitted", which breaks multi-turn
|
||||
Anthropic-format clients that echo thinking blocks back as history.
|
||||
|
|
|
|||
|
|
@ -504,6 +504,7 @@ class FireworksAIConfig(FireworksAIMixin, OpenAIGPTConfig):
|
|||
m = cast(dict, message)
|
||||
m.pop("provider_specific_fields", None)
|
||||
m.pop("thinking_blocks", None)
|
||||
m.pop("reasoning_content", None)
|
||||
|
||||
return messages
|
||||
|
||||
|
|
|
|||
|
|
@ -164,12 +164,13 @@ class HostedVLLMChatConfig(OpenAIGPTConfig):
|
|||
"""
|
||||
Support translating:
|
||||
- video files from file_id or file_data to video_url
|
||||
- thinking_blocks on assistant messages are removed, and content lists
|
||||
are converted to strings for vLLM compatibility
|
||||
- thinking_blocks and reasoning_content on assistant messages are removed,
|
||||
and content lists are converted to strings for vLLM compatibility
|
||||
"""
|
||||
for message in messages:
|
||||
if message["role"] == "assistant":
|
||||
message.pop("thinking_blocks", None)
|
||||
message.pop("reasoning_content", None)
|
||||
existing_content = message.get("content")
|
||||
if isinstance(existing_content, list):
|
||||
text_parts = []
|
||||
|
|
|
|||
|
|
@ -139,7 +139,6 @@ class TestReasoningItemWithoutSummaryText:
|
|||
("content_block_start", 0),
|
||||
("content_block_delta", 0),
|
||||
("content_block_delta", 0),
|
||||
("content_block_delta", 0),
|
||||
("content_block_stop", 0),
|
||||
("content_block_start", 1),
|
||||
("content_block_delta", 1),
|
||||
|
|
@ -148,15 +147,10 @@ class TestReasoningItemWithoutSummaryText:
|
|||
assert chunks[1]["content_block"] == {"type": "thinking", "thinking": "", "signature": ""}
|
||||
assert "".join(c["delta"]["thinking"] for c in chunks[2:4]) == "Weighing options"
|
||||
|
||||
def test_reasoning_item_id_is_streamed_as_the_thinking_signature(self):
|
||||
def test_the_reasoning_item_id_is_never_streamed_as_a_signature(self):
|
||||
"""A stand-in signature would be replayed as a real one, so none is ever sent."""
|
||||
chunks = _drain_async(self._gpt_turn(reasoning_summary_deltas=["Weighing options"]))
|
||||
|
||||
signature_deltas = [c for c in chunks if c.get("delta", {}).get("type") == "signature_delta"]
|
||||
assert [(c["index"], c["delta"]["signature"]) for c in signature_deltas] == [(0, "rs_1")]
|
||||
|
||||
def test_no_signature_delta_without_a_thinking_block(self):
|
||||
chunks = _drain_async(self._gpt_turn(reasoning_summary_deltas=[]))
|
||||
|
||||
assert not [c for c in chunks if c.get("delta", {}).get("type") == "signature_delta"]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -513,15 +513,14 @@ class TestTranslateMessagesToResponsesInput:
|
|||
result = _translate_messages(messages)
|
||||
assert "id" not in result[0]
|
||||
|
||||
def test_thinking_blocks_sharing_a_signature_become_one_reasoning_item(self):
|
||||
"""Summary parts of one upstream reasoning item are regrouped by their signature."""
|
||||
def test_consecutive_thinking_blocks_become_one_reasoning_item(self):
|
||||
"""Summary parts of one upstream reasoning item are regrouped into that item."""
|
||||
messages = [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "thinking", "thinking": "First part.", "signature": "rs_abc123"},
|
||||
{"type": "thinking", "thinking": "Second part.", "signature": "rs_abc123"},
|
||||
{"type": "thinking", "thinking": "A later item.", "signature": "rs_def456"},
|
||||
{"type": "thinking", "thinking": "First part."},
|
||||
{"type": "thinking", "thinking": "Second part."},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
|
@ -533,13 +532,26 @@ class TestTranslateMessagesToResponsesInput:
|
|||
{"type": "summary_text", "text": "First part."},
|
||||
{"type": "summary_text", "text": "Second part."},
|
||||
],
|
||||
},
|
||||
{
|
||||
"type": "reasoning",
|
||||
"summary": [{"type": "summary_text", "text": "A later item."}],
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
def test_a_tool_call_splits_the_reasoning_items_around_it(self):
|
||||
"""Thinking on either side of a tool call belongs to two different reasoning items."""
|
||||
messages = [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "thinking", "thinking": "Before the call."},
|
||||
{"type": "tool_use", "id": "call_1", "name": "get_weather", "input": {"city": "Denver"}},
|
||||
{"type": "thinking", "thinking": "After the call."},
|
||||
],
|
||||
}
|
||||
]
|
||||
result = _translate_messages(messages)
|
||||
assert [item["type"] for item in result] == ["reasoning", "function_call", "reasoning"]
|
||||
assert result[0]["summary"] == [{"type": "summary_text", "text": "Before the call."}]
|
||||
assert result[2]["summary"] == [{"type": "summary_text", "text": "After the call."}]
|
||||
|
||||
def test_thinking_and_text_stay_separate(self):
|
||||
"""The visible answer stays the only thing in the assistant message."""
|
||||
messages = [
|
||||
|
|
@ -1237,12 +1249,12 @@ class TestTranslateResponse:
|
|||
result: Any = _ADAPTER.translate_response(response)
|
||||
assert result["content"] == []
|
||||
|
||||
def test_reasoning_item_id_becomes_thinking_signature(self):
|
||||
"""The reasoning item id rides back as the signature so the next turn can regroup it."""
|
||||
def test_reasoning_item_id_never_becomes_a_thinking_signature(self):
|
||||
"""Only Anthropic can sign a thinking block, so a stand-in signature is never invented."""
|
||||
reasoning = _make_reasoning_item(["Part one.", "Part two."], item_id="rs_abc123")
|
||||
response = _make_mock_response(output=[reasoning])
|
||||
result: Any = _ADAPTER.translate_response(response)
|
||||
assert [block["signature"] for block in result["content"]] == ["rs_abc123", "rs_abc123"]
|
||||
assert [block["signature"] for block in result["content"]] == [None, None]
|
||||
|
||||
def test_dict_reasoning_item_becomes_thinking_block(self):
|
||||
"""A reasoning item arriving as a plain dict is kept, not dropped."""
|
||||
|
|
@ -1257,9 +1269,19 @@ class TestTranslateResponse:
|
|||
)
|
||||
result: Any = _ADAPTER.translate_response(response)
|
||||
assert result["content"] == [
|
||||
{"type": "thinking", "thinking": "Weighing the options.", "signature": "rs_dict_1"}
|
||||
{"type": "thinking", "thinking": "Weighing the options.", "signature": None}
|
||||
]
|
||||
|
||||
def test_thinking_blocks_are_dropped_when_replayed_to_anthropic(self):
|
||||
"""Replaying this turn to an Anthropic model must not send a signature it cannot verify."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
_drop_unsignable_thinking_blocks,
|
||||
)
|
||||
|
||||
response = _make_mock_response(output=[_make_reasoning_item(["Part one."], item_id="rs_abc123")])
|
||||
result: Any = _ADAPTER.translate_response(response)
|
||||
assert _drop_unsignable_thinking_blocks(result["content"]) == []
|
||||
|
||||
def test_usage_mapped_correctly(self):
|
||||
"""Input/output tokens from ResponseAPIUsage are mapped to AnthropicUsage."""
|
||||
response = _make_mock_response(
|
||||
|
|
|
|||
|
|
@ -300,6 +300,7 @@ def test_azure_ai_strips_non_openai_spec_message_fields():
|
|||
"cache_control": {"type": "ephemeral"},
|
||||
}
|
||||
],
|
||||
"reasoning_content": "The user wants me to read a file.",
|
||||
"provider_specific_fields": {"thought_signature": "sig-top"},
|
||||
"tool_calls": [
|
||||
{
|
||||
|
|
@ -327,6 +328,7 @@ def test_azure_ai_strips_non_openai_spec_message_fields():
|
|||
transformed_messages = request["messages"]
|
||||
|
||||
assert not _find_key_anywhere(transformed_messages, "thinking_blocks")
|
||||
assert not _find_key_anywhere(transformed_messages, "reasoning_content")
|
||||
assert not _find_key_anywhere(transformed_messages, "provider_specific_fields")
|
||||
assert not _find_key_anywhere(transformed_messages, "cache_control")
|
||||
|
||||
|
|
|
|||
|
|
@ -473,12 +473,14 @@ def test_transform_messages_helper_strips_thinking_blocks():
|
|||
"thinking_blocks": [
|
||||
{"type": "thinking", "thinking": "internal", "signature": ""}
|
||||
],
|
||||
"reasoning_content": "internal",
|
||||
},
|
||||
]
|
||||
out = config._transform_messages_helper(
|
||||
messages, model="accounts/fireworks/models/glm-5p1", litellm_params={}
|
||||
)
|
||||
assert "thinking_blocks" not in out[1]
|
||||
assert "reasoning_content" not in out[1]
|
||||
assert out[1]["content"] == "I can help."
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -200,6 +200,7 @@ def test_hosted_vllm_thinking_blocks_prepended_to_assistant_content():
|
|||
"signature": "abc123",
|
||||
}
|
||||
],
|
||||
"reasoning_content": "Let me reason about this...",
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
|
|
@ -218,6 +219,7 @@ def test_hosted_vllm_thinking_blocks_prepended_to_assistant_content():
|
|||
assert isinstance(assistant_msg["content"], str)
|
||||
assert assistant_msg["content"] == "Here is my answer."
|
||||
assert "thinking_blocks" not in assistant_msg
|
||||
assert "reasoning_content" not in assistant_msg
|
||||
|
||||
|
||||
def test_hosted_vllm_thinking_blocks_with_list_content():
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue