From 189c530b90d534813733c30f592dc73be1f75b81 Mon Sep 17 00:00:00 2001 From: PRABHU KIRAN VANDRANKI <72809214+VANDRANKI@users.noreply.github.com> Date: Mon, 8 Jun 2026 13:20:10 -0400 Subject: [PATCH] fix(anthropic): omit thinking blocks from Responses API message history replay Fixes #26916 When `anthropic_messages()` was called with a prior assistant `thinking` block in the conversation history, the Responses-API adapter was serializing the thinking content as an `output_text` block. This leaks internal model reasoning back to the model as if it were ordinary visible output. Thinking blocks represent private internal reasoning and have no equivalent history item in the OpenAI Responses API, so they are now silently dropped during history replay. This matches the behavior of the Anthropic API itself (which strips `thinking` blocks on replay when extended thinking is enabled). --- .../responses_adapters/transformation.py | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py index 2badc2a3276..0e1c358d703 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py @@ -166,11 +166,9 @@ class LiteLLMAnthropicToResponsesAPIAdapter: } ) elif btype == "thinking": - thinking_text = block.get("thinking", "") - if thinking_text: - asst_parts.append( - {"type": "output_text", "text": thinking_text} - ) + # Thinking blocks are internal model reasoning and must not be + # forwarded as output_text in replayed history (fixes #26916). + pass if asst_parts: input_items.append( {