From 6ddc7875eaca14beb6fbc270037759999a9b551c Mon Sep 17 00:00:00 2001
From: Cesar Garcia <128240629+Chesars@users.noreply.github.com>
Date: Mon, 15 Dec 2025 05:05:46 -0300
Subject: [PATCH] Fix: add OpenAI-compatible API for Anthropic with
modify_params=True (#17106)
* docs: add OpenAI-compatible API limitations for Anthropic thinking
Document the fundamental incompatibility between Anthropic extended
thinking and OpenAI-compatible API clients. Explains:
- Why thinking_blocks must be resent (stateless vs stateful APIs)
- OpenAI vs Anthropic architecture differences
- Solutions for client developers
* Update docs
* fix: auto-drop thinking param when thinking_blocks missing
When modify_params=True, LiteLLM now automatically drops the 'thinking'
param if the last assistant message with tool_calls is missing
thinking_blocks. This prevents the Anthropic error:
"Expected thinking or redacted_thinking, but found tool_use"
This workaround addresses the OpenAI-Anthropic API incompatibility where
OpenAI-compatible clients don't preserve thinking_blocks.
---
docs/my-website/docs/reasoning_content.md | 101 ++++++++++++++++++
litellm/llms/anthropic/chat/transformation.py | 15 +++
litellm/utils.py | 28 +++++
3 files changed, 144 insertions(+)
diff --git a/docs/my-website/docs/reasoning_content.md b/docs/my-website/docs/reasoning_content.md
index 12db17325d4..fca3df638c7 100644
--- a/docs/my-website/docs/reasoning_content.md
+++ b/docs/my-website/docs/reasoning_content.md
@@ -114,6 +114,107 @@ curl http://0.0.0.0:4000/v1/chat/completions \
Here's how to use `thinking` blocks by Anthropic with tool calling.
+### Important: OpenAI-Compatible API Limitations
+
+:::warning Compatibility Notice
+
+Anthropic extended thinking with tool calling is **not fully compatible** with OpenAI-compatible API clients. This is due to fundamental architectural differences between how OpenAI and Anthropic handle reasoning in multi-turn conversations.
+
+:::
+
+When using Anthropic models with `thinking` enabled and tool calling, you **must include `thinking_blocks`** from the previous assistant response when sending tool results back. Failure to do so will result in a `400 Bad Request` error.
+
+**OpenAI vs Anthropic Architecture:**
+
+| Provider | API Architecture | Reasoning Storage | Multi-turn Handling |
+|----------|------------------|-------------------|---------------------|
+| **OpenAI** (o1, o3) | Responses API (Stateful) | Server-side | Server stores reasoning internally; client sends `previous_response_id` |
+| **Anthropic** (Claude) | Messages API (Stateless) | Client-side | Client must store and resend `thinking_blocks` with every request |
+
+
+1. OpenAI's Chat Completions spec has **no field** for `thinking_blocks`
+2. OpenAI-compatible clients (LibreChat, Open WebUI, Vercel AI SDK, etc.) **ignore** the `thinking_blocks` field in responses
+3. When these clients reconstruct the assistant message for the next turn, the thinking blocks are lost
+4. Anthropic rejects the request because the assistant message doesn't start with a thinking block
+
+:::tip LiteLLM supports thinking_blocks
+LiteLLM's `completion()` API **does support** sending `thinking_blocks` in assistant messages. If you're using LiteLLM directly (not through an OpenAI-compatible client), you can preserve and resend `thinking_blocks` and everything will work correctly.
+:::
+
+**Solutions:**
+
+1. **Use LiteLLM's built-in workaround** (recommended): Set `litellm.modify_params = True` and LiteLLM will automatically handle this incompatibility by dropping the `thinking` param when `thinking_blocks` are missing (see below)
+2. **For client developers**: Explicitly handle and resend the `thinking_blocks` field (see example below)
+3. **Disable extended thinking** when using tools with OpenAI-compatible clients that don't support `thinking_blocks`
+4. **Use Anthropic's native API** directly instead of OpenAI-compatible endpoints
+
+### LiteLLM Built-in Workaround
+
+LiteLLM can automatically handle this incompatibility when `modify_params=True` is set. If the client sends a request with `thinking` enabled but the assistant message with `tool_calls` is missing `thinking_blocks`, LiteLLM will automatically drop the `thinking` param for that turn to avoid the error.
+
+
+
+
+```python showLineNumbers
+import litellm
+
+# Enable automatic parameter modification
+litellm.modify_params = True
+
+# Now this will work even if thinking_blocks are missing from the assistant message
+response = litellm.completion(
+ model="anthropic/claude-sonnet-4-20250514",
+ thinking={"type": "enabled", "budget_tokens": 1024},
+ tools=[...],
+ messages=[
+ {"role": "user", "content": "What's the weather in Madrid?"},
+ {
+ "role": "assistant",
+ "tool_calls": [{"id": "call_123", "type": "function", "function": {"name": "get_weather", "arguments": '{"city": "Madrid"}'}}]
+ # Note: thinking_blocks is missing here - LiteLLM will handle it
+ },
+ {"role": "tool", "tool_call_id": "call_123", "content": "22°C sunny"}
+ ]
+)
+```
+
+
+
+
+```yaml showLineNumbers title="config.yaml"
+litellm_settings:
+ modify_params: true # Enable automatic parameter modification
+
+model_list:
+ - model_name: claude-thinking
+ litellm_params:
+ model: anthropic/claude-sonnet-4-20250514
+ thinking:
+ type: enabled
+ budget_tokens: 1024
+```
+
+
+
+
+:::info
+When `modify_params=True` and LiteLLM drops the `thinking` param, the model will **not** use extended thinking for that specific turn. The conversation will continue normally, but without reasoning for that response.
+:::
+
+**Correct way to include `thinking_blocks`:**
+
+```python
+# After receiving a response with tool_calls, include thinking_blocks when sending back:
+assistant_message = {
+ "role": "assistant",
+ "content": response.choices[0].message.content,
+ "tool_calls": [...],
+ "thinking_blocks": response.choices[0].message.thinking_blocks # ← Required!
+}
+```
+
+---
+
diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py
index 66439930f9a..628121ab11c 100644
--- a/litellm/llms/anthropic/chat/transformation.py
+++ b/litellm/llms/anthropic/chat/transformation.py
@@ -61,6 +61,7 @@ from litellm.utils import (
add_dummy_tool,
get_max_tokens,
has_tool_call_blocks,
+ last_assistant_with_tool_calls_has_no_thinking_blocks,
supports_reasoning,
token_counter,
)
@@ -999,6 +1000,20 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
llm_provider="anthropic",
)
+ # Drop thinking param if thinking is enabled but thinking_blocks are missing
+ # This prevents the error: "Expected thinking or redacted_thinking, but found tool_use"
+ if (
+ optional_params.get("thinking") is not None
+ and messages is not None
+ and last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
+ ):
+ if litellm.modify_params:
+ optional_params.pop("thinking", None)
+ litellm.verbose_logger.warning(
+ "Dropping 'thinking' param because the last assistant message with tool_calls "
+ "has no thinking_blocks. The model won't use extended thinking for this turn."
+ )
+
headers = self.update_headers_with_optional_anthropic_beta(
headers=headers, optional_params=optional_params
)
diff --git a/litellm/utils.py b/litellm/utils.py
index e22109712f7..f5ecc07cc1d 100644
--- a/litellm/utils.py
+++ b/litellm/utils.py
@@ -6875,6 +6875,34 @@ def has_tool_call_blocks(messages: List[AllMessageValues]) -> bool:
return False
+def last_assistant_with_tool_calls_has_no_thinking_blocks(
+ messages: List[AllMessageValues],
+) -> bool:
+ """
+ Returns true if the last assistant message with tool_calls has no thinking_blocks.
+
+ This is used to detect when thinking param should be dropped to avoid
+ Anthropic error: "Expected thinking or redacted_thinking, but found tool_use"
+
+ When thinking is enabled, assistant messages with tool_calls must include thinking_blocks.
+ If the client didn't preserve thinking_blocks, we need to drop the thinking param.
+
+ Related issues: https://github.com/BerriAI/litellm/issues/14194, https://github.com/BerriAI/litellm/issues/9020
+ """
+ # Find the last assistant message with tool_calls
+ last_assistant_with_tools = None
+ for message in messages:
+ if message.get("role") == "assistant" and message.get("tool_calls") is not None:
+ last_assistant_with_tools = message
+
+ if last_assistant_with_tools is None:
+ return False
+
+ # Check if it has thinking_blocks
+ thinking_blocks = last_assistant_with_tools.get("thinking_blocks")
+ return thinking_blocks is None or len(thinking_blocks) == 0
+
+
def add_dummy_tool(custom_llm_provider: str) -> List[ChatCompletionToolParam]:
"""
Prevent Anthropic from raising error when tool_use block exists but no tools are provided.