fix(thinking): recognize adaptive type and drop param for text-only responses

- Recognize 'adaptive' as a valid thinking type in is_thinking_enabled
- Drop thinking param when assistant messages have text without thinking blocks
- Use lazy import for last_assistant_message_has_no_thinking_blocks to
  avoid circular dependency (litellm.utils -> anthropic/bedrock -> litellm.utils)
This commit is contained in:
Quentin Machu 2026-05-03 23:12:15 -04:00 • committed by Quentin Machu
parent e1462c9b17
commit 4f53bbf4d2
No known key found for this signature in database
5 changed files with 264 additions and 16 deletions

View file

@ -1851,16 +1851,24 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
)
# Drop thinking param if thinking is enabled but thinking_blocks are missing
# This prevents the error: "Expected thinking or redacted_thinking, but found tool_use"
# This prevents Anthropic errors:
# - "Expected thinking or redacted_thinking, but found tool_use" (assistant with tool_calls)
# - "Expected thinking or redacted_thinking, but found text" (assistant with text content)
#
# IMPORTANT: Only drop thinking if NO assistant messages have thinking_blocks.
# If any message has thinking_blocks, we must keep thinking enabled, otherwise
# Anthropic errors with: "When thinking is disabled, an assistant message cannot contain thinking"
# Related issue: https://github.com/BerriAI/litellm/issues/18926
# Lazy import to avoid circular dependency (utils -> anthropic -> utils)
from litellm.utils import last_assistant_message_has_no_thinking_blocks
if (
optional_params.get("thinking") is not None
and messages is not None
and last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
and (
last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
or last_assistant_message_has_no_thinking_blocks(messages)
)
and not any_assistant_message_has_thinking_blocks(messages)
):
if litellm.modify_params:

View file

@ -108,9 +108,11 @@ class BaseConfig(ABC):
return type_to_response_format_param(response_format=response_format)
def is_thinking_enabled(self, non_default_params: dict) -> bool:
return (non_default_params.get("thinking") or {}).get(
"type"
) == "enabled" or non_default_params.get("reasoning_effort") is not None
thinking_type = (non_default_params.get("thinking") or {}).get("type")
return (
thinking_type in ("enabled", "adaptive")
or non_default_params.get("reasoning_effort") is not None
)
def is_max_tokens_in_request(self, non_default_params: dict) -> bool:
"""

View file

@ -1593,15 +1593,23 @@ class AmazonConverseConfig(BaseConfig):
)
# Drop thinking param if thinking is enabled but thinking_blocks are missing
# This prevents the error: "Expected thinking or redacted_thinking, but found tool_use"
# This prevents Anthropic errors:
# - "Expected thinking or redacted_thinking, but found tool_use" (assistant with tool_calls)
# - "Expected thinking or redacted_thinking, but found text" (assistant with text content)
#
# IMPORTANT: Only drop thinking if NO assistant messages have thinking_blocks.
# If any message has thinking_blocks, we must keep thinking enabled, otherwise
# Related issues: https://github.com/BerriAI/litellm/issues/14194
# Lazy import to avoid circular dependency (utils -> bedrock -> utils)
from litellm.utils import last_assistant_message_has_no_thinking_blocks
if (
optional_params.get("thinking") is not None
and messages is not None
and last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
and (
last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
or last_assistant_message_has_no_thinking_blocks(messages)
)
and not any_assistant_message_has_thinking_blocks(messages)
):
if litellm.modify_params:

View file

@ -7873,6 +7873,33 @@ def has_tool_call_blocks(messages: List[AllMessageValues]) -> bool:
return False
def _message_has_thinking_blocks(message: AllMessageValues) -> bool:
"""
Check if a single assistant message has thinking blocks.
Checks both the 'thinking_blocks' field (LiteLLM/OpenAI format) and
the 'content' array for thinking/redacted_thinking blocks (Anthropic format).
"""
# Check thinking_blocks field (LiteLLM/OpenAI format)
thinking_blocks = message.get("thinking_blocks")
if thinking_blocks is not None and (
not hasattr(thinking_blocks, "__len__") or len(thinking_blocks) > 0
):
return True
# Check content array for thinking blocks (Anthropic format)
content = message.get("content")
if isinstance(content, list):
for block in content:
if isinstance(block, dict) and block.get("type") in (
"thinking",
"redacted_thinking",
):
return True
return False
def any_assistant_message_has_thinking_blocks(
messages: List[AllMessageValues],
) -> bool:
@ -7888,10 +7915,7 @@ def any_assistant_message_has_thinking_blocks(
"""
for message in messages:
if message.get("role") == "assistant":
thinking_blocks = message.get("thinking_blocks")
if thinking_blocks is not None and (
not hasattr(thinking_blocks, "__len__") or len(thinking_blocks) > 0
):
if _message_has_thinking_blocks(message):
return True
return False
@ -7923,11 +7947,46 @@ def last_assistant_with_tool_calls_has_no_thinking_blocks(
if last_assistant_with_tools is None:
return False
# Check if it has thinking_blocks
thinking_blocks = last_assistant_with_tools.get("thinking_blocks")
return thinking_blocks is None or (
hasattr(thinking_blocks, "__len__") and len(thinking_blocks) == 0
)
return not _message_has_thinking_blocks(last_assistant_with_tools)
def last_assistant_message_has_no_thinking_blocks(
messages: List[AllMessageValues],
) -> bool:
"""
Returns true if the last assistant message has content but no thinking_blocks.
This is used to detect when thinking param should be dropped to avoid
Anthropic error: "Expected thinking or redacted_thinking, but found text"
When thinking is enabled, ALL assistant messages must start with thinking_blocks.
If the client didn't preserve thinking_blocks, we need to drop the thinking param.
IMPORTANT: This should only be used in conjunction with
any_assistant_message_has_thinking_blocks() to ensure we don't drop thinking
when other messages in the conversation contain thinking blocks.
"""
# Only relevant if thinking was previously active in this conversation.
# Without prior thinking blocks, a text-only assistant message just means
# thinking was never enabled — not that blocks were stripped.
if not any_assistant_message_has_thinking_blocks(messages):
return False
# Find the last assistant message
last_assistant = None
for message in messages:
if message.get("role") == "assistant":
last_assistant = message
if last_assistant is None:
return False
# Only flag if message has content (empty messages aren't an issue)
content = last_assistant.get("content")
if not content:
return False
return not _message_has_thinking_blocks(last_assistant)
def add_dummy_tool(custom_llm_provider: str) -> List[ChatCompletionToolParam]:

View file

@ -3698,6 +3698,177 @@ def test_last_assistant_with_tool_calls_has_no_thinking_blocks_issue_18926():
assert should_drop_thinking is False
def test_last_assistant_message_has_no_thinking_blocks_text_only():
"""
Test that the function only fires when thinking was previously active.
A fresh conversation (no prior thinking blocks) must NOT cause thinking to be
dropped — the user may simply be enabling thinking for the first time.
"""
from litellm.utils import (
any_assistant_message_has_thinking_blocks,
last_assistant_message_has_no_thinking_blocks,
last_assistant_with_tool_calls_has_no_thinking_blocks,
)
# Scenario 1: fresh conversation, thinking never used — must NOT drop
messages_no_prior_thinking = [
{"role": "user", "content": "Hello"},
{"role": "assistant", "content": "Hi there!"},
{"role": "user", "content": "What's 2+2?"},
{"role": "assistant", "content": "4"},
{"role": "user", "content": "Thanks"},
]
assert (
last_assistant_with_tool_calls_has_no_thinking_blocks(
messages_no_prior_thinking
)
is False
)
assert (
any_assistant_message_has_thinking_blocks(messages_no_prior_thinking) is False
)
# Must return False — no evidence thinking was ever enabled
assert (
last_assistant_message_has_no_thinking_blocks(messages_no_prior_thinking)
is False
)
# Scenario 2: thinking was used before, but last message has no blocks — MUST drop
messages_with_prior_thinking = [
{"role": "user", "content": "Hello"},
{
"role": "assistant",
"content": [
{"type": "thinking", "thinking": "Let me think..."},
{"type": "text", "text": "Hi!"},
],
},
{"role": "user", "content": "What's 2+2?"},
{"role": "assistant", "content": "4"}, # blocks stripped by client
{"role": "user", "content": "Thanks"},
]
assert (
any_assistant_message_has_thinking_blocks(messages_with_prior_thinking) is True
)
assert (
last_assistant_message_has_no_thinking_blocks(messages_with_prior_thinking)
is True
)
def test_last_assistant_message_has_no_thinking_blocks_with_content_list():
"""
Test detection when last assistant has content list but no thinking blocks,
only when prior thinking blocks exist in the conversation.
"""
from litellm.utils import last_assistant_message_has_no_thinking_blocks
# No prior thinking — should NOT drop
messages_no_prior = [
{"role": "user", "content": "Hello"},
{
"role": "assistant",
"content": [{"type": "text", "text": "Hi there!"}],
},
]
assert last_assistant_message_has_no_thinking_blocks(messages_no_prior) is False
# Prior thinking exists, last message has none — SHOULD drop
messages_with_prior = [
{"role": "user", "content": "Hello"},
{
"role": "assistant",
"content": [
{"type": "thinking", "thinking": "..."},
{"type": "text", "text": "First answer"},
],
},
{"role": "user", "content": "Follow up"},
{
"role": "assistant",
"content": [{"type": "text", "text": "Second answer"}], # blocks stripped
},
]
assert last_assistant_message_has_no_thinking_blocks(messages_with_prior) is True
def test_last_assistant_message_has_thinking_in_content():
"""
Test that function returns False when thinking blocks are in content array
(Anthropic format) rather than in the thinking_blocks field.
"""
from litellm.utils import (
any_assistant_message_has_thinking_blocks,
last_assistant_message_has_no_thinking_blocks,
)
messages = [
{"role": "user", "content": "Hello"},
{
"role": "assistant",
"content": [
{"type": "thinking", "thinking": "Let me think..."},
{"type": "text", "text": "The answer is 42."},
],
},
]
# Content has thinking blocks, so should return False
assert last_assistant_message_has_no_thinking_blocks(messages) is False
# any_assistant check should also detect thinking blocks in content
assert any_assistant_message_has_thinking_blocks(messages) is True
def test_last_assistant_message_no_content():
"""
Test that function returns False when last assistant has no content.
"""
from litellm.utils import last_assistant_message_has_no_thinking_blocks
messages = [
{"role": "user", "content": "Hello"},
{"role": "assistant", "content": None},
]
assert last_assistant_message_has_no_thinking_blocks(messages) is False
def test_no_assistant_messages():
"""
Test that function returns False when there are no assistant messages.
"""
from litellm.utils import last_assistant_message_has_no_thinking_blocks
messages = [
{"role": "user", "content": "Hello"},
]
assert last_assistant_message_has_no_thinking_blocks(messages) is False
def test_thinking_blocks_field_detected_by_any_check():
"""
Test that any_assistant_message_has_thinking_blocks detects thinking blocks
in both the thinking_blocks field and in the content array.
"""
from litellm.utils import any_assistant_message_has_thinking_blocks
# Thinking in content array (Anthropic format)
messages_content = [
{"role": "user", "content": "Hello"},
{
"role": "assistant",
"content": [
{"type": "redacted_thinking", "data": "xxx"},
{"type": "text", "text": "answer"},
],
},
]
assert any_assistant_message_has_thinking_blocks(messages_content) is True
class TestAdditionalDropParamsForNonOpenAIProviders:
"""
Test additional_drop_params functionality for non-OpenAI providers.