mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-30 01:52:18 +00:00
fix(thinking): recognize adaptive type and drop param for text-only responses
- Recognize 'adaptive' as a valid thinking type in is_thinking_enabled - Drop thinking param when assistant messages have text without thinking blocks - Use lazy import for last_assistant_message_has_no_thinking_blocks to avoid circular dependency (litellm.utils -> anthropic/bedrock -> litellm.utils)
This commit is contained in:
parent
e1462c9b17
commit
4f53bbf4d2
5 changed files with 264 additions and 16 deletions
|
|
@ -1851,16 +1851,24 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
)
|
||||
|
||||
# Drop thinking param if thinking is enabled but thinking_blocks are missing
|
||||
# This prevents the error: "Expected thinking or redacted_thinking, but found tool_use"
|
||||
# This prevents Anthropic errors:
|
||||
# - "Expected thinking or redacted_thinking, but found tool_use" (assistant with tool_calls)
|
||||
# - "Expected thinking or redacted_thinking, but found text" (assistant with text content)
|
||||
#
|
||||
# IMPORTANT: Only drop thinking if NO assistant messages have thinking_blocks.
|
||||
# If any message has thinking_blocks, we must keep thinking enabled, otherwise
|
||||
# Anthropic errors with: "When thinking is disabled, an assistant message cannot contain thinking"
|
||||
# Related issue: https://github.com/BerriAI/litellm/issues/18926
|
||||
# Lazy import to avoid circular dependency (utils -> anthropic -> utils)
|
||||
from litellm.utils import last_assistant_message_has_no_thinking_blocks
|
||||
|
||||
if (
|
||||
optional_params.get("thinking") is not None
|
||||
and messages is not None
|
||||
and last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
|
||||
and (
|
||||
last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
|
||||
or last_assistant_message_has_no_thinking_blocks(messages)
|
||||
)
|
||||
and not any_assistant_message_has_thinking_blocks(messages)
|
||||
):
|
||||
if litellm.modify_params:
|
||||
|
|
|
|||
|
|
@ -108,9 +108,11 @@ class BaseConfig(ABC):
|
|||
return type_to_response_format_param(response_format=response_format)
|
||||
|
||||
def is_thinking_enabled(self, non_default_params: dict) -> bool:
|
||||
return (non_default_params.get("thinking") or {}).get(
|
||||
"type"
|
||||
) == "enabled" or non_default_params.get("reasoning_effort") is not None
|
||||
thinking_type = (non_default_params.get("thinking") or {}).get("type")
|
||||
return (
|
||||
thinking_type in ("enabled", "adaptive")
|
||||
or non_default_params.get("reasoning_effort") is not None
|
||||
)
|
||||
|
||||
def is_max_tokens_in_request(self, non_default_params: dict) -> bool:
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -1593,15 +1593,23 @@ class AmazonConverseConfig(BaseConfig):
|
|||
)
|
||||
|
||||
# Drop thinking param if thinking is enabled but thinking_blocks are missing
|
||||
# This prevents the error: "Expected thinking or redacted_thinking, but found tool_use"
|
||||
# This prevents Anthropic errors:
|
||||
# - "Expected thinking or redacted_thinking, but found tool_use" (assistant with tool_calls)
|
||||
# - "Expected thinking or redacted_thinking, but found text" (assistant with text content)
|
||||
#
|
||||
# IMPORTANT: Only drop thinking if NO assistant messages have thinking_blocks.
|
||||
# If any message has thinking_blocks, we must keep thinking enabled, otherwise
|
||||
# Related issues: https://github.com/BerriAI/litellm/issues/14194
|
||||
# Lazy import to avoid circular dependency (utils -> bedrock -> utils)
|
||||
from litellm.utils import last_assistant_message_has_no_thinking_blocks
|
||||
|
||||
if (
|
||||
optional_params.get("thinking") is not None
|
||||
and messages is not None
|
||||
and last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
|
||||
and (
|
||||
last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
|
||||
or last_assistant_message_has_no_thinking_blocks(messages)
|
||||
)
|
||||
and not any_assistant_message_has_thinking_blocks(messages)
|
||||
):
|
||||
if litellm.modify_params:
|
||||
|
|
|
|||
|
|
@ -7873,6 +7873,33 @@ def has_tool_call_blocks(messages: List[AllMessageValues]) -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def _message_has_thinking_blocks(message: AllMessageValues) -> bool:
|
||||
"""
|
||||
Check if a single assistant message has thinking blocks.
|
||||
|
||||
Checks both the 'thinking_blocks' field (LiteLLM/OpenAI format) and
|
||||
the 'content' array for thinking/redacted_thinking blocks (Anthropic format).
|
||||
"""
|
||||
# Check thinking_blocks field (LiteLLM/OpenAI format)
|
||||
thinking_blocks = message.get("thinking_blocks")
|
||||
if thinking_blocks is not None and (
|
||||
not hasattr(thinking_blocks, "__len__") or len(thinking_blocks) > 0
|
||||
):
|
||||
return True
|
||||
|
||||
# Check content array for thinking blocks (Anthropic format)
|
||||
content = message.get("content")
|
||||
if isinstance(content, list):
|
||||
for block in content:
|
||||
if isinstance(block, dict) and block.get("type") in (
|
||||
"thinking",
|
||||
"redacted_thinking",
|
||||
):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def any_assistant_message_has_thinking_blocks(
|
||||
messages: List[AllMessageValues],
|
||||
) -> bool:
|
||||
|
|
@ -7888,10 +7915,7 @@ def any_assistant_message_has_thinking_blocks(
|
|||
"""
|
||||
for message in messages:
|
||||
if message.get("role") == "assistant":
|
||||
thinking_blocks = message.get("thinking_blocks")
|
||||
if thinking_blocks is not None and (
|
||||
not hasattr(thinking_blocks, "__len__") or len(thinking_blocks) > 0
|
||||
):
|
||||
if _message_has_thinking_blocks(message):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
|
@ -7923,11 +7947,46 @@ def last_assistant_with_tool_calls_has_no_thinking_blocks(
|
|||
if last_assistant_with_tools is None:
|
||||
return False
|
||||
|
||||
# Check if it has thinking_blocks
|
||||
thinking_blocks = last_assistant_with_tools.get("thinking_blocks")
|
||||
return thinking_blocks is None or (
|
||||
hasattr(thinking_blocks, "__len__") and len(thinking_blocks) == 0
|
||||
)
|
||||
return not _message_has_thinking_blocks(last_assistant_with_tools)
|
||||
|
||||
|
||||
def last_assistant_message_has_no_thinking_blocks(
|
||||
messages: List[AllMessageValues],
|
||||
) -> bool:
|
||||
"""
|
||||
Returns true if the last assistant message has content but no thinking_blocks.
|
||||
|
||||
This is used to detect when thinking param should be dropped to avoid
|
||||
Anthropic error: "Expected thinking or redacted_thinking, but found text"
|
||||
|
||||
When thinking is enabled, ALL assistant messages must start with thinking_blocks.
|
||||
If the client didn't preserve thinking_blocks, we need to drop the thinking param.
|
||||
|
||||
IMPORTANT: This should only be used in conjunction with
|
||||
any_assistant_message_has_thinking_blocks() to ensure we don't drop thinking
|
||||
when other messages in the conversation contain thinking blocks.
|
||||
"""
|
||||
# Only relevant if thinking was previously active in this conversation.
|
||||
# Without prior thinking blocks, a text-only assistant message just means
|
||||
# thinking was never enabled — not that blocks were stripped.
|
||||
if not any_assistant_message_has_thinking_blocks(messages):
|
||||
return False
|
||||
|
||||
# Find the last assistant message
|
||||
last_assistant = None
|
||||
for message in messages:
|
||||
if message.get("role") == "assistant":
|
||||
last_assistant = message
|
||||
|
||||
if last_assistant is None:
|
||||
return False
|
||||
|
||||
# Only flag if message has content (empty messages aren't an issue)
|
||||
content = last_assistant.get("content")
|
||||
if not content:
|
||||
return False
|
||||
|
||||
return not _message_has_thinking_blocks(last_assistant)
|
||||
|
||||
|
||||
def add_dummy_tool(custom_llm_provider: str) -> List[ChatCompletionToolParam]:
|
||||
|
|
|
|||
|
|
@ -3698,6 +3698,177 @@ def test_last_assistant_with_tool_calls_has_no_thinking_blocks_issue_18926():
|
|||
assert should_drop_thinking is False
|
||||
|
||||
|
||||
def test_last_assistant_message_has_no_thinking_blocks_text_only():
|
||||
"""
|
||||
Test that the function only fires when thinking was previously active.
|
||||
|
||||
A fresh conversation (no prior thinking blocks) must NOT cause thinking to be
|
||||
dropped — the user may simply be enabling thinking for the first time.
|
||||
"""
|
||||
from litellm.utils import (
|
||||
any_assistant_message_has_thinking_blocks,
|
||||
last_assistant_message_has_no_thinking_blocks,
|
||||
last_assistant_with_tool_calls_has_no_thinking_blocks,
|
||||
)
|
||||
|
||||
# Scenario 1: fresh conversation, thinking never used — must NOT drop
|
||||
messages_no_prior_thinking = [
|
||||
{"role": "user", "content": "Hello"},
|
||||
{"role": "assistant", "content": "Hi there!"},
|
||||
{"role": "user", "content": "What's 2+2?"},
|
||||
{"role": "assistant", "content": "4"},
|
||||
{"role": "user", "content": "Thanks"},
|
||||
]
|
||||
assert (
|
||||
last_assistant_with_tool_calls_has_no_thinking_blocks(
|
||||
messages_no_prior_thinking
|
||||
)
|
||||
is False
|
||||
)
|
||||
assert (
|
||||
any_assistant_message_has_thinking_blocks(messages_no_prior_thinking) is False
|
||||
)
|
||||
# Must return False — no evidence thinking was ever enabled
|
||||
assert (
|
||||
last_assistant_message_has_no_thinking_blocks(messages_no_prior_thinking)
|
||||
is False
|
||||
)
|
||||
|
||||
# Scenario 2: thinking was used before, but last message has no blocks — MUST drop
|
||||
messages_with_prior_thinking = [
|
||||
{"role": "user", "content": "Hello"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "thinking", "thinking": "Let me think..."},
|
||||
{"type": "text", "text": "Hi!"},
|
||||
],
|
||||
},
|
||||
{"role": "user", "content": "What's 2+2?"},
|
||||
{"role": "assistant", "content": "4"}, # blocks stripped by client
|
||||
{"role": "user", "content": "Thanks"},
|
||||
]
|
||||
assert (
|
||||
any_assistant_message_has_thinking_blocks(messages_with_prior_thinking) is True
|
||||
)
|
||||
assert (
|
||||
last_assistant_message_has_no_thinking_blocks(messages_with_prior_thinking)
|
||||
is True
|
||||
)
|
||||
|
||||
|
||||
def test_last_assistant_message_has_no_thinking_blocks_with_content_list():
|
||||
"""
|
||||
Test detection when last assistant has content list but no thinking blocks,
|
||||
only when prior thinking blocks exist in the conversation.
|
||||
"""
|
||||
from litellm.utils import last_assistant_message_has_no_thinking_blocks
|
||||
|
||||
# No prior thinking — should NOT drop
|
||||
messages_no_prior = [
|
||||
{"role": "user", "content": "Hello"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [{"type": "text", "text": "Hi there!"}],
|
||||
},
|
||||
]
|
||||
assert last_assistant_message_has_no_thinking_blocks(messages_no_prior) is False
|
||||
|
||||
# Prior thinking exists, last message has none — SHOULD drop
|
||||
messages_with_prior = [
|
||||
{"role": "user", "content": "Hello"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "thinking", "thinking": "..."},
|
||||
{"type": "text", "text": "First answer"},
|
||||
],
|
||||
},
|
||||
{"role": "user", "content": "Follow up"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [{"type": "text", "text": "Second answer"}], # blocks stripped
|
||||
},
|
||||
]
|
||||
assert last_assistant_message_has_no_thinking_blocks(messages_with_prior) is True
|
||||
|
||||
|
||||
def test_last_assistant_message_has_thinking_in_content():
|
||||
"""
|
||||
Test that function returns False when thinking blocks are in content array
|
||||
(Anthropic format) rather than in the thinking_blocks field.
|
||||
"""
|
||||
from litellm.utils import (
|
||||
any_assistant_message_has_thinking_blocks,
|
||||
last_assistant_message_has_no_thinking_blocks,
|
||||
)
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "Hello"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "thinking", "thinking": "Let me think..."},
|
||||
{"type": "text", "text": "The answer is 42."},
|
||||
],
|
||||
},
|
||||
]
|
||||
|
||||
# Content has thinking blocks, so should return False
|
||||
assert last_assistant_message_has_no_thinking_blocks(messages) is False
|
||||
|
||||
# any_assistant check should also detect thinking blocks in content
|
||||
assert any_assistant_message_has_thinking_blocks(messages) is True
|
||||
|
||||
|
||||
def test_last_assistant_message_no_content():
|
||||
"""
|
||||
Test that function returns False when last assistant has no content.
|
||||
"""
|
||||
from litellm.utils import last_assistant_message_has_no_thinking_blocks
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "Hello"},
|
||||
{"role": "assistant", "content": None},
|
||||
]
|
||||
|
||||
assert last_assistant_message_has_no_thinking_blocks(messages) is False
|
||||
|
||||
|
||||
def test_no_assistant_messages():
|
||||
"""
|
||||
Test that function returns False when there are no assistant messages.
|
||||
"""
|
||||
from litellm.utils import last_assistant_message_has_no_thinking_blocks
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "Hello"},
|
||||
]
|
||||
|
||||
assert last_assistant_message_has_no_thinking_blocks(messages) is False
|
||||
|
||||
|
||||
def test_thinking_blocks_field_detected_by_any_check():
|
||||
"""
|
||||
Test that any_assistant_message_has_thinking_blocks detects thinking blocks
|
||||
in both the thinking_blocks field and in the content array.
|
||||
"""
|
||||
from litellm.utils import any_assistant_message_has_thinking_blocks
|
||||
|
||||
# Thinking in content array (Anthropic format)
|
||||
messages_content = [
|
||||
{"role": "user", "content": "Hello"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": [
|
||||
{"type": "redacted_thinking", "data": "xxx"},
|
||||
{"type": "text", "text": "answer"},
|
||||
],
|
||||
},
|
||||
]
|
||||
assert any_assistant_message_has_thinking_blocks(messages_content) is True
|
||||
|
||||
|
||||
class TestAdditionalDropParamsForNonOpenAIProviders:
|
||||
"""
|
||||
Test additional_drop_params functionality for non-OpenAI providers.
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue