diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 57609cfcd26..853c236ae03 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1851,16 +1851,24 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): ) # Drop thinking param if thinking is enabled but thinking_blocks are missing - # This prevents the error: "Expected thinking or redacted_thinking, but found tool_use" + # This prevents Anthropic errors: + # - "Expected thinking or redacted_thinking, but found tool_use" (assistant with tool_calls) + # - "Expected thinking or redacted_thinking, but found text" (assistant with text content) # # IMPORTANT: Only drop thinking if NO assistant messages have thinking_blocks. # If any message has thinking_blocks, we must keep thinking enabled, otherwise # Anthropic errors with: "When thinking is disabled, an assistant message cannot contain thinking" # Related issue: https://github.com/BerriAI/litellm/issues/18926 + # Lazy import to avoid circular dependency (utils -> anthropic -> utils) + from litellm.utils import last_assistant_message_has_no_thinking_blocks + if ( optional_params.get("thinking") is not None and messages is not None - and last_assistant_with_tool_calls_has_no_thinking_blocks(messages) + and ( + last_assistant_with_tool_calls_has_no_thinking_blocks(messages) + or last_assistant_message_has_no_thinking_blocks(messages) + ) and not any_assistant_message_has_thinking_blocks(messages) ): if litellm.modify_params: diff --git a/litellm/llms/base_llm/chat/transformation.py b/litellm/llms/base_llm/chat/transformation.py index 5f35a58ce1f..0ce18ff39f8 100644 --- a/litellm/llms/base_llm/chat/transformation.py +++ b/litellm/llms/base_llm/chat/transformation.py @@ -108,9 +108,11 @@ class BaseConfig(ABC): return type_to_response_format_param(response_format=response_format) def is_thinking_enabled(self, non_default_params: dict) -> bool: - return (non_default_params.get("thinking") or {}).get( - "type" - ) == "enabled" or non_default_params.get("reasoning_effort") is not None + thinking_type = (non_default_params.get("thinking") or {}).get("type") + return ( + thinking_type in ("enabled", "adaptive") + or non_default_params.get("reasoning_effort") is not None + ) def is_max_tokens_in_request(self, non_default_params: dict) -> bool: """ diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index bd6c5b6e7cd..a63476eefc5 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -1593,15 +1593,23 @@ class AmazonConverseConfig(BaseConfig): ) # Drop thinking param if thinking is enabled but thinking_blocks are missing - # This prevents the error: "Expected thinking or redacted_thinking, but found tool_use" + # This prevents Anthropic errors: + # - "Expected thinking or redacted_thinking, but found tool_use" (assistant with tool_calls) + # - "Expected thinking or redacted_thinking, but found text" (assistant with text content) # # IMPORTANT: Only drop thinking if NO assistant messages have thinking_blocks. # If any message has thinking_blocks, we must keep thinking enabled, otherwise # Related issues: https://github.com/BerriAI/litellm/issues/14194 + # Lazy import to avoid circular dependency (utils -> bedrock -> utils) + from litellm.utils import last_assistant_message_has_no_thinking_blocks + if ( optional_params.get("thinking") is not None and messages is not None - and last_assistant_with_tool_calls_has_no_thinking_blocks(messages) + and ( + last_assistant_with_tool_calls_has_no_thinking_blocks(messages) + or last_assistant_message_has_no_thinking_blocks(messages) + ) and not any_assistant_message_has_thinking_blocks(messages) ): if litellm.modify_params: diff --git a/litellm/utils.py b/litellm/utils.py index 5a9dccc089e..92a9777864a 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -7873,6 +7873,33 @@ def has_tool_call_blocks(messages: List[AllMessageValues]) -> bool: return False +def _message_has_thinking_blocks(message: AllMessageValues) -> bool: + """ + Check if a single assistant message has thinking blocks. + + Checks both the 'thinking_blocks' field (LiteLLM/OpenAI format) and + the 'content' array for thinking/redacted_thinking blocks (Anthropic format). + """ + # Check thinking_blocks field (LiteLLM/OpenAI format) + thinking_blocks = message.get("thinking_blocks") + if thinking_blocks is not None and ( + not hasattr(thinking_blocks, "__len__") or len(thinking_blocks) > 0 + ): + return True + + # Check content array for thinking blocks (Anthropic format) + content = message.get("content") + if isinstance(content, list): + for block in content: + if isinstance(block, dict) and block.get("type") in ( + "thinking", + "redacted_thinking", + ): + return True + + return False + + def any_assistant_message_has_thinking_blocks( messages: List[AllMessageValues], ) -> bool: @@ -7888,10 +7915,7 @@ def any_assistant_message_has_thinking_blocks( """ for message in messages: if message.get("role") == "assistant": - thinking_blocks = message.get("thinking_blocks") - if thinking_blocks is not None and ( - not hasattr(thinking_blocks, "__len__") or len(thinking_blocks) > 0 - ): + if _message_has_thinking_blocks(message): return True return False @@ -7923,11 +7947,46 @@ def last_assistant_with_tool_calls_has_no_thinking_blocks( if last_assistant_with_tools is None: return False - # Check if it has thinking_blocks - thinking_blocks = last_assistant_with_tools.get("thinking_blocks") - return thinking_blocks is None or ( - hasattr(thinking_blocks, "__len__") and len(thinking_blocks) == 0 - ) + return not _message_has_thinking_blocks(last_assistant_with_tools) + + +def last_assistant_message_has_no_thinking_blocks( + messages: List[AllMessageValues], +) -> bool: + """ + Returns true if the last assistant message has content but no thinking_blocks. + + This is used to detect when thinking param should be dropped to avoid + Anthropic error: "Expected thinking or redacted_thinking, but found text" + + When thinking is enabled, ALL assistant messages must start with thinking_blocks. + If the client didn't preserve thinking_blocks, we need to drop the thinking param. + + IMPORTANT: This should only be used in conjunction with + any_assistant_message_has_thinking_blocks() to ensure we don't drop thinking + when other messages in the conversation contain thinking blocks. + """ + # Only relevant if thinking was previously active in this conversation. + # Without prior thinking blocks, a text-only assistant message just means + # thinking was never enabled — not that blocks were stripped. + if not any_assistant_message_has_thinking_blocks(messages): + return False + + # Find the last assistant message + last_assistant = None + for message in messages: + if message.get("role") == "assistant": + last_assistant = message + + if last_assistant is None: + return False + + # Only flag if message has content (empty messages aren't an issue) + content = last_assistant.get("content") + if not content: + return False + + return not _message_has_thinking_blocks(last_assistant) def add_dummy_tool(custom_llm_provider: str) -> List[ChatCompletionToolParam]: diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 6a78653ec99..2647a523fb6 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -3698,6 +3698,177 @@ def test_last_assistant_with_tool_calls_has_no_thinking_blocks_issue_18926(): assert should_drop_thinking is False +def test_last_assistant_message_has_no_thinking_blocks_text_only(): + """ + Test that the function only fires when thinking was previously active. + + A fresh conversation (no prior thinking blocks) must NOT cause thinking to be + dropped — the user may simply be enabling thinking for the first time. + """ + from litellm.utils import ( + any_assistant_message_has_thinking_blocks, + last_assistant_message_has_no_thinking_blocks, + last_assistant_with_tool_calls_has_no_thinking_blocks, + ) + + # Scenario 1: fresh conversation, thinking never used — must NOT drop + messages_no_prior_thinking = [ + {"role": "user", "content": "Hello"}, + {"role": "assistant", "content": "Hi there!"}, + {"role": "user", "content": "What's 2+2?"}, + {"role": "assistant", "content": "4"}, + {"role": "user", "content": "Thanks"}, + ] + assert ( + last_assistant_with_tool_calls_has_no_thinking_blocks( + messages_no_prior_thinking + ) + is False + ) + assert ( + any_assistant_message_has_thinking_blocks(messages_no_prior_thinking) is False + ) + # Must return False — no evidence thinking was ever enabled + assert ( + last_assistant_message_has_no_thinking_blocks(messages_no_prior_thinking) + is False + ) + + # Scenario 2: thinking was used before, but last message has no blocks — MUST drop + messages_with_prior_thinking = [ + {"role": "user", "content": "Hello"}, + { + "role": "assistant", + "content": [ + {"type": "thinking", "thinking": "Let me think..."}, + {"type": "text", "text": "Hi!"}, + ], + }, + {"role": "user", "content": "What's 2+2?"}, + {"role": "assistant", "content": "4"}, # blocks stripped by client + {"role": "user", "content": "Thanks"}, + ] + assert ( + any_assistant_message_has_thinking_blocks(messages_with_prior_thinking) is True + ) + assert ( + last_assistant_message_has_no_thinking_blocks(messages_with_prior_thinking) + is True + ) + + +def test_last_assistant_message_has_no_thinking_blocks_with_content_list(): + """ + Test detection when last assistant has content list but no thinking blocks, + only when prior thinking blocks exist in the conversation. + """ + from litellm.utils import last_assistant_message_has_no_thinking_blocks + + # No prior thinking — should NOT drop + messages_no_prior = [ + {"role": "user", "content": "Hello"}, + { + "role": "assistant", + "content": [{"type": "text", "text": "Hi there!"}], + }, + ] + assert last_assistant_message_has_no_thinking_blocks(messages_no_prior) is False + + # Prior thinking exists, last message has none — SHOULD drop + messages_with_prior = [ + {"role": "user", "content": "Hello"}, + { + "role": "assistant", + "content": [ + {"type": "thinking", "thinking": "..."}, + {"type": "text", "text": "First answer"}, + ], + }, + {"role": "user", "content": "Follow up"}, + { + "role": "assistant", + "content": [{"type": "text", "text": "Second answer"}], # blocks stripped + }, + ] + assert last_assistant_message_has_no_thinking_blocks(messages_with_prior) is True + + +def test_last_assistant_message_has_thinking_in_content(): + """ + Test that function returns False when thinking blocks are in content array + (Anthropic format) rather than in the thinking_blocks field. + """ + from litellm.utils import ( + any_assistant_message_has_thinking_blocks, + last_assistant_message_has_no_thinking_blocks, + ) + + messages = [ + {"role": "user", "content": "Hello"}, + { + "role": "assistant", + "content": [ + {"type": "thinking", "thinking": "Let me think..."}, + {"type": "text", "text": "The answer is 42."}, + ], + }, + ] + + # Content has thinking blocks, so should return False + assert last_assistant_message_has_no_thinking_blocks(messages) is False + + # any_assistant check should also detect thinking blocks in content + assert any_assistant_message_has_thinking_blocks(messages) is True + + +def test_last_assistant_message_no_content(): + """ + Test that function returns False when last assistant has no content. + """ + from litellm.utils import last_assistant_message_has_no_thinking_blocks + + messages = [ + {"role": "user", "content": "Hello"}, + {"role": "assistant", "content": None}, + ] + + assert last_assistant_message_has_no_thinking_blocks(messages) is False + + +def test_no_assistant_messages(): + """ + Test that function returns False when there are no assistant messages. + """ + from litellm.utils import last_assistant_message_has_no_thinking_blocks + + messages = [ + {"role": "user", "content": "Hello"}, + ] + + assert last_assistant_message_has_no_thinking_blocks(messages) is False + + +def test_thinking_blocks_field_detected_by_any_check(): + """ + Test that any_assistant_message_has_thinking_blocks detects thinking blocks + in both the thinking_blocks field and in the content array. + """ + from litellm.utils import any_assistant_message_has_thinking_blocks + + # Thinking in content array (Anthropic format) + messages_content = [ + {"role": "user", "content": "Hello"}, + { + "role": "assistant", + "content": [ + {"type": "redacted_thinking", "data": "xxx"}, + {"type": "text", "text": "answer"}, + ], + }, + ] + assert any_assistant_message_has_thinking_blocks(messages_content) is True + + class TestAdditionalDropParamsForNonOpenAIProviders: """ Test additional_drop_params functionality for non-OpenAI providers.