From 81fefc69c9b77470d94e2247e910380dd0972810 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Benedikt=20=C3=93skarsson?= Date: Thu, 8 Jan 2026 00:44:59 +0000 Subject: [PATCH 1/3] fix(bedrock): handle thinking with tool calls for Claude 4 models --- .../bedrock/chat/converse_transformation.py | 38 +- litellm/llms/bedrock/chat/invoke_handler.py | 36 +- litellm/utils.py | 58 ++- .../chat/test_converse_transformation.py | 486 +++++++++++------- 4 files changed, 398 insertions(+), 220 deletions(-) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 13dbec3952a..9935cce68dc 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -53,7 +53,12 @@ from litellm.types.utils import ( PromptTokensDetailsWrapper, Usage, ) -from litellm.utils import add_dummy_tool, has_tool_call_blocks, supports_reasoning +from litellm.utils import ( + add_dummy_tool, + has_tool_call_blocks, + last_assistant_with_tool_calls_has_no_thinking_blocks, + supports_reasoning, +) from ..common_utils import ( BedrockError, @@ -729,7 +734,7 @@ class AmazonConverseConfig(BaseConfig): return optional_params """ - Follow similar approach to anthropic - translate to a single tool call. + Follow similar approach to anthropic - translate to a single tool call. When using tools in this way: - https://docs.anthropic.com/en/docs/build-with-claude/tool-use#json-mode - You usually want to provide a single tool @@ -912,16 +917,16 @@ class AmazonConverseConfig(BaseConfig): inference_params = { k: v for k, v in inference_params.items() if k in total_supported_params } - + # Only set the topK value in for models that support it additional_request_params.update( self._handle_top_k_value(model, inference_params) ) - + # Filter out internal/MCP-related parameters that shouldn't be sent to the API # These are LiteLLM internal parameters, not API parameters additional_request_params = filter_internal_params(additional_request_params) - + # Filter out non-serializable objects (exceptions, callables, logging objects, etc.) # from additional_request_params to prevent JSON serialization errors # This filters: Exception objects, callable objects (functions), Logging objects, etc. @@ -1021,9 +1026,24 @@ class AmazonConverseConfig(BaseConfig): llm_provider="bedrock", ) + # Drop thinking param if thinking is enabled but thinking_blocks are missing + # This prevents the error: "Expected thinking or redacted_thinking, but found tool_use" + # Related issues: https://github.com/BerriAI/litellm/issues/14194 + if ( + optional_params.get("thinking") is not None + and messages is not None + and last_assistant_with_tool_calls_has_no_thinking_blocks(messages) + ): + if litellm.modify_params: + optional_params.pop("thinking", None) + litellm.verbose_logger.warning( + "Dropping 'thinking' param because the last assistant message with tool_calls " + "has no thinking_blocks. The model won't use extended thinking for this turn." + ) + # Prepare and separate parameters - inference_params, additional_request_params, request_metadata = ( - self._prepare_request_params(optional_params, model) + inference_params, additional_request_params, request_metadata = self._prepare_request_params( + optional_params, model ) original_tools = inference_params.pop("tools", []) @@ -1410,11 +1430,11 @@ class AmazonConverseConfig(BaseConfig): ) """ - Bedrock Response Object has optional message block + Bedrock Response Object has optional message block completion_response["output"].get("message", None) - A message block looks like this (Example 1): + A message block looks like this (Example 1): "output": { "message": { "role": "assistant", diff --git a/litellm/llms/bedrock/chat/invoke_handler.py b/litellm/llms/bedrock/chat/invoke_handler.py index 49292545208..ea612da52c6 100644 --- a/litellm/llms/bedrock/chat/invoke_handler.py +++ b/litellm/llms/bedrock/chat/invoke_handler.py @@ -374,6 +374,29 @@ class BedrockLLM(BaseAWSLLM): def __init__(self) -> None: super().__init__() + @staticmethod + def is_claude_messages_api_model(model: str) -> bool: + """ + Check if the model uses the Claude Messages API (Claude 3+). + + Handles: + - Regional prefixes: eu.anthropic.claude-*, us.anthropic.claude-* + - Claude 3 models: claude-3-haiku, claude-3-sonnet, claude-3-opus, claude-3-5-*, claude-3-7-* + - Claude 4 models: claude-opus-4, claude-sonnet-4, claude-haiku-4 + """ + # Normalize model string to lowercase for matching + model_lower = model.lower() + + # Claude 3+ indicators (all use Messages API) + messages_api_indicators = [ + "claude-3", # Claude 3.x models + "claude-opus-4", # Claude Opus 4 + "claude-sonnet-4", # Claude Sonnet 4 + "claude-haiku-4", # Claude Haiku 4 + ] + + return any(indicator in model_lower for indicator in messages_api_indicators) + def convert_messages_to_prompt( self, model, messages, provider, custom_prompt_dict ) -> Tuple[str, Optional[list]]: @@ -465,7 +488,7 @@ class BedrockLLM(BaseAWSLLM): completion_response["generations"][0]["finish_reason"] ) elif provider == "anthropic": - if model.startswith("anthropic.claude-3"): + if self.is_claude_messages_api_model(model): json_schemas: dict = {} _is_function_call = False ## Handle Tool Calling @@ -595,13 +618,12 @@ class BedrockLLM(BaseAWSLLM): outputText = choice["message"].get("content") elif "text" in choice: # fallback for completion format outputText = choice["text"] - # Set finish reason if "finish_reason" in choice: model_response.choices[0].finish_reason = map_finish_reason( choice["finish_reason"] ) - + # Set usage if available if "usage" in completion_response: usage = completion_response["usage"] @@ -842,7 +864,7 @@ class BedrockLLM(BaseAWSLLM): ] = True # cohere requires stream = True in inference params data = json.dumps({"prompt": prompt, **inference_params}) elif provider == "anthropic": - if model.startswith("anthropic.claude-3"): + if self.is_claude_messages_api_model(model): # Separate system prompt from rest of message system_prompt_idx: list[int] = [] system_messages: list[str] = [] @@ -940,13 +962,13 @@ class BedrockLLM(BaseAWSLLM): # Use AmazonBedrockOpenAIConfig for proper OpenAI transformation openai_config = AmazonBedrockOpenAIConfig() supported_params = openai_config.get_supported_openai_params(model=model) - + # Filter to only supported OpenAI params filtered_params = { - k: v for k, v in inference_params.items() + k: v for k, v in inference_params.items() if k in supported_params } - + # OpenAI uses messages format, not prompt data = json.dumps({"messages": messages, **filtered_params}) else: diff --git a/litellm/utils.py b/litellm/utils.py index fbbaa94f7a1..5011b45c0f7 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -621,7 +621,7 @@ def load_credentials_from_list(kwargs: dict): """ # Access CredentialAccessor via module to trigger lazy loading if needed CredentialAccessor = getattr(sys.modules[__name__], 'CredentialAccessor') - + credential_name = kwargs.get("litellm_credential_name") if credential_name and litellm.credential_list: credential_accessor = CredentialAccessor.get_credential_values(credential_name) @@ -648,7 +648,7 @@ def _is_gemini_model(model: Optional[str], custom_llm_provider: Optional[str]) - if custom_llm_provider in ["vertex_ai", "vertex_ai_beta"]: return model is not None and "gemini" in model.lower() return True - + # Check if model name contains gemini return model is not None and "gemini" in model.lower() @@ -670,7 +670,7 @@ def _process_assistant_message_tool_calls( """ role = msg_copy.get("role") tool_calls = msg_copy.get("tool_calls") - + if role == "assistant" and isinstance(tool_calls, list): new_tool_calls = [] for tc in tool_calls: @@ -683,17 +683,17 @@ def _process_assistant_message_tool_calls( else: new_tool_calls.append(tc) continue - + # Remove thought signature from ID if present if isinstance(tc_dict.get("id"), str): if thought_signature_separator in tc_dict["id"]: tc_dict["id"] = _remove_thought_signature_from_id( tc_dict["id"], thought_signature_separator ) - + new_tool_calls.append(tc_dict) msg_copy["tool_calls"] = new_tool_calls - + return msg_copy @@ -708,7 +708,7 @@ def _process_tool_message_id(msg_copy: dict, thought_signature_separator: str) - msg_copy["tool_call_id"] = _remove_thought_signature_from_id( msg_copy["tool_call_id"], thought_signature_separator ) - + return msg_copy @@ -719,7 +719,7 @@ def _remove_thought_signatures_from_messages( Remove thought signatures from tool call IDs in all messages. """ processed_messages = [] - + for msg in messages: # Handle Pydantic models (convert to dict) if hasattr(msg, "model_dump"): @@ -730,17 +730,17 @@ def _remove_thought_signatures_from_messages( # Unknown type, keep as is processed_messages.append(msg) continue - + # Process assistant messages with tool_calls msg_dict = _process_assistant_message_tool_calls( msg_dict, thought_signature_separator ) - + # Process tool messages with tool_call_id msg_dict = _process_tool_message_id(msg_dict, thought_signature_separator) - + processed_messages.append(msg_dict) - + return processed_messages @@ -960,7 +960,7 @@ def function_setup( # noqa: PLR0915 input=buffer.getvalue(), model=model, ) - + ### REMOVE THOUGHT SIGNATURES FROM TOOL CALL IDS FOR NON-GEMINI MODELS ### # Gemini models embed thought signatures in tool call IDs. When sending # messages with tool calls to non-Gemini providers, we need to remove these @@ -976,7 +976,7 @@ def function_setup( # noqa: PLR0915 # Get custom_llm_provider to determine target provider custom_llm_provider = kwargs.get("custom_llm_provider") - + # If custom_llm_provider not in kwargs, try to determine it from the model if not custom_llm_provider and model: try: @@ -987,18 +987,18 @@ def function_setup( # noqa: PLR0915 except Exception: # If we can't determine the provider, skip this processing pass - + # Only process if target is NOT a Gemini model if not _is_gemini_model(model, custom_llm_provider): verbose_logger.debug( "Removing thought signatures from tool call IDs for non-Gemini model" ) - + # Process messages to remove thought signatures processed_messages = _remove_thought_signatures_from_messages( messages, THOUGHT_SIGNATURE_SEPARATOR ) - + # Update messages in kwargs or args if "messages" in kwargs: kwargs["messages"] = processed_messages @@ -2977,7 +2977,7 @@ def get_optional_params_embeddings( # noqa: PLR0915 ): # Lazy load get_supported_openai_params get_supported_openai_params = getattr(sys.modules[__name__], 'get_supported_openai_params') - + # retrieve all parameters passed to the function passed_params = locals() custom_llm_provider = passed_params.pop("custom_llm_provider", None) @@ -4063,7 +4063,14 @@ def get_optional_params( # noqa: PLR0915 ), ) elif "anthropic" in bedrock_base_model and bedrock_route == "invoke": - if bedrock_base_model.startswith("anthropic.claude-3"): + # Check for Claude 3+ models (Messages API) including regional prefixes and Claude 4 + # Models like eu.anthropic.claude-opus-4-5, us.anthropic.claude-3-5-sonnet, etc. + bedrock_base_model_lower = bedrock_base_model.lower() + is_messages_api_model = any( + indicator in bedrock_base_model_lower + for indicator in ["claude-3", "claude-opus-4", "claude-sonnet-4", "claude-haiku-4"] + ) + if is_messages_api_model: optional_params = ( litellm.AmazonAnthropicClaudeConfig().map_openai_params( non_default_params=non_default_params, @@ -6911,7 +6918,7 @@ def get_valid_models( # init litellm_params ################################# from litellm.types.router import LiteLLM_Params - + if litellm_params is None: litellm_params = LiteLLM_Params(model="") if api_key is not None: @@ -7513,7 +7520,7 @@ class ProviderConfigManager: return litellm.IBMWatsonXAIConfig() elif litellm.LlmProviders.EMPOWER == provider: return litellm.EmpowerChatConfig() - elif litellm.LlmProviders.MINIMAX == provider: + elif litellm.LlmProviders.MINIMAX == provider: return litellm.MinimaxChatConfig() elif litellm.LlmProviders.GITHUB == provider: return litellm.GithubChatConfig() @@ -8314,8 +8321,7 @@ class ProviderConfigManager: from litellm.llms.vertex_ai.ocr.common_utils import get_vertex_ai_ocr_config return get_vertex_ai_ocr_config(model=model) - - MistralOCRConfig = getattr(sys.modules[__name__], 'MistralOCRConfig') +MistralOCRConfig = getattr(sys.modules[__name__], 'MistralOCRConfig') PROVIDER_TO_CONFIG_MAP = { litellm.LlmProviders.MISTRAL: MistralOCRConfig, } @@ -8752,12 +8758,12 @@ def __getattr__(name: str) -> Any: """Lazy import handler for utils module with cached registry for improved performance.""" # Use cached registry from _lazy_imports instead of importing tuples every time from litellm._lazy_imports import _get_lazy_import_registry - + registry = _get_lazy_import_registry() - + # Check if name is in registry and call the cached handler function if name in registry: handler_func = registry[name] return handler_func(name) - + raise AttributeError(f"module {__name__!r} has no attribute {name!r}") diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index e603f94ab87..473847c04fd 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -275,10 +275,10 @@ def test_get_supported_openai_params(): def test_get_supported_openai_params_bedrock_converse(): """ - Test that all documented bedrock converse models have the same set of supported openai params when using + Test that all documented bedrock converse models have the same set of supported openai params when using `bedrock/converse/` or `bedrock/` prefix. - Note: This test is critical for routing, if we ever remove `litellm.BEDROCK_CONVERSE_MODELS`, + Note: This test is critical for routing, if we ever remove `litellm.BEDROCK_CONVERSE_MODELS`, please update this test to read `bedrock_converse` models from the model cost map. """ for model in litellm.BEDROCK_CONVERSE_MODELS: @@ -380,7 +380,7 @@ def test_transform_response_with_computer_use_tool(): @property def text(self): return json.dumps(response_json) - + config = AmazonConverseConfig() model_response = ModelResponse() optional_params = { @@ -471,7 +471,7 @@ def test_transform_response_with_bash_tool(): @property def text(self): return json.dumps(response_json) - + config = AmazonConverseConfig() model_response = ModelResponse() optional_params = { @@ -525,7 +525,7 @@ def test_transform_response_with_structured_response_being_called(): "toolUseId": "tooluse_456", "name": "json_tool_call", "input": { - "Current_Temperature": 62, + "Current_Temperature": 62, "Weather_Explanation": "San Francisco typically has mild, cool weather year-round due to its coastal location and marine influence. The city is known for its fog, moderate temperatures, and relatively stable climate with little seasonal variation."}, } } @@ -550,51 +550,51 @@ def test_transform_response_with_structured_response_being_called(): @property def text(self): return json.dumps(response_json) - + config = AmazonConverseConfig() model_response = ModelResponse() optional_params = { "json_mode": True, "tools": [ { - 'type': 'function', + 'type': 'function', 'function': { - 'name': 'get_weather', - 'description': 'Get the current weather in a given location', + 'name': 'get_weather', + 'description': 'Get the current weather in a given location', 'parameters': { - 'type': 'object', + 'type': 'object', 'properties': { 'location': { - 'type': 'string', + 'type': 'string', 'description': 'The city and state, e.g. San Francisco, CA' - }, + }, 'unit': { - 'type': 'string', + 'type': 'string', 'enum': ['celsius', 'fahrenheit'] } - }, + }, 'required': ['location'] } } - }, + }, { - 'type': 'function', + 'type': 'function', 'function': { - 'name': 'json_tool_call', + 'name': 'json_tool_call', 'parameters': { - '$schema': 'http://json-schema.org/draft-07/schema#', - 'type': 'object', - 'required': ['Weather_Explanation', 'Current_Temperature'], + '$schema': 'http://json-schema.org/draft-07/schema#', + 'type': 'object', + 'required': ['Weather_Explanation', 'Current_Temperature'], 'properties': { 'Weather_Explanation': { - 'type': ['string', 'null'], + 'type': ['string', 'null'], 'description': '1-2 sentences explaining the weather in the location' - }, + }, 'Current_Temperature': { - 'type': ['number', 'null'], + 'type': ['number', 'null'], 'description': 'Current temperature in the location' } - }, + }, 'additionalProperties': False } } @@ -629,36 +629,36 @@ def test_transform_response_with_structured_response_calling_tool(): response_json = { "metrics": { "latencyMs": 1148 - }, + }, "output": { - "message": + "message": { "content": [ { "text": "I\'ll check the current weather in San Francisco for you." - }, + }, { "toolUse": { "input": { "location": "San Francisco, CA", "unit": "celsius" - }, - "name": "get_weather", + }, + "name": "get_weather", "toolUseId": "tooluse_oKk__QrqSUmufMw3Q7vGaQ" } } - ], + ], "role": "assistant" } - }, - "stopReason": "tool_use", + }, + "stopReason": "tool_use", "usage": { - "cacheReadInputTokenCount": 0, - "cacheReadInputTokens": 0, - "cacheWriteInputTokenCount": 0, - "cacheWriteInputTokens": 0, - "inputTokens": 534, - "outputTokens": 69, + "cacheReadInputTokenCount": 0, + "cacheReadInputTokens": 0, + "cacheWriteInputTokenCount": 0, + "cacheWriteInputTokens": 0, + "inputTokens": 534, + "outputTokens": 69, "totalTokens": 603 } } @@ -669,51 +669,51 @@ def test_transform_response_with_structured_response_calling_tool(): @property def text(self): return json.dumps(response_json) - + config = AmazonConverseConfig() model_response = ModelResponse() optional_params = { "json_mode": True, "tools": [ { - 'type': 'function', + 'type': 'function', 'function': { - 'name': 'get_weather', - 'description': 'Get the current weather in a given location', + 'name': 'get_weather', + 'description': 'Get the current weather in a given location', 'parameters': { - 'type': 'object', + 'type': 'object', 'properties': { 'location': { - 'type': 'string', + 'type': 'string', 'description': 'The city and state, e.g. San Francisco, CA' - }, + }, 'unit': { - 'type': 'string', + 'type': 'string', 'enum': ['celsius', 'fahrenheit'] } - }, + }, 'required': ['location'] } } - }, + }, { - 'type': 'function', + 'type': 'function', 'function': { - 'name': 'json_tool_call', + 'name': 'json_tool_call', 'parameters': { - '$schema': 'http://json-schema.org/draft-07/schema#', - 'type': 'object', - 'required': ['Weather_Explanation', 'Current_Temperature'], + '$schema': 'http://json-schema.org/draft-07/schema#', + 'type': 'object', + 'required': ['Weather_Explanation', 'Current_Temperature'], 'properties': { 'Weather_Explanation': { - 'type': ['string', 'null'], + 'type': ['string', 'null'], 'description': '1-2 sentences explaining the weather in the location' - }, + }, 'Current_Temperature': { - 'type': ['number', 'null'], + 'type': ['number', 'null'], 'description': 'Current temperature in the location' } - }, + }, 'additionalProperties': False } } @@ -743,7 +743,7 @@ def test_transform_response_with_structured_response_calling_tool(): @pytest.mark.asyncio async def test_bedrock_bash_tool_acompletion(): """Test Bedrock with bash tool for ls command using acompletion.""" - + # Test with bash tool instead of computer tool tools = [ { @@ -751,14 +751,14 @@ async def test_bedrock_bash_tool_acompletion(): "name": "bash", } ] - + messages = [ { - "role": "user", + "role": "user", "content": "run ls command and find all python files" } ] - + try: response = await litellm.acompletion( model="bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0", @@ -771,13 +771,13 @@ async def test_bedrock_bash_tool_acompletion(): assert False, "Expected authentication error but got successful response" except Exception as e: error_str = str(e).lower() - + # Check if it's an expected authentication/credentials error auth_error_indicators = [ - "credentials", "authentication", "unauthorized", "access denied", + "credentials", "authentication", "unauthorized", "access denied", "aws", "region", "profile", "token", "invalid", "signature" ] - + if any(auth_error in error_str for auth_error in auth_error_indicators): # This is expected - request formatting succeeded, auth failed as expected assert True @@ -789,7 +789,7 @@ async def test_bedrock_bash_tool_acompletion(): @pytest.mark.asyncio async def test_bedrock_computer_use_acompletion(): """Test Bedrock computer use with acompletion function.""" - + # Test with computer use tool tools = [ { @@ -800,10 +800,10 @@ async def test_bedrock_computer_use_acompletion(): "display_number": 0, } ] - + messages = [ { - "role": "user", + "role": "user", "content": [ { "type": "text", @@ -818,7 +818,7 @@ async def test_bedrock_computer_use_acompletion(): ] } ] - + try: response = await litellm.acompletion( model="bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0", @@ -831,13 +831,13 @@ async def test_bedrock_computer_use_acompletion(): assert False, "Expected authentication error but got successful response" except Exception as e: error_str = str(e).lower() - + # Check if it's an expected authentication/credentials error auth_error_indicators = [ - "credentials", "authentication", "unauthorized", "access denied", + "credentials", "authentication", "unauthorized", "access denied", "aws", "region", "profile", "token", "invalid", "signature" ] - + if any(auth_error in error_str for auth_error in auth_error_indicators): # This is expected - request formatting succeeded, auth failed as expected assert True @@ -849,9 +849,9 @@ async def test_bedrock_computer_use_acompletion(): @pytest.mark.asyncio async def test_transformation_directly(): """Test the transformation directly to verify the request structure.""" - + config = AmazonConverseConfig() - + tools = [ { "type": "computer_20241022", @@ -865,14 +865,14 @@ async def test_transformation_directly(): "name": "bash", } ] - + messages = [ { "role": "user", "content": "run ls command and find all python files" } ] - + # Transform request request_data = config.transform_request( model="anthropic.claude-3-5-sonnet-20241022-v2:0", @@ -881,19 +881,19 @@ async def test_transformation_directly(): litellm_params={}, headers={} ) - + # Verify the structure assert "additionalModelRequestFields" in request_data additional_fields = request_data["additionalModelRequestFields"] - + # Check that anthropic_beta is set correctly for computer use assert "anthropic_beta" in additional_fields assert additional_fields["anthropic_beta"] == ["computer-use-2024-10-22"] - + # Check that tools are present assert "tools" in additional_fields assert len(additional_fields["tools"]) == 2 - + # Verify tool types tool_types = [tool.get("type") for tool in additional_fields["tools"]] assert "computer_20241022" in tool_types @@ -933,7 +933,7 @@ def test_transform_request_helper_includes_anthropic_beta_and_tools_bash(): def test_transform_request_with_multiple_tools(): """Test transformation with multiple tools including computer, bash, and function tools.""" config = AmazonConverseConfig() - + # Use the exact payload from the user's error tools = [ { @@ -974,14 +974,14 @@ def test_transform_request_with_multiple_tools(): } } ] - + messages = [ { "role": "user", "content": "run ls command and find all python files" } ] - + # Transform request request_data = config.transform_request( model="anthropic.claude-3-5-sonnet-20241022-v2:0", @@ -990,25 +990,25 @@ def test_transform_request_with_multiple_tools(): litellm_params={}, headers={} ) - + # Verify the structure assert "additionalModelRequestFields" in request_data additional_fields = request_data["additionalModelRequestFields"] - + # Check that anthropic_beta is set correctly for computer use assert "anthropic_beta" in additional_fields assert additional_fields["anthropic_beta"] == ["computer-use-2024-10-22"] - + # Check that tools are present assert "tools" in additional_fields assert len(additional_fields["tools"]) == 3 # computer, bash, text_editor tools - + # Verify tool types tool_types = [tool.get("type") for tool in additional_fields["tools"]] assert "computer_20241022" in tool_types assert "bash_20241022" in tool_types assert "text_editor_20241022" in tool_types - + # Function tools are processed separately and not included in computer use tools # They would be in toolConfig if present @@ -1016,7 +1016,7 @@ def test_transform_request_with_multiple_tools(): def test_transform_request_with_computer_tool_only(): """Test transformation with only computer tool.""" config = AmazonConverseConfig() - + tools = [ { "type": "computer_20241022", @@ -1026,10 +1026,10 @@ def test_transform_request_with_computer_tool_only(): "display_number": 0, } ] - + messages = [ { - "role": "user", + "role": "user", "content": [ { "type": "text", @@ -1044,7 +1044,7 @@ def test_transform_request_with_computer_tool_only(): ] } ] - + # Transform request request_data = config.transform_request( model="anthropic.claude-3-5-sonnet-20241022-v2:0", @@ -1053,15 +1053,15 @@ def test_transform_request_with_computer_tool_only(): litellm_params={}, headers={} ) - + # Verify the structure assert "additionalModelRequestFields" in request_data additional_fields = request_data["additionalModelRequestFields"] - + # Check that anthropic_beta is set correctly for computer use assert "anthropic_beta" in additional_fields assert additional_fields["anthropic_beta"] == ["computer-use-2024-10-22"] - + # Check that tools are present assert "tools" in additional_fields assert len(additional_fields["tools"]) == 1 @@ -1071,21 +1071,21 @@ def test_transform_request_with_computer_tool_only(): def test_transform_request_with_bash_tool_only(): """Test transformation with only bash tool.""" config = AmazonConverseConfig() - + tools = [ { "type": "bash_20241022", "name": "bash", } ] - + messages = [ { - "role": "user", + "role": "user", "content": "run ls command and find all python files" } ] - + # Transform request request_data = config.transform_request( model="anthropic.claude-3-5-sonnet-20241022-v2:0", @@ -1094,15 +1094,15 @@ def test_transform_request_with_bash_tool_only(): litellm_params={}, headers={} ) - + # Verify the structure assert "additionalModelRequestFields" in request_data additional_fields = request_data["additionalModelRequestFields"] - + # Check that anthropic_beta is set correctly for computer use assert "anthropic_beta" in additional_fields assert additional_fields["anthropic_beta"] == ["computer-use-2024-10-22"] - + # Check that tools are present assert "tools" in additional_fields assert len(additional_fields["tools"]) == 1 @@ -1112,21 +1112,21 @@ def test_transform_request_with_bash_tool_only(): def test_transform_request_with_text_editor_tool(): """Test transformation with text editor tool.""" config = AmazonConverseConfig() - + tools = [ { "type": "text_editor_20241022", "name": "str_replace_editor", } ] - + messages = [ { "role": "user", "content": "Edit this text file" } ] - + # Transform request request_data = config.transform_request( model="anthropic.claude-3-5-sonnet-20241022-v2:0", @@ -1135,15 +1135,15 @@ def test_transform_request_with_text_editor_tool(): litellm_params={}, headers={} ) - + # Verify the structure assert "additionalModelRequestFields" in request_data additional_fields = request_data["additionalModelRequestFields"] - + # Check that anthropic_beta is set correctly for computer use assert "anthropic_beta" in additional_fields assert additional_fields["anthropic_beta"] == ["computer-use-2024-10-22"] - + # Check that tools are present assert "tools" in additional_fields assert len(additional_fields["tools"]) == 1 @@ -1153,7 +1153,7 @@ def test_transform_request_with_text_editor_tool(): def test_transform_request_with_function_tool(): """Test transformation with function tool.""" config = AmazonConverseConfig() - + tools = [ { "type": "function", @@ -1174,14 +1174,14 @@ def test_transform_request_with_function_tool(): } } ] - + messages = [ { "role": "user", "content": "What's the weather like in San Francisco?" } ] - + # Transform request request_data = config.transform_request( model="anthropic.claude-3-5-sonnet-20241022-v2:0", @@ -1190,11 +1190,11 @@ def test_transform_request_with_function_tool(): litellm_params={}, headers={} ) - + # Verify the structure assert "additionalModelRequestFields" in request_data additional_fields = request_data["additionalModelRequestFields"] - + # Function tools are not computer use tools, so they don't get anthropic_beta # They are processed through the regular tool config assert "toolConfig" in request_data @@ -1206,7 +1206,7 @@ def test_transform_request_with_function_tool(): def test_map_openai_params_with_response_format(): """Test map_openai_params with response_format.""" config = AmazonConverseConfig() - + tools = [ { "type": "function", @@ -1277,12 +1277,12 @@ async def test_assistant_message_cache_control(): messages = [ {"role": "user", "content": "Hello"}, { - "role": "assistant", + "role": "assistant", "content": "Hi there!", "cache_control": {"type": "ephemeral"} } ] - + result = _bedrock_converse_messages_pt( messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", @@ -1294,7 +1294,7 @@ async def test_assistant_message_cache_control(): model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + assert result == async_result async_result = await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async( @@ -1302,14 +1302,14 @@ async def test_assistant_message_cache_control(): model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + assert result == async_result - + # Should have user message and assistant message assert len(result) == 2 assert result[0]["role"] == "user" assert result[1]["role"] == "assistant" - + # Assistant message should have text content and cachePoint assistant_content = result[1]["content"] assert len(assistant_content) == 2 @@ -1325,7 +1325,7 @@ async def test_assistant_message_list_content_cache_control(): BedrockConverseMessagesProcessor, _bedrock_converse_messages_pt, ) - + messages = [ {"role": "user", "content": "Hello"}, { @@ -1339,7 +1339,7 @@ async def test_assistant_message_list_content_cache_control(): ] } ] - + result = _bedrock_converse_messages_pt( messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", @@ -1351,9 +1351,9 @@ async def test_assistant_message_list_content_cache_control(): model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + assert result == async_result - + # Assistant message should have text content and cachePoint assistant_content = result[1]["content"] assert len(assistant_content) == 2 @@ -1369,7 +1369,7 @@ async def test_tool_message_cache_control(): BedrockConverseMessagesProcessor, _bedrock_converse_messages_pt, ) - + messages = [ {"role": "user", "content": "What's the weather?"}, { @@ -1395,7 +1395,7 @@ async def test_tool_message_cache_control(): ] } ] - + result = _bedrock_converse_messages_pt( messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", @@ -1407,20 +1407,20 @@ async def test_tool_message_cache_control(): model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + assert result == async_result - + # Should have user, assistant, and user (tool results) messages assert len(result) == 3 - + # Last message should contain tool result and cachePoint tool_message_content = result[2]["content"] assert len(tool_message_content) == 2 - + # First should be tool result assert "toolResult" in tool_message_content[0] assert tool_message_content[0]["toolResult"]["content"][0]["text"] == "Weather data: sunny, 25°C" - + # Second should be cachePoint assert "cachePoint" in tool_message_content[1] assert tool_message_content[1]["cachePoint"]["type"] == "default" @@ -1433,7 +1433,7 @@ async def test_tool_message_string_content_cache_control(): BedrockConverseMessagesProcessor, _bedrock_converse_messages_pt, ) - + messages = [ {"role": "user", "content": "What's the weather?"}, { @@ -1442,7 +1442,7 @@ async def test_tool_message_string_content_cache_control(): "tool_calls": [ { "id": "call_123", - "type": "function", + "type": "function", "function": {"name": "get_weather", "arguments": "{}"} } ] @@ -1454,7 +1454,7 @@ async def test_tool_message_string_content_cache_control(): "cache_control": {"type": "ephemeral"} } ] - + result = _bedrock_converse_messages_pt( messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", @@ -1466,17 +1466,17 @@ async def test_tool_message_string_content_cache_control(): model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + assert result == async_result - + # Last message should contain tool result and cachePoint tool_message_content = result[2]["content"] assert len(tool_message_content) == 2 - + # First should be tool result assert "toolResult" in tool_message_content[0] assert tool_message_content[0]["toolResult"]["content"][0]["text"] == "Weather: sunny, 25°C" - + # Second should be cachePoint assert "cachePoint" in tool_message_content[1] assert tool_message_content[1]["cachePoint"]["type"] == "default" @@ -1489,7 +1489,7 @@ async def test_assistant_tool_calls_cache_control(): BedrockConverseMessagesProcessor, _bedrock_converse_messages_pt, ) - + messages = [ {"role": "user", "content": "Calculate 2+2"}, { @@ -1505,7 +1505,7 @@ async def test_assistant_tool_calls_cache_control(): ] } ] - + result = _bedrock_converse_messages_pt( messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", @@ -1517,18 +1517,18 @@ async def test_assistant_tool_calls_cache_control(): model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + assert result == async_result - + # Assistant message should have tool use and cachePoint assistant_content = result[1]["content"] assert len(assistant_content) == 2 - + # First should be tool use assert "toolUse" in assistant_content[0] assert assistant_content[0]["toolUse"]["name"] == "calc" assert assistant_content[0]["toolUse"]["toolUseId"] == "call_proxy_123" - + # Second should be cachePoint assert "cachePoint" in assistant_content[1] assert assistant_content[1]["cachePoint"]["type"] == "default" @@ -1541,7 +1541,7 @@ async def test_multiple_tool_calls_with_mixed_cache_control(): BedrockConverseMessagesProcessor, _bedrock_converse_messages_pt, ) - + messages = [ {"role": "user", "content": "Do multiple calculations"}, { @@ -1563,7 +1563,7 @@ async def test_multiple_tool_calls_with_mixed_cache_control(): ] } ] - + result = _bedrock_converse_messages_pt( messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", @@ -1575,21 +1575,21 @@ async def test_multiple_tool_calls_with_mixed_cache_control(): model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + assert result == async_result - + # Assistant message should have: toolUse1, cachePoint, toolUse2 assistant_content = result[1]["content"] assert len(assistant_content) == 3 - + # First tool use with cache assert "toolUse" in assistant_content[0] assert assistant_content[0]["toolUse"]["toolUseId"] == "call_1" - + # Cache point for first tool assert "cachePoint" in assistant_content[1] assert assistant_content[1]["cachePoint"]["type"] == "default" - + # Second tool use without cache assert "toolUse" in assistant_content[2] assert assistant_content[2]["toolUse"]["toolUseId"] == "call_2" @@ -1602,7 +1602,7 @@ async def test_no_cache_control_no_cache_point(): BedrockConverseMessagesProcessor, _bedrock_converse_messages_pt, ) - + messages = [ {"role": "user", "content": "Hello"}, {"role": "assistant", "content": "Hi there!"}, # No cache_control @@ -1612,7 +1612,7 @@ async def test_no_cache_control_no_cache_point(): "content": "Tool result" # No cache_control } ] - + result = _bedrock_converse_messages_pt( messages=messages, model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", @@ -1624,14 +1624,14 @@ async def test_no_cache_control_no_cache_point(): model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + assert result == async_result - + # Assistant message should only have text content, no cachePoint assistant_content = result[1]["content"] assert len(assistant_content) == 1 assert assistant_content[0]["text"] == "Hi there!" - + # Tool message should only have tool result, no cachePoint tool_content = result[2]["content"] assert len(tool_content) == 1 @@ -1867,11 +1867,11 @@ def test_guarded_text_with_tool_calls(): # First should be regular text assert "text" in content[0] assert content[0]["text"] == "What's the weather?" - + # Second should be guardContent assert "guardContent" in content[1] assert content[1]["guardContent"]["text"]["text"] == "Please be careful with sensitive information" - + # Other messages should not have guardContent for i in range(1, 3): content = result[i]["content"] @@ -2115,7 +2115,7 @@ def test_auto_convert_in_full_transformation(): # Verify the transformation worked assert "messages" in result assert len(result["messages"]) == 1 - + # The message should have guardContent message = result["messages"][0] assert "content" in message @@ -2626,79 +2626,79 @@ def test_empty_assistant_message_handling(): {"role": "assistant", "content": ""}, # Empty content {"role": "user", "content": "How are you?"} ] - + # Enable modify_params to prevent consecutive user message merging original_modify_params = litellm.modify_params litellm.modify_params = True - + try: result = _bedrock_converse_messages_pt( messages=messages, model="anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + # Should have 3 messages: user, assistant (with placeholder), user assert len(result) == 3 assert result[0]["role"] == "user" assert result[1]["role"] == "assistant" assert result[2]["role"] == "user" - + # Assistant message should have placeholder text instead of empty content assert len(result[1]["content"]) == 1 assert result[1]["content"][0]["text"] == "Please continue." - + # Test case 2: Whitespace-only content messages = [ {"role": "user", "content": "Hello"}, {"role": "assistant", "content": " "}, # Whitespace-only content {"role": "user", "content": "How are you?"} ] - + result = _bedrock_converse_messages_pt( messages=messages, model="anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + # Assistant message should have placeholder text instead of whitespace assert len(result[1]["content"]) == 1 assert result[1]["content"][0]["text"] == "Please continue." - + # Test case 3: Empty list content messages = [ {"role": "user", "content": "Hello"}, {"role": "assistant", "content": [{"type": "text", "text": ""}]}, # Empty text in list {"role": "user", "content": "How are you?"} ] - + result = _bedrock_converse_messages_pt( messages=messages, model="anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + # Assistant message should have placeholder text instead of empty text assert len(result[1]["content"]) == 1 assert result[1]["content"][0]["text"] == "Please continue." - + # Test case 4: Normal content should not be affected messages = [ {"role": "user", "content": "Hello"}, {"role": "assistant", "content": "I'm doing well, thank you!"}, # Normal content {"role": "user", "content": "How are you?"} ] - + result = _bedrock_converse_messages_pt( messages=messages, model="anthropic.claude-3-5-sonnet-20240620-v1:0", llm_provider="bedrock_converse" ) - + # Assistant message should keep original content assert len(result[1]["content"]) == 1 assert result[1]["content"][0]["text"] == "I'm doing well, thank you!" - + finally: # Restore original modify_params setting litellm.modify_params = original_modify_params @@ -2707,31 +2707,161 @@ def test_empty_assistant_message_handling(): def test_is_nova_lite_2_model(): """Test the _is_nova_lite_2_model() method for detecting Nova 2 models.""" config = AmazonConverseConfig() - + # Test with amazon.nova-2-lite-v1:0 assert config._is_nova_lite_2_model("amazon.nova-2-lite-v1:0") is True - + # Test with regional variants assert config._is_nova_lite_2_model("us.amazon.nova-2-lite-v1:0") is True assert config._is_nova_lite_2_model("eu.amazon.nova-2-lite-v1:0") is True assert config._is_nova_lite_2_model("apac.amazon.nova-2-lite-v1:0") is True - + # Test with other Nova 2 variants (pro, micro) assert config._is_nova_lite_2_model("amazon.nova-pro-1-5-v1:0") is False assert config._is_nova_lite_2_model("amazon.nova-micro-1-5-v1:0") is False assert config._is_nova_lite_2_model("us.amazon.nova-pro-1-5-v1:0") is False assert config._is_nova_lite_2_model("eu.amazon.nova-micro-1-5-v1:0") is False - + # Test with non-Nova-1.5 lite models (should return False) assert config._is_nova_lite_2_model("amazon.nova-lite-v1:0") is False assert config._is_nova_lite_2_model("amazon.nova-pro-v1:0") is False assert config._is_nova_lite_2_model("amazon.nova-micro-v1:0") is False - + # Test with Nova v1:0 models (should return False) assert config._is_nova_lite_2_model("us.amazon.nova-lite-v1:0") is False assert config._is_nova_lite_2_model("eu.amazon.nova-pro-v1:0") is False - + # Test with completely different models (should return False) assert config._is_nova_lite_2_model("anthropic.claude-3-5-sonnet-20240620-v1:0") is False assert config._is_nova_lite_2_model("meta.llama3-70b-instruct-v1:0") is False assert config._is_nova_lite_2_model("mistral.mistral-7b-instruct-v0:2") is False + +def test_drop_thinking_param_when_thinking_blocks_missing(): + """ + Test that thinking param is dropped when modify_params=True and + thinking_blocks are missing from assistant message with tool_calls. + + This prevents the Anthropic/Bedrock error: + "Expected thinking or redacted_thinking, but found tool_use" + + Related issue: https://github.com/BerriAI/litellm/issues/14194 + """ + from litellm.utils import last_assistant_with_tool_calls_has_no_thinking_blocks + + # Save original modify_params setting + original_modify_params = litellm.modify_params + + try: + # Test case 1: thinking should be dropped when modify_params=True + # and assistant message has tool_calls but no thinking_blocks + litellm.modify_params = True + + messages_without_thinking_blocks = [ + {"role": "user", "content": "Search for weather"}, + { + "role": "assistant", + "content": "", + "tool_calls": [ + { + "id": "call_123", + "type": "function", + "function": {"name": "search", "arguments": "{}"}, + } + ], + # No thinking_blocks - simulates OpenAI-compatible client + }, + {"role": "tool", "content": "Weather is sunny", "tool_call_id": "call_123"}, + ] + + optional_params = {"thinking": {"type": "enabled", "budget_tokens": 1000}} + + # Verify the condition is detected + assert last_assistant_with_tool_calls_has_no_thinking_blocks( + messages_without_thinking_blocks + ), "Should detect missing thinking_blocks" + + # Simulate what _transform_request_helper does + if ( + optional_params.get("thinking") is not None + and messages_without_thinking_blocks is not None + and last_assistant_with_tool_calls_has_no_thinking_blocks( + messages_without_thinking_blocks + ) + ): + if litellm.modify_params: + optional_params.pop("thinking", None) + + assert "thinking" not in optional_params, ( + "thinking param should be dropped when modify_params=True " + "and thinking_blocks are missing" + ) + + # Test case 2: thinking should NOT be dropped when thinking_blocks are present + messages_with_thinking_blocks = [ + {"role": "user", "content": "Search for weather"}, + { + "role": "assistant", + "content": "", + "tool_calls": [ + { + "id": "call_123", + "type": "function", + "function": {"name": "search", "arguments": "{}"}, + } + ], + "thinking_blocks": [ + {"type": "thinking", "thinking": "Let me search for weather..."} + ], + }, + {"role": "tool", "content": "Weather is sunny", "tool_call_id": "call_123"}, + ] + + optional_params_with_thinking = { + "thinking": {"type": "enabled", "budget_tokens": 1000} + } + + # Verify the condition is NOT detected when thinking_blocks are present + assert not last_assistant_with_tool_calls_has_no_thinking_blocks( + messages_with_thinking_blocks + ), "Should NOT detect missing thinking_blocks when they are present" + + # Simulate what _transform_request_helper does + if ( + optional_params_with_thinking.get("thinking") is not None + and messages_with_thinking_blocks is not None + and last_assistant_with_tool_calls_has_no_thinking_blocks( + messages_with_thinking_blocks + ) + ): + if litellm.modify_params: + optional_params_with_thinking.pop("thinking", None) + + assert "thinking" in optional_params_with_thinking, ( + "thinking param should NOT be dropped when thinking_blocks are present" + ) + + # Test case 3: thinking should NOT be dropped when modify_params=False + litellm.modify_params = False + + optional_params_no_modify = { + "thinking": {"type": "enabled", "budget_tokens": 1000} + } + + # Simulate what _transform_request_helper does + if ( + optional_params_no_modify.get("thinking") is not None + and messages_without_thinking_blocks is not None + and last_assistant_with_tool_calls_has_no_thinking_blocks( + messages_without_thinking_blocks + ) + ): + if litellm.modify_params: + optional_params_no_modify.pop("thinking", None) + + assert "thinking" in optional_params_no_modify, ( + "thinking param should NOT be dropped when modify_params=False" + ) + + finally: + # Restore original modify_params setting + litellm.modify_params = original_modify_params From d3fdac84682e4d972d3c170925a5225224886135 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Benedikt=20=C3=93skarsson?= Date: Thu, 8 Jan 2026 02:04:52 +0000 Subject: [PATCH 2/3] chore: fix formatting error --- litellm/utils.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/litellm/utils.py b/litellm/utils.py index 5011b45c0f7..e088b35037a 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -8321,7 +8321,8 @@ class ProviderConfigManager: from litellm.llms.vertex_ai.ocr.common_utils import get_vertex_ai_ocr_config return get_vertex_ai_ocr_config(model=model) -MistralOCRConfig = getattr(sys.modules[__name__], 'MistralOCRConfig') + + MistralOCRConfig = getattr(sys.modules[__name__], 'MistralOCRConfig') PROVIDER_TO_CONFIG_MAP = { litellm.LlmProviders.MISTRAL: MistralOCRConfig, } From 0367e9c9f11b7f8f68166a7d1bb68c56523a8776 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Benedikt=20=C3=93skarsson?= Date: Mon, 19 Jan 2026 14:50:57 +0000 Subject: [PATCH 3/3] fix: pr ammends. --- litellm/llms/bedrock/chat/converse_transformation.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index fd8c0d75473..cb26a22edfa 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -55,6 +55,7 @@ from litellm.types.utils import ( ) from litellm.utils import ( add_dummy_tool, + any_assistant_message_has_thinking_blocks, has_tool_call_blocks, last_assistant_with_tool_calls_has_no_thinking_blocks, supports_reasoning, @@ -1077,11 +1078,15 @@ class AmazonConverseConfig(BaseConfig): # Drop thinking param if thinking is enabled but thinking_blocks are missing # This prevents the error: "Expected thinking or redacted_thinking, but found tool_use" + # + # IMPORTANT: Only drop thinking if NO assistant messages have thinking_blocks. + # If any message has thinking_blocks, we must keep thinking enabled, otherwise # Related issues: https://github.com/BerriAI/litellm/issues/14194 if ( optional_params.get("thinking") is not None and messages is not None and last_assistant_with_tool_calls_has_no_thinking_blocks(messages) + and not any_assistant_message_has_thinking_blocks(messages) ): if litellm.modify_params: optional_params.pop("thinking", None)