Merge pull request #18787 from aproorg/fix/bedrock-thinking-tool-call-2

fix(bedrock): handle thinking with tool calls for Claude 4 models
This commit is contained in:
Sameer Kankute 2026-01-20 09:43:07 +05:30 • committed by GitHub
commit 2ae308028d
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 386 additions and 202 deletions

View file

@ -53,7 +53,13 @@ from litellm.types.utils import (
PromptTokensDetailsWrapper,
Usage,
)
from litellm.utils import add_dummy_tool, has_tool_call_blocks, supports_reasoning
from litellm.utils import (
add_dummy_tool,
any_assistant_message_has_thinking_blocks,
has_tool_call_blocks,
last_assistant_with_tool_calls_has_no_thinking_blocks,
supports_reasoning,
)
from ..common_utils import (
BedrockError,
@ -773,7 +779,7 @@ class AmazonConverseConfig(BaseConfig):
return optional_params
"""
Follow similar approach to anthropic - translate to a single tool call.
Follow similar approach to anthropic - translate to a single tool call.
When using tools in this way: - https://docs.anthropic.com/en/docs/build-with-claude/tool-use#json-mode
- You usually want to provide a single tool
@ -1070,9 +1076,28 @@ class AmazonConverseConfig(BaseConfig):
llm_provider="bedrock",
)
# Drop thinking param if thinking is enabled but thinking_blocks are missing
# This prevents the error: "Expected thinking or redacted_thinking, but found tool_use"
#
# IMPORTANT: Only drop thinking if NO assistant messages have thinking_blocks.
# If any message has thinking_blocks, we must keep thinking enabled, otherwise
# Related issues: https://github.com/BerriAI/litellm/issues/14194
if (
optional_params.get("thinking") is not None
and messages is not None
and last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
and not any_assistant_message_has_thinking_blocks(messages)
):
if litellm.modify_params:
optional_params.pop("thinking", None)
litellm.verbose_logger.warning(
"Dropping 'thinking' param because the last assistant message with tool_calls "
"has no thinking_blocks. The model won't use extended thinking for this turn."
)
# Prepare and separate parameters
inference_params, additional_request_params, request_metadata = (
self._prepare_request_params(optional_params, model)
inference_params, additional_request_params, request_metadata = self._prepare_request_params(
optional_params, model
)
original_tools = inference_params.pop("tools", [])
@ -1459,11 +1484,11 @@ class AmazonConverseConfig(BaseConfig):
)
"""
Bedrock Response Object has optional message block
Bedrock Response Object has optional message block
completion_response["output"].get("message", None)
A message block looks like this (Example 1):
A message block looks like this (Example 1):
"output": {
"message": {
"role": "assistant",

View file

@ -374,6 +374,29 @@ class BedrockLLM(BaseAWSLLM):
def __init__(self) -> None:
super().__init__()
@staticmethod
def is_claude_messages_api_model(model: str) -> bool:
"""
Check if the model uses the Claude Messages API (Claude 3+).
Handles:
- Regional prefixes: eu.anthropic.claude-*, us.anthropic.claude-*
- Claude 3 models: claude-3-haiku, claude-3-sonnet, claude-3-opus, claude-3-5-*, claude-3-7-*
- Claude 4 models: claude-opus-4, claude-sonnet-4, claude-haiku-4
"""
# Normalize model string to lowercase for matching
model_lower = model.lower()
# Claude 3+ indicators (all use Messages API)
messages_api_indicators = [
"claude-3", # Claude 3.x models
"claude-opus-4", # Claude Opus 4
"claude-sonnet-4", # Claude Sonnet 4
"claude-haiku-4", # Claude Haiku 4
]
return any(indicator in model_lower for indicator in messages_api_indicators)
def convert_messages_to_prompt(
self, model, messages, provider, custom_prompt_dict
) -> Tuple[str, Optional[list]]:
@ -465,7 +488,7 @@ class BedrockLLM(BaseAWSLLM):
completion_response["generations"][0]["finish_reason"]
)
elif provider == "anthropic":
if model.startswith("anthropic.claude-3"):
if self.is_claude_messages_api_model(model):
json_schemas: dict = {}
_is_function_call = False
## Handle Tool Calling
@ -595,13 +618,12 @@ class BedrockLLM(BaseAWSLLM):
outputText = choice["message"].get("content")
elif "text" in choice: # fallback for completion format
outputText = choice["text"]
# Set finish reason
if "finish_reason" in choice:
model_response.choices[0].finish_reason = map_finish_reason(
choice["finish_reason"]
)
# Set usage if available
if "usage" in completion_response:
usage = completion_response["usage"]
@ -838,7 +860,7 @@ class BedrockLLM(BaseAWSLLM):
] = True # cohere requires stream = True in inference params
data = json.dumps({"prompt": prompt, **inference_params})
elif provider == "anthropic":
if model.startswith("anthropic.claude-3"):
if self.is_claude_messages_api_model(model):
# Separate system prompt from rest of message
system_prompt_idx: list[int] = []
system_messages: list[str] = []
@ -936,13 +958,13 @@ class BedrockLLM(BaseAWSLLM):
# Use AmazonBedrockOpenAIConfig for proper OpenAI transformation
openai_config = AmazonBedrockOpenAIConfig()
supported_params = openai_config.get_supported_openai_params(model=model)
# Filter to only supported OpenAI params
filtered_params = {
k: v for k, v in inference_params.items()
k: v for k, v in inference_params.items()
if k in supported_params
}
# OpenAI uses messages format, not prompt
data = json.dumps({"messages": messages, **filtered_params})
else:

View file

@ -4186,7 +4186,14 @@ def get_optional_params( # noqa: PLR0915
),
)
elif "anthropic" in bedrock_base_model and bedrock_route == "invoke":
if bedrock_base_model.startswith("anthropic.claude-3"):
# Check for Claude 3+ models (Messages API) including regional prefixes and Claude 4
# Models like eu.anthropic.claude-opus-4-5, us.anthropic.claude-3-5-sonnet, etc.
bedrock_base_model_lower = bedrock_base_model.lower()
is_messages_api_model = any(
indicator in bedrock_base_model_lower
for indicator in ["claude-3", "claude-opus-4", "claude-sonnet-4", "claude-haiku-4"]
)
if is_messages_api_model:
optional_params = (
litellm.AmazonAnthropicClaudeConfig().map_openai_params(
non_default_params=non_default_params,