[Fix] Claude Code + Bedrock Converse Usage - ensure budget tokens are passed to converse api correctly (#19107)

* test_bedrock_converse_budget_tokens_preserved

* test_openai_model_with_thinking_converts_to_reasoning_effort

* fix translate_anthropic_thinking_to_reasoning_effort

* test_bedrock_converse_budget_tokens_preserved

* test_anthropic_messages_bedrock_converse_with_thinking
This commit is contained in:
Ishaan Jaff 2026-01-14 12:02:27 -08:00 • committed by GitHub
parent f0e34a46a7
commit 747829dadb
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 244 additions and 19 deletions

View file

@ -17,7 +17,6 @@ from openai.types.chat.chat_completion_chunk import Choice as OpenAIStreamingCho
from litellm.litellm_core_utils.prompt_templates.common_utils import (
parse_tool_call_arguments,
)
from litellm.types.llms.anthropic import (
AllAnthropicToolsValues,
AnthopicMessagesAssistantMessageParam,
@ -210,7 +209,7 @@ class LiteLLMAnthropicMessagesAdapter:
# Convert Anthropic image format to OpenAI format
source = content.get("source", {})
openai_image_url = (
self._translate_anthropic_image_to_openai(source)
self._translate_anthropic_image_to_openai(cast(dict, source))
)
if openai_image_url:
@ -240,7 +239,7 @@ class LiteLLMAnthropicMessagesAdapter:
# Combine all content items into a single tool message
# to avoid creating multiple tool_result blocks with the same ID
# (each tool_use must have exactly one tool_result)
content_items = content.get("content", [])
content_items = list(content.get("content", []))
# For single-item content, maintain backward compatibility with string/url format
if len(content_items) == 1:
@ -266,7 +265,7 @@ class LiteLLMAnthropicMessagesAdapter:
source = c.get("source", {})
openai_image_url = (
self._translate_anthropic_image_to_openai(
source
cast(dict, source)
)
or ""
)
@ -306,7 +305,7 @@ class LiteLLMAnthropicMessagesAdapter:
source = c.get("source", {})
openai_image_url = (
self._translate_anthropic_image_to_openai(
source
cast(dict, source)
)
or ""
)
@ -363,7 +362,7 @@ class LiteLLMAnthropicMessagesAdapter:
}
signature = (
self._extract_signature_from_tool_use_content(
content
cast(Dict[str, Any], content)
)
)
@ -424,14 +423,21 @@ class LiteLLMAnthropicMessagesAdapter:
return new_messages
def translate_anthropic_thinking_to_openai(
self, thinking: Dict[str, Any]
@staticmethod
def translate_anthropic_thinking_to_reasoning_effort(
thinking: Dict[str, Any]
) -> Optional[str]:
"""
Translate Anthropic's thinking parameter to OpenAI's reasoning_effort.
Anthropic thinking format: {'type': 'enabled'|'disabled', 'budget_tokens': int}
OpenAI reasoning_effort: 'none' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'default'
Mapping:
- budget_tokens >= 10000 -> 'high'
- budget_tokens >= 5000 -> 'medium'
- budget_tokens >= 2000 -> 'low'
- budget_tokens < 2000 -> 'minimal'
"""
if not isinstance(thinking, dict):
return None
@ -453,6 +459,53 @@ class LiteLLMAnthropicMessagesAdapter:
return None
@staticmethod
def is_anthropic_claude_model(model: str) -> bool:
"""
Check if the model is an Anthropic Claude model that supports the thinking parameter.
Returns True for:
- anthropic/* models
- bedrock/*anthropic* models (including converse)
- vertex_ai/*claude* models
"""
model_lower = model.lower()
return (
"anthropic" in model_lower
or "claude" in model_lower
)
@staticmethod
def translate_thinking_for_model(
thinking: Dict[str, Any],
model: str,
) -> Dict[str, Any]:
"""
Translate Anthropic thinking parameter based on the target model.
For Claude/Anthropic models: returns {'thinking': <original_thinking>}
- Preserves exact budget_tokens value
For non-Claude models: returns {'reasoning_effort': <mapped_value>}
- Converts thinking to reasoning_effort to avoid UnsupportedParamsError
Args:
thinking: Anthropic thinking dict with 'type' and 'budget_tokens'
model: The target model name
Returns:
Dict with either 'thinking' or 'reasoning_effort' key
"""
if LiteLLMAnthropicMessagesAdapter.is_anthropic_claude_model(model):
return {"thinking": thinking}
else:
reasoning_effort = LiteLLMAnthropicMessagesAdapter.translate_anthropic_thinking_to_reasoning_effort(
thinking
)
if reasoning_effort:
return {"reasoning_effort": reasoning_effort}
return {}
def translate_anthropic_tool_choice_to_openai(
self, tool_choice: AnthropicMessagesToolChoice
) -> ChatCompletionToolChoiceValues:
@ -566,11 +619,15 @@ class LiteLLMAnthropicMessagesAdapter:
if "thinking" in anthropic_message_request:
thinking = anthropic_message_request["thinking"]
if thinking:
reasoning_effort = self.translate_anthropic_thinking_to_openai(
thinking=cast(Dict[str, Any], thinking)
)
if reasoning_effort:
new_kwargs["reasoning_effort"] = reasoning_effort
model = new_kwargs.get("model", "")
if self.is_anthropic_claude_model(model):
new_kwargs["thinking"] = thinking # type: ignore
else:
reasoning_effort = self.translate_anthropic_thinking_to_reasoning_effort(
cast(Dict[str, Any], thinking)
)
if reasoning_effort:
new_kwargs["reasoning_effort"] = reasoning_effort
translatable_params = self.translatable_anthropic_params()
for k, v in anthropic_message_request.items():

View file

@ -1,10 +1,10 @@
model_list:
- model_name: anthropic/*
- model_name: us.anthropic.claude-sonnet-4-20250514-v1:0
litellm_params:
model: anthropic/*
- model_name: openai/*
litellm_params:
model: openai/*
model: bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0
model_info:
litellm_provider: bedrock_converse
mode: chat
general_settings:
store_prompts_in_spend_logs: true

View file

@ -66,3 +66,35 @@ async def test_anthropic_messages_litellm_router_bedrock():
INSTANCE_BASE_ANTHROPIC_MESSAGES_TEST._validate_response(response)
@pytest.mark.asyncio
async def test_anthropic_messages_bedrock_converse_with_thinking():
"""
Test that bedrock/converse model works with thinking parameter.
Validates the request body from issue where budget_tokens was being lost.
"""
router = Router(
model_list=[
{
"model_name": "bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0",
"litellm_params": {
"model": "bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0",
},
},
]
)
messages = [{"role": "user", "content": "What is 2+2?"}]
response = await router.aanthropic_messages(
messages=messages,
model="bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0",
max_tokens=1026,
thinking={
"type": "enabled",
"budget_tokens": 1025
},
)
print("bedrock response: ", response)
# Verify response
INSTANCE_BASE_ANTHROPIC_MESSAGES_TEST._validate_response(response)

View file

@ -1,3 +1,4 @@
import json
import os
import sys
@ -6,8 +7,10 @@ from fastapi.testclient import TestClient
sys.path.insert(0, os.path.abspath("../../../../.."))
from unittest.mock import MagicMock, patch
from unittest.mock import AsyncMock, MagicMock, patch
from litellm.anthropic_interface import messages
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.types.utils import Delta, ModelResponse, StreamingChoices
@ -87,3 +90,136 @@ def test_anthropic_experimental_pass_through_messages_handler_custom_llm_provide
assert call_kwargs["custom_llm_provider"] == "my-custom-llm"
assert call_kwargs["model"] == "my-custom-llm/my-custom-model"
assert call_kwargs["api_key"] == "test-api-key"
@pytest.mark.asyncio
async def test_bedrock_converse_budget_tokens_preserved():
"""
Test that budget_tokens value in thinking parameter is correctly passed to Bedrock Converse API
when using messages.acreate with bedrock/converse model.
The bug was that the messages -> completion adapter was converting thinking to reasoning_effort
and losing the original budget_tokens value, causing it to use the default (128) instead.
"""
client = AsyncHTTPHandler()
with patch.object(client, "post") as mock_post:
mock_response = AsyncMock()
mock_response.status_code = 200
mock_response.headers = {}
mock_response.text = "mock response"
mock_response.json.return_value = {
"output": {
"message": {
"role": "assistant",
"content": [{"text": "4"}]
}
},
"stopReason": "end_turn",
"usage": {
"inputTokens": 10,
"outputTokens": 5,
"totalTokens": 15
}
}
mock_post.return_value = mock_response
try:
await messages.acreate(
client=client,
max_tokens=1024,
messages=[{"role": "user", "content": "What is 2+2?"}],
model="bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0",
thinking={
"budget_tokens": 1024,
"type": "enabled"
},
)
except Exception:
pass # Expected due to mock response format
mock_post.assert_called_once()
call_kwargs = mock_post.call_args.kwargs
json_data = call_kwargs.get("json") or json.loads(call_kwargs.get("data", "{}"))
print("Request json: ", json.dumps(json_data, indent=4, default=str))
additional_fields = json_data.get("additionalModelRequestFields", {})
thinking_config = additional_fields.get("thinking", {})
assert "thinking" in additional_fields, "thinking parameter should be in additionalModelRequestFields"
assert thinking_config.get("type") == "enabled", "thinking.type should be 'enabled'"
assert thinking_config.get("budget_tokens") == 1024, f"thinking.budget_tokens should be 1024, but got {thinking_config.get('budget_tokens')}"
def test_openai_model_with_thinking_converts_to_reasoning_effort():
"""
Test that when using a non-Anthropic model (like OpenAI gpt-5.2) with thinking parameter,
the thinking is converted to reasoning_effort and NOT passed as thinking.
This ensures we don't regress on issue #16052 where non-Anthropic models would fail
with UnsupportedParamsError when thinking was passed directly.
"""
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
anthropic_messages_handler,
)
with patch("litellm.completion", return_value="test-response") as mock_completion:
try:
anthropic_messages_handler(
max_tokens=1024,
messages=[{"role": "user", "content": "What is 2+2?"}],
model="openai/gpt-5.2",
api_key="test-api-key",
thinking={
"type": "enabled",
"budget_tokens": 1024
},
)
except Exception as e:
print(f"Error: {e}")
mock_completion.assert_called_once()
call_kwargs = mock_completion.call_args.kwargs
# Verify reasoning_effort is set (converted from thinking)
assert "reasoning_effort" in call_kwargs, "reasoning_effort should be passed to completion"
assert call_kwargs["reasoning_effort"] == "minimal", f"reasoning_effort should be 'minimal' for budget_tokens=1024, got {call_kwargs.get('reasoning_effort')}"
# Verify thinking is NOT passed (non-Claude model)
assert "thinking" not in call_kwargs, "thinking should NOT be passed for non-Claude models"
class TestThinkingParameterTransformation:
"""Core tests for thinking parameter transformation logic."""
def test_claude_model_preserves_thinking_with_budget_tokens(self):
"""Test that Claude models get thinking parameter passed through with exact budget_tokens."""
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
LiteLLMAnthropicMessagesAdapter,
)
thinking = {"type": "enabled", "budget_tokens": 5000}
result = LiteLLMAnthropicMessagesAdapter.translate_thinking_for_model(
thinking=thinking,
model="bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0",
)
assert result == {"thinking": thinking}
assert result["thinking"]["budget_tokens"] == 5000
def test_non_claude_model_converts_thinking_to_reasoning_effort(self):
"""Test that non-Claude models convert thinking to reasoning_effort."""
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
LiteLLMAnthropicMessagesAdapter,
)
thinking = {"type": "enabled", "budget_tokens": 1024}
result = LiteLLMAnthropicMessagesAdapter.translate_thinking_for_model(
thinking=thinking,
model="openai/gpt-5.2",
)
assert result == {"reasoning_effort": "minimal"}
assert "thinking" not in result