mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
[Fix] Claude Code + Bedrock Converse Usage - ensure budget tokens are passed to converse api correctly (#19107)
* test_bedrock_converse_budget_tokens_preserved * test_openai_model_with_thinking_converts_to_reasoning_effort * fix translate_anthropic_thinking_to_reasoning_effort * test_bedrock_converse_budget_tokens_preserved * test_anthropic_messages_bedrock_converse_with_thinking
This commit is contained in:
parent
f0e34a46a7
commit
747829dadb
4 changed files with 244 additions and 19 deletions
|
|
@ -17,7 +17,6 @@ from openai.types.chat.chat_completion_chunk import Choice as OpenAIStreamingCho
|
|||
from litellm.litellm_core_utils.prompt_templates.common_utils import (
|
||||
parse_tool_call_arguments,
|
||||
)
|
||||
|
||||
from litellm.types.llms.anthropic import (
|
||||
AllAnthropicToolsValues,
|
||||
AnthopicMessagesAssistantMessageParam,
|
||||
|
|
@ -210,7 +209,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
# Convert Anthropic image format to OpenAI format
|
||||
source = content.get("source", {})
|
||||
openai_image_url = (
|
||||
self._translate_anthropic_image_to_openai(source)
|
||||
self._translate_anthropic_image_to_openai(cast(dict, source))
|
||||
)
|
||||
|
||||
if openai_image_url:
|
||||
|
|
@ -240,7 +239,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
# Combine all content items into a single tool message
|
||||
# to avoid creating multiple tool_result blocks with the same ID
|
||||
# (each tool_use must have exactly one tool_result)
|
||||
content_items = content.get("content", [])
|
||||
content_items = list(content.get("content", []))
|
||||
|
||||
# For single-item content, maintain backward compatibility with string/url format
|
||||
if len(content_items) == 1:
|
||||
|
|
@ -266,7 +265,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
source = c.get("source", {})
|
||||
openai_image_url = (
|
||||
self._translate_anthropic_image_to_openai(
|
||||
source
|
||||
cast(dict, source)
|
||||
)
|
||||
or ""
|
||||
)
|
||||
|
|
@ -306,7 +305,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
source = c.get("source", {})
|
||||
openai_image_url = (
|
||||
self._translate_anthropic_image_to_openai(
|
||||
source
|
||||
cast(dict, source)
|
||||
)
|
||||
or ""
|
||||
)
|
||||
|
|
@ -363,7 +362,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
}
|
||||
signature = (
|
||||
self._extract_signature_from_tool_use_content(
|
||||
content
|
||||
cast(Dict[str, Any], content)
|
||||
)
|
||||
)
|
||||
|
||||
|
|
@ -424,14 +423,21 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
|
||||
return new_messages
|
||||
|
||||
def translate_anthropic_thinking_to_openai(
|
||||
self, thinking: Dict[str, Any]
|
||||
@staticmethod
|
||||
def translate_anthropic_thinking_to_reasoning_effort(
|
||||
thinking: Dict[str, Any]
|
||||
) -> Optional[str]:
|
||||
"""
|
||||
Translate Anthropic's thinking parameter to OpenAI's reasoning_effort.
|
||||
|
||||
Anthropic thinking format: {'type': 'enabled'|'disabled', 'budget_tokens': int}
|
||||
OpenAI reasoning_effort: 'none' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'default'
|
||||
|
||||
Mapping:
|
||||
- budget_tokens >= 10000 -> 'high'
|
||||
- budget_tokens >= 5000 -> 'medium'
|
||||
- budget_tokens >= 2000 -> 'low'
|
||||
- budget_tokens < 2000 -> 'minimal'
|
||||
"""
|
||||
if not isinstance(thinking, dict):
|
||||
return None
|
||||
|
|
@ -453,6 +459,53 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def is_anthropic_claude_model(model: str) -> bool:
|
||||
"""
|
||||
Check if the model is an Anthropic Claude model that supports the thinking parameter.
|
||||
|
||||
Returns True for:
|
||||
- anthropic/* models
|
||||
- bedrock/*anthropic* models (including converse)
|
||||
- vertex_ai/*claude* models
|
||||
"""
|
||||
model_lower = model.lower()
|
||||
return (
|
||||
"anthropic" in model_lower
|
||||
or "claude" in model_lower
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def translate_thinking_for_model(
|
||||
thinking: Dict[str, Any],
|
||||
model: str,
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Translate Anthropic thinking parameter based on the target model.
|
||||
|
||||
For Claude/Anthropic models: returns {'thinking': <original_thinking>}
|
||||
- Preserves exact budget_tokens value
|
||||
|
||||
For non-Claude models: returns {'reasoning_effort': <mapped_value>}
|
||||
- Converts thinking to reasoning_effort to avoid UnsupportedParamsError
|
||||
|
||||
Args:
|
||||
thinking: Anthropic thinking dict with 'type' and 'budget_tokens'
|
||||
model: The target model name
|
||||
|
||||
Returns:
|
||||
Dict with either 'thinking' or 'reasoning_effort' key
|
||||
"""
|
||||
if LiteLLMAnthropicMessagesAdapter.is_anthropic_claude_model(model):
|
||||
return {"thinking": thinking}
|
||||
else:
|
||||
reasoning_effort = LiteLLMAnthropicMessagesAdapter.translate_anthropic_thinking_to_reasoning_effort(
|
||||
thinking
|
||||
)
|
||||
if reasoning_effort:
|
||||
return {"reasoning_effort": reasoning_effort}
|
||||
return {}
|
||||
|
||||
def translate_anthropic_tool_choice_to_openai(
|
||||
self, tool_choice: AnthropicMessagesToolChoice
|
||||
) -> ChatCompletionToolChoiceValues:
|
||||
|
|
@ -566,11 +619,15 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
if "thinking" in anthropic_message_request:
|
||||
thinking = anthropic_message_request["thinking"]
|
||||
if thinking:
|
||||
reasoning_effort = self.translate_anthropic_thinking_to_openai(
|
||||
thinking=cast(Dict[str, Any], thinking)
|
||||
)
|
||||
if reasoning_effort:
|
||||
new_kwargs["reasoning_effort"] = reasoning_effort
|
||||
model = new_kwargs.get("model", "")
|
||||
if self.is_anthropic_claude_model(model):
|
||||
new_kwargs["thinking"] = thinking # type: ignore
|
||||
else:
|
||||
reasoning_effort = self.translate_anthropic_thinking_to_reasoning_effort(
|
||||
cast(Dict[str, Any], thinking)
|
||||
)
|
||||
if reasoning_effort:
|
||||
new_kwargs["reasoning_effort"] = reasoning_effort
|
||||
|
||||
translatable_params = self.translatable_anthropic_params()
|
||||
for k, v in anthropic_message_request.items():
|
||||
|
|
|
|||
|
|
@ -1,10 +1,10 @@
|
|||
model_list:
|
||||
- model_name: anthropic/*
|
||||
- model_name: us.anthropic.claude-sonnet-4-20250514-v1:0
|
||||
litellm_params:
|
||||
model: anthropic/*
|
||||
- model_name: openai/*
|
||||
litellm_params:
|
||||
model: openai/*
|
||||
model: bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0
|
||||
model_info:
|
||||
litellm_provider: bedrock_converse
|
||||
mode: chat
|
||||
|
||||
general_settings:
|
||||
store_prompts_in_spend_logs: true
|
||||
|
|
@ -66,3 +66,35 @@ async def test_anthropic_messages_litellm_router_bedrock():
|
|||
INSTANCE_BASE_ANTHROPIC_MESSAGES_TEST._validate_response(response)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_anthropic_messages_bedrock_converse_with_thinking():
|
||||
"""
|
||||
Test that bedrock/converse model works with thinking parameter.
|
||||
Validates the request body from issue where budget_tokens was being lost.
|
||||
"""
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0",
|
||||
"litellm_params": {
|
||||
"model": "bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0",
|
||||
},
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
messages = [{"role": "user", "content": "What is 2+2?"}]
|
||||
|
||||
response = await router.aanthropic_messages(
|
||||
messages=messages,
|
||||
model="bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0",
|
||||
max_tokens=1026,
|
||||
thinking={
|
||||
"type": "enabled",
|
||||
"budget_tokens": 1025
|
||||
},
|
||||
)
|
||||
print("bedrock response: ", response)
|
||||
|
||||
# Verify response
|
||||
INSTANCE_BASE_ANTHROPIC_MESSAGES_TEST._validate_response(response)
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
|
|
@ -6,8 +7,10 @@ from fastapi.testclient import TestClient
|
|||
|
||||
sys.path.insert(0, os.path.abspath("../../../../.."))
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from litellm.anthropic_interface import messages
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
|
||||
from litellm.types.utils import Delta, ModelResponse, StreamingChoices
|
||||
|
||||
|
||||
|
|
@ -87,3 +90,136 @@ def test_anthropic_experimental_pass_through_messages_handler_custom_llm_provide
|
|||
assert call_kwargs["custom_llm_provider"] == "my-custom-llm"
|
||||
assert call_kwargs["model"] == "my-custom-llm/my-custom-model"
|
||||
assert call_kwargs["api_key"] == "test-api-key"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_bedrock_converse_budget_tokens_preserved():
|
||||
"""
|
||||
Test that budget_tokens value in thinking parameter is correctly passed to Bedrock Converse API
|
||||
when using messages.acreate with bedrock/converse model.
|
||||
|
||||
The bug was that the messages -> completion adapter was converting thinking to reasoning_effort
|
||||
and losing the original budget_tokens value, causing it to use the default (128) instead.
|
||||
"""
|
||||
client = AsyncHTTPHandler()
|
||||
|
||||
with patch.object(client, "post") as mock_post:
|
||||
mock_response = AsyncMock()
|
||||
mock_response.status_code = 200
|
||||
mock_response.headers = {}
|
||||
mock_response.text = "mock response"
|
||||
mock_response.json.return_value = {
|
||||
"output": {
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": [{"text": "4"}]
|
||||
}
|
||||
},
|
||||
"stopReason": "end_turn",
|
||||
"usage": {
|
||||
"inputTokens": 10,
|
||||
"outputTokens": 5,
|
||||
"totalTokens": 15
|
||||
}
|
||||
}
|
||||
mock_post.return_value = mock_response
|
||||
|
||||
try:
|
||||
await messages.acreate(
|
||||
client=client,
|
||||
max_tokens=1024,
|
||||
messages=[{"role": "user", "content": "What is 2+2?"}],
|
||||
model="bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0",
|
||||
thinking={
|
||||
"budget_tokens": 1024,
|
||||
"type": "enabled"
|
||||
},
|
||||
)
|
||||
except Exception:
|
||||
pass # Expected due to mock response format
|
||||
|
||||
mock_post.assert_called_once()
|
||||
|
||||
call_kwargs = mock_post.call_args.kwargs
|
||||
json_data = call_kwargs.get("json") or json.loads(call_kwargs.get("data", "{}"))
|
||||
print("Request json: ", json.dumps(json_data, indent=4, default=str))
|
||||
|
||||
additional_fields = json_data.get("additionalModelRequestFields", {})
|
||||
thinking_config = additional_fields.get("thinking", {})
|
||||
|
||||
assert "thinking" in additional_fields, "thinking parameter should be in additionalModelRequestFields"
|
||||
assert thinking_config.get("type") == "enabled", "thinking.type should be 'enabled'"
|
||||
assert thinking_config.get("budget_tokens") == 1024, f"thinking.budget_tokens should be 1024, but got {thinking_config.get('budget_tokens')}"
|
||||
|
||||
|
||||
def test_openai_model_with_thinking_converts_to_reasoning_effort():
|
||||
"""
|
||||
Test that when using a non-Anthropic model (like OpenAI gpt-5.2) with thinking parameter,
|
||||
the thinking is converted to reasoning_effort and NOT passed as thinking.
|
||||
|
||||
This ensures we don't regress on issue #16052 where non-Anthropic models would fail
|
||||
with UnsupportedParamsError when thinking was passed directly.
|
||||
"""
|
||||
from litellm.llms.anthropic.experimental_pass_through.messages.handler import (
|
||||
anthropic_messages_handler,
|
||||
)
|
||||
|
||||
with patch("litellm.completion", return_value="test-response") as mock_completion:
|
||||
try:
|
||||
anthropic_messages_handler(
|
||||
max_tokens=1024,
|
||||
messages=[{"role": "user", "content": "What is 2+2?"}],
|
||||
model="openai/gpt-5.2",
|
||||
api_key="test-api-key",
|
||||
thinking={
|
||||
"type": "enabled",
|
||||
"budget_tokens": 1024
|
||||
},
|
||||
)
|
||||
except Exception as e:
|
||||
print(f"Error: {e}")
|
||||
|
||||
mock_completion.assert_called_once()
|
||||
|
||||
call_kwargs = mock_completion.call_args.kwargs
|
||||
|
||||
# Verify reasoning_effort is set (converted from thinking)
|
||||
assert "reasoning_effort" in call_kwargs, "reasoning_effort should be passed to completion"
|
||||
assert call_kwargs["reasoning_effort"] == "minimal", f"reasoning_effort should be 'minimal' for budget_tokens=1024, got {call_kwargs.get('reasoning_effort')}"
|
||||
|
||||
# Verify thinking is NOT passed (non-Claude model)
|
||||
assert "thinking" not in call_kwargs, "thinking should NOT be passed for non-Claude models"
|
||||
|
||||
|
||||
class TestThinkingParameterTransformation:
|
||||
"""Core tests for thinking parameter transformation logic."""
|
||||
|
||||
def test_claude_model_preserves_thinking_with_budget_tokens(self):
|
||||
"""Test that Claude models get thinking parameter passed through with exact budget_tokens."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
|
||||
LiteLLMAnthropicMessagesAdapter,
|
||||
)
|
||||
|
||||
thinking = {"type": "enabled", "budget_tokens": 5000}
|
||||
result = LiteLLMAnthropicMessagesAdapter.translate_thinking_for_model(
|
||||
thinking=thinking,
|
||||
model="bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0",
|
||||
)
|
||||
|
||||
assert result == {"thinking": thinking}
|
||||
assert result["thinking"]["budget_tokens"] == 5000
|
||||
|
||||
def test_non_claude_model_converts_thinking_to_reasoning_effort(self):
|
||||
"""Test that non-Claude models convert thinking to reasoning_effort."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import (
|
||||
LiteLLMAnthropicMessagesAdapter,
|
||||
)
|
||||
|
||||
thinking = {"type": "enabled", "budget_tokens": 1024}
|
||||
result = LiteLLMAnthropicMessagesAdapter.translate_thinking_for_model(
|
||||
thinking=thinking,
|
||||
model="openai/gpt-5.2",
|
||||
)
|
||||
|
||||
assert result == {"reasoning_effort": "minimal"}
|
||||
assert "thinking" not in result
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue