From 747829dadb82a1c55d664cc36f5c574edce4d307 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Wed, 14 Jan 2026 12:02:27 -0800 Subject: [PATCH] [Fix] Claude Code + Bedrock Converse Usage - ensure budget tokens are passed to converse api correctly (#19107) * test_bedrock_converse_budget_tokens_preserved * test_openai_model_with_thinking_converts_to_reasoning_effort * fix translate_anthropic_thinking_to_reasoning_effort * test_bedrock_converse_budget_tokens_preserved * test_anthropic_messages_bedrock_converse_with_thinking --- .../adapters/transformation.py | 83 +++++++++-- litellm/proxy/proxy_config.yaml | 10 +- .../test_bedrock_anthropic_messages_test.py | 32 ++++ ...erimental_pass_through_messages_handler.py | 138 +++++++++++++++++- 4 files changed, 244 insertions(+), 19 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index 06092755b17..cb2110aee9a 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -17,7 +17,6 @@ from openai.types.chat.chat_completion_chunk import Choice as OpenAIStreamingCho from litellm.litellm_core_utils.prompt_templates.common_utils import ( parse_tool_call_arguments, ) - from litellm.types.llms.anthropic import ( AllAnthropicToolsValues, AnthopicMessagesAssistantMessageParam, @@ -210,7 +209,7 @@ class LiteLLMAnthropicMessagesAdapter: # Convert Anthropic image format to OpenAI format source = content.get("source", {}) openai_image_url = ( - self._translate_anthropic_image_to_openai(source) + self._translate_anthropic_image_to_openai(cast(dict, source)) ) if openai_image_url: @@ -240,7 +239,7 @@ class LiteLLMAnthropicMessagesAdapter: # Combine all content items into a single tool message # to avoid creating multiple tool_result blocks with the same ID # (each tool_use must have exactly one tool_result) - content_items = content.get("content", []) + content_items = list(content.get("content", [])) # For single-item content, maintain backward compatibility with string/url format if len(content_items) == 1: @@ -266,7 +265,7 @@ class LiteLLMAnthropicMessagesAdapter: source = c.get("source", {}) openai_image_url = ( self._translate_anthropic_image_to_openai( - source + cast(dict, source) ) or "" ) @@ -306,7 +305,7 @@ class LiteLLMAnthropicMessagesAdapter: source = c.get("source", {}) openai_image_url = ( self._translate_anthropic_image_to_openai( - source + cast(dict, source) ) or "" ) @@ -363,7 +362,7 @@ class LiteLLMAnthropicMessagesAdapter: } signature = ( self._extract_signature_from_tool_use_content( - content + cast(Dict[str, Any], content) ) ) @@ -424,14 +423,21 @@ class LiteLLMAnthropicMessagesAdapter: return new_messages - def translate_anthropic_thinking_to_openai( - self, thinking: Dict[str, Any] + @staticmethod + def translate_anthropic_thinking_to_reasoning_effort( + thinking: Dict[str, Any] ) -> Optional[str]: """ Translate Anthropic's thinking parameter to OpenAI's reasoning_effort. Anthropic thinking format: {'type': 'enabled'|'disabled', 'budget_tokens': int} OpenAI reasoning_effort: 'none' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'default' + + Mapping: + - budget_tokens >= 10000 -> 'high' + - budget_tokens >= 5000 -> 'medium' + - budget_tokens >= 2000 -> 'low' + - budget_tokens < 2000 -> 'minimal' """ if not isinstance(thinking, dict): return None @@ -453,6 +459,53 @@ class LiteLLMAnthropicMessagesAdapter: return None + @staticmethod + def is_anthropic_claude_model(model: str) -> bool: + """ + Check if the model is an Anthropic Claude model that supports the thinking parameter. + + Returns True for: + - anthropic/* models + - bedrock/*anthropic* models (including converse) + - vertex_ai/*claude* models + """ + model_lower = model.lower() + return ( + "anthropic" in model_lower + or "claude" in model_lower + ) + + @staticmethod + def translate_thinking_for_model( + thinking: Dict[str, Any], + model: str, + ) -> Dict[str, Any]: + """ + Translate Anthropic thinking parameter based on the target model. + + For Claude/Anthropic models: returns {'thinking': } + - Preserves exact budget_tokens value + + For non-Claude models: returns {'reasoning_effort': } + - Converts thinking to reasoning_effort to avoid UnsupportedParamsError + + Args: + thinking: Anthropic thinking dict with 'type' and 'budget_tokens' + model: The target model name + + Returns: + Dict with either 'thinking' or 'reasoning_effort' key + """ + if LiteLLMAnthropicMessagesAdapter.is_anthropic_claude_model(model): + return {"thinking": thinking} + else: + reasoning_effort = LiteLLMAnthropicMessagesAdapter.translate_anthropic_thinking_to_reasoning_effort( + thinking + ) + if reasoning_effort: + return {"reasoning_effort": reasoning_effort} + return {} + def translate_anthropic_tool_choice_to_openai( self, tool_choice: AnthropicMessagesToolChoice ) -> ChatCompletionToolChoiceValues: @@ -566,11 +619,15 @@ class LiteLLMAnthropicMessagesAdapter: if "thinking" in anthropic_message_request: thinking = anthropic_message_request["thinking"] if thinking: - reasoning_effort = self.translate_anthropic_thinking_to_openai( - thinking=cast(Dict[str, Any], thinking) - ) - if reasoning_effort: - new_kwargs["reasoning_effort"] = reasoning_effort + model = new_kwargs.get("model", "") + if self.is_anthropic_claude_model(model): + new_kwargs["thinking"] = thinking # type: ignore + else: + reasoning_effort = self.translate_anthropic_thinking_to_reasoning_effort( + cast(Dict[str, Any], thinking) + ) + if reasoning_effort: + new_kwargs["reasoning_effort"] = reasoning_effort translatable_params = self.translatable_anthropic_params() for k, v in anthropic_message_request.items(): diff --git a/litellm/proxy/proxy_config.yaml b/litellm/proxy/proxy_config.yaml index 576ba24aac2..54a923e3bbb 100644 --- a/litellm/proxy/proxy_config.yaml +++ b/litellm/proxy/proxy_config.yaml @@ -1,10 +1,10 @@ model_list: - - model_name: anthropic/* + - model_name: us.anthropic.claude-sonnet-4-20250514-v1:0 litellm_params: - model: anthropic/* - - model_name: openai/* - litellm_params: - model: openai/* + model: bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0 + model_info: + litellm_provider: bedrock_converse + mode: chat general_settings: store_prompts_in_spend_logs: true \ No newline at end of file diff --git a/tests/pass_through_unit_tests/test_bedrock_anthropic_messages_test.py b/tests/pass_through_unit_tests/test_bedrock_anthropic_messages_test.py index 41edc8572cd..155af6b6a95 100644 --- a/tests/pass_through_unit_tests/test_bedrock_anthropic_messages_test.py +++ b/tests/pass_through_unit_tests/test_bedrock_anthropic_messages_test.py @@ -66,3 +66,35 @@ async def test_anthropic_messages_litellm_router_bedrock(): INSTANCE_BASE_ANTHROPIC_MESSAGES_TEST._validate_response(response) +@pytest.mark.asyncio +async def test_anthropic_messages_bedrock_converse_with_thinking(): + """ + Test that bedrock/converse model works with thinking parameter. + Validates the request body from issue where budget_tokens was being lost. + """ + router = Router( + model_list=[ + { + "model_name": "bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0", + "litellm_params": { + "model": "bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0", + }, + }, + ] + ) + + messages = [{"role": "user", "content": "What is 2+2?"}] + + response = await router.aanthropic_messages( + messages=messages, + model="bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0", + max_tokens=1026, + thinking={ + "type": "enabled", + "budget_tokens": 1025 + }, + ) + print("bedrock response: ", response) + + # Verify response + INSTANCE_BASE_ANTHROPIC_MESSAGES_TEST._validate_response(response) diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index 653f9e8e31e..66d62aae1ec 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -1,3 +1,4 @@ +import json import os import sys @@ -6,8 +7,10 @@ from fastapi.testclient import TestClient sys.path.insert(0, os.path.abspath("../../../../..")) -from unittest.mock import MagicMock, patch +from unittest.mock import AsyncMock, MagicMock, patch +from litellm.anthropic_interface import messages +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler from litellm.types.utils import Delta, ModelResponse, StreamingChoices @@ -87,3 +90,136 @@ def test_anthropic_experimental_pass_through_messages_handler_custom_llm_provide assert call_kwargs["custom_llm_provider"] == "my-custom-llm" assert call_kwargs["model"] == "my-custom-llm/my-custom-model" assert call_kwargs["api_key"] == "test-api-key" + + +@pytest.mark.asyncio +async def test_bedrock_converse_budget_tokens_preserved(): + """ + Test that budget_tokens value in thinking parameter is correctly passed to Bedrock Converse API + when using messages.acreate with bedrock/converse model. + + The bug was that the messages -> completion adapter was converting thinking to reasoning_effort + and losing the original budget_tokens value, causing it to use the default (128) instead. + """ + client = AsyncHTTPHandler() + + with patch.object(client, "post") as mock_post: + mock_response = AsyncMock() + mock_response.status_code = 200 + mock_response.headers = {} + mock_response.text = "mock response" + mock_response.json.return_value = { + "output": { + "message": { + "role": "assistant", + "content": [{"text": "4"}] + } + }, + "stopReason": "end_turn", + "usage": { + "inputTokens": 10, + "outputTokens": 5, + "totalTokens": 15 + } + } + mock_post.return_value = mock_response + + try: + await messages.acreate( + client=client, + max_tokens=1024, + messages=[{"role": "user", "content": "What is 2+2?"}], + model="bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0", + thinking={ + "budget_tokens": 1024, + "type": "enabled" + }, + ) + except Exception: + pass # Expected due to mock response format + + mock_post.assert_called_once() + + call_kwargs = mock_post.call_args.kwargs + json_data = call_kwargs.get("json") or json.loads(call_kwargs.get("data", "{}")) + print("Request json: ", json.dumps(json_data, indent=4, default=str)) + + additional_fields = json_data.get("additionalModelRequestFields", {}) + thinking_config = additional_fields.get("thinking", {}) + + assert "thinking" in additional_fields, "thinking parameter should be in additionalModelRequestFields" + assert thinking_config.get("type") == "enabled", "thinking.type should be 'enabled'" + assert thinking_config.get("budget_tokens") == 1024, f"thinking.budget_tokens should be 1024, but got {thinking_config.get('budget_tokens')}" + + +def test_openai_model_with_thinking_converts_to_reasoning_effort(): + """ + Test that when using a non-Anthropic model (like OpenAI gpt-5.2) with thinking parameter, + the thinking is converted to reasoning_effort and NOT passed as thinking. + + This ensures we don't regress on issue #16052 where non-Anthropic models would fail + with UnsupportedParamsError when thinking was passed directly. + """ + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + anthropic_messages_handler, + ) + + with patch("litellm.completion", return_value="test-response") as mock_completion: + try: + anthropic_messages_handler( + max_tokens=1024, + messages=[{"role": "user", "content": "What is 2+2?"}], + model="openai/gpt-5.2", + api_key="test-api-key", + thinking={ + "type": "enabled", + "budget_tokens": 1024 + }, + ) + except Exception as e: + print(f"Error: {e}") + + mock_completion.assert_called_once() + + call_kwargs = mock_completion.call_args.kwargs + + # Verify reasoning_effort is set (converted from thinking) + assert "reasoning_effort" in call_kwargs, "reasoning_effort should be passed to completion" + assert call_kwargs["reasoning_effort"] == "minimal", f"reasoning_effort should be 'minimal' for budget_tokens=1024, got {call_kwargs.get('reasoning_effort')}" + + # Verify thinking is NOT passed (non-Claude model) + assert "thinking" not in call_kwargs, "thinking should NOT be passed for non-Claude models" + + +class TestThinkingParameterTransformation: + """Core tests for thinking parameter transformation logic.""" + + def test_claude_model_preserves_thinking_with_budget_tokens(self): + """Test that Claude models get thinking parameter passed through with exact budget_tokens.""" + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + LiteLLMAnthropicMessagesAdapter, + ) + + thinking = {"type": "enabled", "budget_tokens": 5000} + result = LiteLLMAnthropicMessagesAdapter.translate_thinking_for_model( + thinking=thinking, + model="bedrock/converse/us.anthropic.claude-sonnet-4-20250514-v1:0", + ) + + assert result == {"thinking": thinking} + assert result["thinking"]["budget_tokens"] == 5000 + + def test_non_claude_model_converts_thinking_to_reasoning_effort(self): + """Test that non-Claude models convert thinking to reasoning_effort.""" + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + LiteLLMAnthropicMessagesAdapter, + ) + + thinking = {"type": "enabled", "budget_tokens": 1024} + result = LiteLLMAnthropicMessagesAdapter.translate_thinking_for_model( + thinking=thinking, + model="openai/gpt-5.2", + ) + + assert result == {"reasoning_effort": "minimal"} + assert "thinking" not in result