From 7e02eb219b612b020cd2cc04a2640e26ad433a6f Mon Sep 17 00:00:00 2001 From: aishwary-dongre <87765118+aishwary-dongre@users.noreply.github.com> Date: Fri, 4 Sep 2026 11:35:34 +0000 Subject: [PATCH 1/3] fix(bedrock): clamp reasoning_effort thinking budget to max_tokens on converse Fixes #39627. _handle_reasoning_effort_parameter assigned the mapped thinking budget without capping it against max_tokens, so any request with max_tokens <= the mapped budget (e.g. medium's 2048) got a 400 from Bedrock. Now routes through AnthropicConfig.cap_thinking_budget_to_max_tokens, matching the existing adaptive-downgrade branch and the /v1/messages fix in 71a9516. --- .../bedrock/chat/converse_transformation.py | 11 ++++-- .../chat/test_converse_transformation.py | 34 +++++++++++++++++++ 2 files changed, 43 insertions(+), 2 deletions(-) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index e097805f54a..80f053cda61 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -418,7 +418,9 @@ class AmazonConverseConfig(BaseConfig): } } - def _handle_reasoning_effort_parameter(self, model: str, reasoning_effort: str, optional_params: dict) -> None: + def _handle_reasoning_effort_parameter( + self, model: str, reasoning_effort: str, optional_params: dict, max_tokens: int | None = None + ) -> None: """ Handle the reasoning_effort parameter based on the model type. @@ -443,6 +445,8 @@ class AmazonConverseConfig(BaseConfig): custom_llm_provider="bedrock", llm_provider="bedrock_converse", ) + if mapped_thinking is not None: + mapped_thinking = AnthropicConfig.cap_thinking_budget_to_max_tokens(mapped_thinking, max_tokens) if mapped_thinking is None: optional_params.pop("thinking", None) optional_params.pop("output_config", None) @@ -948,7 +952,10 @@ class AmazonConverseConfig(BaseConfig): ) elif param == "reasoning_effort" and isinstance(value, str): self._handle_reasoning_effort_parameter( - model=model, reasoning_effort=value, optional_params=optional_params + model=model, + reasoning_effort=value, + optional_params=optional_params, + max_tokens=non_default_params.get("max_completion_tokens") or non_default_params.get("max_tokens"), ) elif param == "output_config" and isinstance(value, dict): mapped_output_config = dict(value) diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index cb05cdb9451..2e566eee0cc 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -462,6 +462,40 @@ def test_reasoning_effort_none_omits_thinking_for_anthropic_converse(model): assert "thinking" not in optional_params +@pytest.mark.parametrize( + "max_tokens,expect_thinking,expect_budget_below_max_tokens", + [ + (1024, False, None), # at/below Bedrock's min thinking budget -> thinking dropped entirely + (2048, True, True), # medium's default budget (2048) equals max_tokens -> must be clamped below it + (4096, True, False), # budget already fits comfortably under max_tokens -> left unchanged + ], +) +def test_reasoning_effort_thinking_budget_clamped_to_max_tokens_converse( + max_tokens, expect_thinking, expect_budget_below_max_tokens +): + """Regression #39627: a deployment-level reasoning_effort must not send a + thinking.budget_tokens that is >= max_tokens on Bedrock Converse. Bedrock + requires maxTokens strictly greater than thinking.budget_tokens, so the + mapped budget must be clamped the same way the adaptive-downgrade branch + already clamps it, or dropped when even the 1024 minimum can't fit.""" + config = AmazonConverseConfig() + + optional_params = config.map_openai_params( + non_default_params={"max_tokens": max_tokens, "reasoning_effort": "medium"}, + optional_params={}, + model="us.anthropic.claude-haiku-4-5-20251001-v1:0", + drop_params=False, + ) + + if not expect_thinking: + assert optional_params.get("thinking") is None + return + + thinking = optional_params["thinking"] + budget = thinking["budget_tokens"] + assert budget < max_tokens if expect_budget_below_max_tokens else budget == 2048 + + @pytest.mark.parametrize( "model,effort,expected_effort", [ From abc80708c1840f9fa4ac92daa3968602e81d8d97 Mon Sep 17 00:00:00 2001 From: aishwary-dongre <87765118+aishwary-dongre@users.noreply.github.com> Date: Fri, 4 Sep 2026 17:22:50 +0000 Subject: [PATCH 2/3] address greptile review: drop restating comments, type test params --- .../bedrock/chat/test_converse_transformation.py | 16 ++++++---------- 1 file changed, 6 insertions(+), 10 deletions(-) diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index 2e566eee0cc..459207f8b88 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -465,19 +465,15 @@ def test_reasoning_effort_none_omits_thinking_for_anthropic_converse(model): @pytest.mark.parametrize( "max_tokens,expect_thinking,expect_budget_below_max_tokens", [ - (1024, False, None), # at/below Bedrock's min thinking budget -> thinking dropped entirely - (2048, True, True), # medium's default budget (2048) equals max_tokens -> must be clamped below it - (4096, True, False), # budget already fits comfortably under max_tokens -> left unchanged + (1024, False, None), + (2048, True, True), + (4096, True, False), ], ) def test_reasoning_effort_thinking_budget_clamped_to_max_tokens_converse( - max_tokens, expect_thinking, expect_budget_below_max_tokens -): - """Regression #39627: a deployment-level reasoning_effort must not send a - thinking.budget_tokens that is >= max_tokens on Bedrock Converse. Bedrock - requires maxTokens strictly greater than thinking.budget_tokens, so the - mapped budget must be clamped the same way the adaptive-downgrade branch - already clamps it, or dropped when even the 1024 minimum can't fit.""" + max_tokens: int, expect_thinking: bool, expect_budget_below_max_tokens: bool | None +) -> None: + """Regression #39627.""" config = AmazonConverseConfig() optional_params = config.map_openai_params( From dc29b9693309dadad10933474149f96717e24f17 Mon Sep 17 00:00:00 2001 From: aishwary-dongre <87765118+aishwary-dongre@users.noreply.github.com> Date: Fri, 4 Sep 2026 18:09:10 +0000 Subject: [PATCH 3/3] fix: remove Final from mapped_thinking to allow clamped reassignment basedpyright flagged mapped_thinking as Final while the clamp fix reassigns it after capping the budget. Dropping Final resolves the reportGeneralTypeIssues gate failure without changing behavior. --- litellm/llms/bedrock/chat/converse_transformation.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 80f053cda61..74053876ed7 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -439,7 +439,7 @@ class AmazonConverseConfig(BaseConfig): reasoning_config: Final = self._transform_reasoning_effort_to_reasoning_config(reasoning_effort) optional_params.update(reasoning_config) else: - mapped_thinking: Final = AnthropicConfig._map_reasoning_effort( + mapped_thinking = AnthropicConfig._map_reasoning_effort( reasoning_effort=reasoning_effort, model=model, custom_llm_provider="bedrock",