mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-29 01:42:19 +00:00
fix(bedrock): clamp reasoning_effort thinking budget to max_tokens on converse
Fixes #39627. _handle_reasoning_effort_parameter assigned the mapped
thinking budget without capping it against max_tokens, so any request
with max_tokens <= the mapped budget (e.g. medium's 2048) got a 400
from Bedrock. Now routes through AnthropicConfig.cap_thinking_budget_to_max_tokens,
matching the existing adaptive-downgrade branch and the /v1/messages fix in 71a9516.
This commit is contained in:
parent
c8635ecc67
commit
7e02eb219b
2 changed files with 43 additions and 2 deletions
|
|
@ -418,7 +418,9 @@ class AmazonConverseConfig(BaseConfig):
|
|||
}
|
||||
}
|
||||
|
||||
def _handle_reasoning_effort_parameter(self, model: str, reasoning_effort: str, optional_params: dict) -> None:
|
||||
def _handle_reasoning_effort_parameter(
|
||||
self, model: str, reasoning_effort: str, optional_params: dict, max_tokens: int | None = None
|
||||
) -> None:
|
||||
"""
|
||||
Handle the reasoning_effort parameter based on the model type.
|
||||
|
||||
|
|
@ -443,6 +445,8 @@ class AmazonConverseConfig(BaseConfig):
|
|||
custom_llm_provider="bedrock",
|
||||
llm_provider="bedrock_converse",
|
||||
)
|
||||
if mapped_thinking is not None:
|
||||
mapped_thinking = AnthropicConfig.cap_thinking_budget_to_max_tokens(mapped_thinking, max_tokens)
|
||||
if mapped_thinking is None:
|
||||
optional_params.pop("thinking", None)
|
||||
optional_params.pop("output_config", None)
|
||||
|
|
@ -948,7 +952,10 @@ class AmazonConverseConfig(BaseConfig):
|
|||
)
|
||||
elif param == "reasoning_effort" and isinstance(value, str):
|
||||
self._handle_reasoning_effort_parameter(
|
||||
model=model, reasoning_effort=value, optional_params=optional_params
|
||||
model=model,
|
||||
reasoning_effort=value,
|
||||
optional_params=optional_params,
|
||||
max_tokens=non_default_params.get("max_completion_tokens") or non_default_params.get("max_tokens"),
|
||||
)
|
||||
elif param == "output_config" and isinstance(value, dict):
|
||||
mapped_output_config = dict(value)
|
||||
|
|
|
|||
|
|
@ -462,6 +462,40 @@ def test_reasoning_effort_none_omits_thinking_for_anthropic_converse(model):
|
|||
assert "thinking" not in optional_params
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"max_tokens,expect_thinking,expect_budget_below_max_tokens",
|
||||
[
|
||||
(1024, False, None), # at/below Bedrock's min thinking budget -> thinking dropped entirely
|
||||
(2048, True, True), # medium's default budget (2048) equals max_tokens -> must be clamped below it
|
||||
(4096, True, False), # budget already fits comfortably under max_tokens -> left unchanged
|
||||
],
|
||||
)
|
||||
def test_reasoning_effort_thinking_budget_clamped_to_max_tokens_converse(
|
||||
max_tokens, expect_thinking, expect_budget_below_max_tokens
|
||||
):
|
||||
"""Regression #39627: a deployment-level reasoning_effort must not send a
|
||||
thinking.budget_tokens that is >= max_tokens on Bedrock Converse. Bedrock
|
||||
requires maxTokens strictly greater than thinking.budget_tokens, so the
|
||||
mapped budget must be clamped the same way the adaptive-downgrade branch
|
||||
already clamps it, or dropped when even the 1024 minimum can't fit."""
|
||||
config = AmazonConverseConfig()
|
||||
|
||||
optional_params = config.map_openai_params(
|
||||
non_default_params={"max_tokens": max_tokens, "reasoning_effort": "medium"},
|
||||
optional_params={},
|
||||
model="us.anthropic.claude-haiku-4-5-20251001-v1:0",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
if not expect_thinking:
|
||||
assert optional_params.get("thinking") is None
|
||||
return
|
||||
|
||||
thinking = optional_params["thinking"]
|
||||
budget = thinking["budget_tokens"]
|
||||
assert budget < max_tokens if expect_budget_below_max_tokens else budget == 2048
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,effort,expected_effort",
|
||||
[
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue