mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(bedrock): clamp thinking.budget_tokens to minimum 1024
Bedrock rejects thinking.budget_tokens values below 1024 with a 400 error. This adds automatic clamping in the LiteLLM transformation layer so callers (e.g. router with reasoning_effort="low") don't need to know about the provider-specific minimum. Fixes #21297 Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
parent
44bb1dafdb
commit
37da38fdaa
3 changed files with 75 additions and 1 deletions
|
|
@ -319,6 +319,9 @@ NON_LLM_CONNECTION_TIMEOUT = int(
|
|||
MAX_EXCEPTION_MESSAGE_LENGTH = int(os.getenv("MAX_EXCEPTION_MESSAGE_LENGTH", 2000))
|
||||
MAX_STRING_LENGTH_PROMPT_IN_DB = int(os.getenv("MAX_STRING_LENGTH_PROMPT_IN_DB", 2048))
|
||||
BEDROCK_MAX_POLICY_SIZE = int(os.getenv("BEDROCK_MAX_POLICY_SIZE", 75))
|
||||
BEDROCK_MIN_THINKING_BUDGET_TOKENS = int(
|
||||
os.getenv("BEDROCK_MIN_THINKING_BUDGET_TOKENS", 1024)
|
||||
)
|
||||
REPLICATE_POLLING_DELAY_SECONDS = float(
|
||||
os.getenv("REPLICATE_POLLING_DELAY_SECONDS", 0.5)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -11,7 +11,10 @@ import httpx
|
|||
|
||||
import litellm
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.constants import RESPONSE_FORMAT_TOOL_NAME
|
||||
from litellm.constants import (
|
||||
BEDROCK_MIN_THINKING_BUDGET_TOKENS,
|
||||
RESPONSE_FORMAT_TOOL_NAME,
|
||||
)
|
||||
from litellm.litellm_core_utils.core_helpers import (
|
||||
filter_exceptions_from_params,
|
||||
filter_internal_params,
|
||||
|
|
@ -434,6 +437,25 @@ class AmazonConverseConfig(BaseConfig):
|
|||
reasoning_effort=reasoning_effort, model=model
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _clamp_thinking_budget_tokens(optional_params: dict) -> None:
|
||||
"""
|
||||
Clamp thinking.budget_tokens to the Bedrock minimum (1024).
|
||||
|
||||
Bedrock returns a 400 error if budget_tokens < 1024.
|
||||
"""
|
||||
thinking = optional_params.get("thinking")
|
||||
if isinstance(thinking, dict):
|
||||
budget = thinking.get("budget_tokens")
|
||||
if isinstance(budget, int) and budget < BEDROCK_MIN_THINKING_BUDGET_TOKENS:
|
||||
verbose_logger.debug(
|
||||
"Bedrock requires thinking.budget_tokens >= %d, got %d. "
|
||||
"Clamping to minimum.",
|
||||
BEDROCK_MIN_THINKING_BUDGET_TOKENS,
|
||||
budget,
|
||||
)
|
||||
thinking["budget_tokens"] = BEDROCK_MIN_THINKING_BUDGET_TOKENS
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> List[str]:
|
||||
from litellm.utils import supports_function_calling
|
||||
|
||||
|
|
@ -871,9 +893,14 @@ class AmazonConverseConfig(BaseConfig):
|
|||
Checks 'non_default_params' for 'thinking' and 'max_tokens'
|
||||
|
||||
if 'thinking' is enabled and 'max_tokens' is not specified, set 'max_tokens' to the thinking token budget + DEFAULT_MAX_TOKENS
|
||||
|
||||
Also clamps thinking.budget_tokens to the Bedrock minimum (1024) to
|
||||
prevent 400 errors from the Bedrock API.
|
||||
"""
|
||||
from litellm.constants import DEFAULT_MAX_TOKENS
|
||||
|
||||
self._clamp_thinking_budget_tokens(optional_params)
|
||||
|
||||
is_thinking_enabled = self.is_thinking_enabled(optional_params)
|
||||
is_max_tokens_in_request = self.is_max_tokens_in_request(non_default_params)
|
||||
if is_thinking_enabled and not is_max_tokens_in_request:
|
||||
|
|
|
|||
|
|
@ -2934,3 +2934,47 @@ def test_drop_thinking_param_when_thinking_blocks_missing():
|
|||
finally:
|
||||
# Restore original modify_params setting
|
||||
litellm.modify_params = original_modify_params
|
||||
|
||||
|
||||
class TestBedrockMinThinkingBudgetTokens:
|
||||
"""Test that thinking.budget_tokens is clamped to the Bedrock minimum (1024)."""
|
||||
|
||||
def _map_params(
|
||||
self, thinking_value, model="anthropic.claude-3-7-sonnet-20250219-v1:0"
|
||||
):
|
||||
"""Helper to call map_openai_params with the given thinking value."""
|
||||
config = AmazonConverseConfig()
|
||||
non_default_params = {"thinking": thinking_value}
|
||||
optional_params = {"thinking": thinking_value}
|
||||
return config.map_openai_params(
|
||||
non_default_params=non_default_params,
|
||||
optional_params=optional_params,
|
||||
model=model,
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
def test_budget_tokens_below_minimum_is_clamped(self):
|
||||
"""budget_tokens < 1024 should be clamped to 1024."""
|
||||
result = self._map_params({"type": "enabled", "budget_tokens": 499})
|
||||
assert result["thinking"]["budget_tokens"] == 1024
|
||||
|
||||
def test_budget_tokens_at_minimum_is_unchanged(self):
|
||||
"""budget_tokens == 1024 should remain 1024."""
|
||||
result = self._map_params({"type": "enabled", "budget_tokens": 1024})
|
||||
assert result["thinking"]["budget_tokens"] == 1024
|
||||
|
||||
def test_budget_tokens_above_minimum_is_unchanged(self):
|
||||
"""budget_tokens > 1024 should remain unchanged."""
|
||||
result = self._map_params({"type": "enabled", "budget_tokens": 2048})
|
||||
assert result["thinking"]["budget_tokens"] == 2048
|
||||
|
||||
def test_no_thinking_param_does_not_error(self):
|
||||
"""When thinking is not provided, map_openai_params should not raise."""
|
||||
config = AmazonConverseConfig()
|
||||
result = config.map_openai_params(
|
||||
non_default_params={},
|
||||
optional_params={},
|
||||
model="anthropic.claude-3-7-sonnet-20250219-v1:0",
|
||||
drop_params=False,
|
||||
)
|
||||
assert "thinking" not in result or result.get("thinking") is None
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue