From 6889ae713db7691fe36d46d2e93faaac288d22af Mon Sep 17 00:00:00 2001 From: Abdel Date: Thu, 17 Sep 2026 01:39:44 +0000 Subject: [PATCH] fix(bedrock): fill default maxTokens for anthropic converse models from the cost map bedrock applies a 4096 output token default when inferenceConfig.maxTokens is omitted, which silently truncates anthropic models whose documented maximum is far higher fill maxTokens from the litellm cost map when the request has no max_tokens, the thinking branch did not set one, and the model is an anthropic id, unknown models stay unchanged fixes #41529 --- .../bedrock/chat/converse_transformation.py | 33 ++++++++++ .../chat/test_converse_transformation.py | 65 +++++++++++++++++++ 2 files changed, 98 insertions(+) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 2a2f3052b2a..38c04bc246d 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -1151,6 +1151,16 @@ class AmazonConverseConfig(BaseConfig): non_default_params=non_default_params, optional_params=optional_params ) + # Bedrock applies a 4096 output-token default when maxTokens is omitted, silently truncating anthropic models + if ( + not self.is_max_tokens_in_request(non_default_params) + and "maxTokens" not in optional_params + and "anthropic" in model + ): + default_max_tokens: Final = self._get_default_max_tokens_for_model(model) + if default_max_tokens is not None: + optional_params["maxTokens"] = default_max_tokens + final_is_thinking_enabled: Final = self.is_thinking_enabled(optional_params) if final_is_thinking_enabled and "tool_choice" in optional_params: tool_choice_block: Final = optional_params["tool_choice"] @@ -1293,6 +1303,29 @@ class AmazonConverseConfig(BaseConfig): if thinking_token_budget is not None: optional_params["maxTokens"] = thinking_token_budget + DEFAULT_MAX_TOKENS + @staticmethod + def _get_default_max_tokens_for_model(model: str) -> int | None: + """ + Resolve the model's documented max output tokens from the litellm cost map. + + Returns None when the model is not mapped, so the request stays unchanged. + Bedrock model ids carry an optional cross-region prefix that cost map keys do + not include, so strip it and retry. + """ + + def lookup(candidate: str) -> int | None: + entry: Final = litellm.model_cost.get(candidate) + if entry is None: + return None + if "max_output_tokens" in entry: + return entry["max_output_tokens"] + return entry.get("max_tokens") + + first_segment, separator, rest = model.partition(".") + region_prefixed: Final = separator == "." and first_segment in ("us", "eu", "apac", "global", "ca", "sa") + candidates: Final = (model, rest) if region_prefixed else (model,) + return next((value for value in map(lookup, candidates) if value is not None), None) + @overload def get_cache_point_block( self, diff --git a/tests/unit/llms/bedrock/chat/test_converse_transformation.py b/tests/unit/llms/bedrock/chat/test_converse_transformation.py index 499096621c5..3ae56c5e24a 100644 --- a/tests/unit/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/unit/llms/bedrock/chat/test_converse_transformation.py @@ -7913,3 +7913,68 @@ def test_supports_sampling_params_prefixed_and_anthropic_fallback(monkeypatch: p ) assert AmazonConverseConfig._supports_sampling_params("custom-test-reasoning-model") is False assert AmazonConverseConfig._supports_sampling_params("anthropic.claude-custom-unregistered") is True + + +def test_anthropic_converse_default_max_tokens_from_cost_map(local_model_cost_map): + """Bedrock caps output at 4096 when maxTokens is omitted; fill the model max from the cost map instead.""" + config = AmazonConverseConfig() + + optional_params = config.map_openai_params( + model="us.anthropic.claude-sonnet-5", + non_default_params={}, + optional_params={}, + drop_params=False, + ) + assert optional_params["maxTokens"] == 128000 + + data = config._transform_request_helper( + model="us.anthropic.claude-sonnet-5", + system_content_blocks=[], + optional_params=optional_params, + messages=None, + ) + assert data["inferenceConfig"]["maxTokens"] == 128000 + + +def test_anthropic_converse_explicit_max_tokens_still_wins(local_model_cost_map): + config = AmazonConverseConfig() + optional_params = config.map_openai_params( + model="us.anthropic.claude-sonnet-5", + non_default_params={"max_tokens": 111}, + optional_params={}, + drop_params=False, + ) + assert optional_params["maxTokens"] == 111 + + +def test_anthropic_converse_default_max_tokens_strips_region_prefix(local_model_cost_map): + config = AmazonConverseConfig() + optional_params = config.map_openai_params( + model="eu.anthropic.claude-3-7-sonnet-20240620-v1:0", + non_default_params={}, + optional_params={}, + drop_params=False, + ) + assert optional_params["maxTokens"] == 8192 + + +def test_non_anthropic_converse_gets_no_default_max_tokens(local_model_cost_map): + config = AmazonConverseConfig() + optional_params = config.map_openai_params( + model="us.meta.llama4-maverick-17b-instruct-v1:0", + non_default_params={}, + optional_params={}, + drop_params=False, + ) + assert "maxTokens" not in optional_params + + +def test_anthropic_converse_unknown_model_gets_no_default_max_tokens(local_model_cost_map): + config = AmazonConverseConfig() + optional_params = config.map_openai_params( + model="us.anthropic.not-a-real-model-v1:0", + non_default_params={}, + optional_params={}, + drop_params=False, + ) + assert "maxTokens" not in optional_params