fix(bedrock): forward output_config.effort for adaptive-thinking Claude

Claude Sonnet 4.6 / Opus 4.6 / Opus 4.7 expose Anthropic's effort parameter
through 'output_config.effort' (Messages API and Bedrock Invoke) and through
'additionalModelRequestFields.thinking.effort' (Bedrock Converse). LiteLLM
was stripping output_config on every Bedrock route, so Sonnet 4.6 callers
saw the effort never reach the wire body.

This fix:
- Bedrock Invoke /v1/messages: preserve output_config for adaptive-thinking
  models (Claude 4.6/4.7) and for Opus 4.5 with the effort-2025-11-24 beta
  header (auto-attached via the existing is_effort_used path).
- Bedrock Invoke /completion: same — stop stripping output_config for the
  supported model set.
- Bedrock Converse: fold output_config.effort into thinking.effort and
  switch reasoning_effort to populate thinking.effort on adaptive-thinking
  models. output_config itself is still removed before the request signs,
  matching Bedrock's wire shape.
- Map effort-2025-11-24 to itself for 'bedrock' in the beta-headers JSON
  so the auto-attach for Opus 4.5 isn't dropped on the way out.
- Add adaptive-thinking and Opus-4.5-with-beta tests on all three Bedrock
  routes; haiku/sonnet-4 stays in the strip path.

Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-05-01 18:02:34 +00:00
parent 8300657af9
commit f52cbae60c
No known key found for this signature in database
8 changed files with 354 additions and 8 deletions

View file

@ -103,7 +103,7 @@
"computer-use-2025-11-24": "computer-use-2025-11-24",
"context-1m-2025-08-07": "context-1m-2025-08-07",
"context-management-2025-06-27": null,
"effort-2025-11-24": null,
"effort-2025-11-24": "effort-2025-11-24",
"fast-mode-2026-02-01": null,
"files-api-2025-04-14": null,
"fine-grained-tool-streaming-2025-05-14": null,

View file

@ -32,6 +32,7 @@ from litellm.litellm_core_utils.prompt_templates.factory import (
_bedrock_tools_pt,
)
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException
from litellm.types.llms.bedrock import *
from litellm.types.llms.openai import (
@ -452,6 +453,70 @@ class AmazonConverseConfig(BaseConfig):
optional_params["thinking"] = AnthropicConfig._map_reasoning_effort(
reasoning_effort=reasoning_effort, model=model
)
# Claude 4.6/4.7 adaptive thinking takes ``effort`` inside the
# ``thinking`` block on Bedrock Converse (Anthropic's
# ``output_config.effort`` is API-only). Surface the mapped
# effort here so it survives into ``additionalModelRequestFields``.
if (
AnthropicModelInfo._is_adaptive_thinking_model(model)
and isinstance(optional_params.get("thinking"), dict)
and optional_params["thinking"].get("type") == "adaptive"
and "effort" not in optional_params["thinking"]
):
effort_map = {
"low": "low",
"minimal": "low",
"medium": "medium",
"high": "high",
"xhigh": "xhigh",
"max": "max",
}
optional_params["thinking"]["effort"] = effort_map.get(
reasoning_effort, reasoning_effort
)
@staticmethod
def _fold_output_config_effort_into_thinking(
inference_params: dict, model: str
) -> None:
"""
Fold ``output_config.effort`` into ``thinking.effort`` for Bedrock
Converse on adaptive-thinking models (Claude 4.6 / 4.7).
Bedrock Converse exposes Anthropic's ``effort`` parameter inside
``additionalModelRequestFields.thinking`` (per the AWS adaptive
thinking docs), not as a top-level ``output_config`` block. Callers
commonly send the Anthropic Messages API shape — ``output_config:
{effort: ...}`` — when chaining a Messages-style request through
Converse, which leaves the value on the floor.
This helper preserves caller intent: if the model accepts adaptive
thinking and the caller did not already set ``thinking.effort``, we
attach the effort there. ``output_config`` is still stripped from
``inference_params`` separately so it never reaches the wire body.
"""
if not AnthropicModelInfo._is_adaptive_thinking_model(model):
return
output_config = inference_params.get("output_config")
if not isinstance(output_config, dict):
return
effort = output_config.get("effort")
if not (effort and isinstance(effort, str)):
return
thinking = inference_params.get("thinking")
if isinstance(thinking, dict):
# Don't override an explicit thinking.effort set by the caller.
if "effort" not in thinking:
thinking["effort"] = effort
# Make sure the type is adaptive — only adaptive thinking accepts
# effort. If the caller passed ``enabled``, leave it alone; the
# downstream `_translate_legacy_thinking_for_adaptive_model` flow
# owns that translation.
else:
inference_params["thinking"] = {
"type": "adaptive",
"effort": effort,
}
@staticmethod
def _clamp_thinking_budget_tokens(optional_params: dict) -> None:
@ -1192,6 +1257,17 @@ class AmazonConverseConfig(BaseConfig):
+ supported_config_params
)
inference_params.pop("json_mode", None) # used for handling json_schema
# Bedrock Converse exposes the Anthropic ``effort`` parameter through
# ``additionalModelRequestFields.thinking.effort`` (per AWS adaptive
# thinking docs). If the caller supplied ``output_config.effort`` —
# the Anthropic Messages API shape — fold it into ``thinking`` for
# Claude 4.6/4.7 adaptive-thinking models so the value reaches the
# wire body. For all other models the field is unsupported and gets
# stripped below.
self._fold_output_config_effort_into_thinking(
inference_params=inference_params, model=model
)
# Anthropic-only key. Bedrock expects `outputConfig` (camelCase) and
# will reject `output_config` if it leaks through pass-through routes.
inference_params.pop("output_config", None)
@ -1204,9 +1280,6 @@ class AmazonConverseConfig(BaseConfig):
output_config: Optional[OutputConfigBlock] = inference_params.pop(
"outputConfig", None
)
inference_params.pop(
"output_config", None
) # Bedrock Converse doesn't support it
# keep supported params in 'inference_params', and set all model-specific params in 'additional_request_params'
additional_request_params = {

View file

@ -11,6 +11,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import (
convert_url_to_base64,
)
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation import (
AmazonInvokeConfig,
)
@ -31,6 +32,21 @@ else:
LiteLLMLoggingObj = Any
def _supports_effort_on_bedrock_invoke(model: str) -> bool:
"""
Bedrock Invoke (legacy /completion path) accepts ``output_config.effort``
on Claude 4.6/4.7 adaptive-thinking models (no beta header) and on Claude
Opus 4.5 with the ``effort-2025-11-24`` beta header. All other Claude
models reject the field.
"""
if AnthropicModelInfo._is_adaptive_thinking_model(model):
return True
model_lower = model.lower()
return any(
p in model_lower for p in ("opus-4.5", "opus_4.5", "opus-4-5", "opus_4_5")
)
class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
"""
Reference:
@ -169,7 +185,12 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
anthropic_request.pop("model", None)
anthropic_request.pop("stream", None)
anthropic_request.pop("output_format", None)
anthropic_request.pop("output_config", None)
# Bedrock accepts ``output_config`` (carrying ``effort``) on Claude 4.6/4.7
# adaptive-thinking models natively, and on Claude Opus 4.5 with the
# ``effort-2025-11-24`` beta header. Older Claude models reject the
# field — strip it for them so the request signs cleanly.
if not _supports_effort_on_bedrock_invoke(model):
anthropic_request.pop("output_config", None)
if "anthropic_version" not in anthropic_request:
anthropic_request["anthropic_version"] = self.anthropic_version

View file

@ -290,6 +290,19 @@ class AmazonAnthropicClaudeMessagesConfig(
)
return True
def _supports_effort_on_bedrock(self, model: str) -> bool:
"""
Whether Bedrock accepts ``output_config.effort`` for this model.
Adaptive-thinking models (Claude 4.6 / 4.7) take ``effort`` natively
with no beta header. Claude Opus 4.5 takes ``effort`` only when the
``effort-2025-11-24`` beta header is attached. Older Claude models
reject the field outright.
"""
return AnthropicModelInfo._is_adaptive_thinking_model(
model
) or self._is_claude_opus_4_5(model)
def _is_claude_opus_4_5(self, model: str) -> bool:
"""
Check if the model is Claude Opus 4.5.
@ -511,6 +524,16 @@ class AmazonAnthropicClaudeMessagesConfig(
anthropic_messages_request=anthropic_messages_request,
)
# 5b. ``output_config`` (carries the Anthropic ``effort`` parameter)
# is accepted by Bedrock on:
# - Claude 4.6/4.7 adaptive-thinking models (no beta header needed)
# - Claude Opus 4.5 with the ``effort-2025-11-24`` beta header
# Other Claude models on Bedrock reject the field, so strip it for
# them. The allowlist (step 7 below) only preserves keys that survive
# this filter.
if not self._supports_effort_on_bedrock(model):
anthropic_messages_request.pop("output_config", None)
# 5a. Remove `custom` field from tools (Bedrock doesn't support it)
# Claude Code sends `custom: {defer_loading: true}` on tool definitions,
# which causes Bedrock to reject the request with "Extra inputs are not permitted"

View file

@ -1041,3 +1041,12 @@ class BedrockInvokeAnthropicMessagesRequest(TypedDict, total=False):
# `metadata` is part of the common Anthropic Messages API shape.
thinking: dict
metadata: dict
# `output_config` carries the Anthropic ``effort`` parameter. Bedrock
# accepts it on the Invoke Messages API for Claude 4.6+ adaptive thinking
# (no beta header) and for Opus 4.5 with the ``effort-2025-11-24`` beta.
# Older Claude models on Bedrock reject this field, so the runtime strips
# it for them — see ``AmazonAnthropicClaudeMessagesConfig`` for that
# model-aware filtering. Including it here lets the allowlist preserve it
# for the supported-model path.
output_config: dict

View file

@ -440,6 +440,51 @@ def test_output_config_removed_from_bedrock_chat_invoke_request():
assert result["max_tokens"] == 100
def test_output_config_preserved_for_adaptive_thinking_on_bedrock_invoke():
"""
Bedrock Invoke (legacy /completion path) must forward ``output_config``
for Claude 4.6/4.7 adaptive-thinking models — the field reaches the wire
body. This mirrors the /v1/messages behavior for the same models.
"""
config = AmazonAnthropicClaudeConfig()
messages = [{"role": "user", "content": "test"}]
optional_params = {
"max_tokens": 100,
"output_config": {"effort": "high"},
}
result = config.transform_request(
model="anthropic.claude-sonnet-4-6-v1:0",
messages=messages,
optional_params=optional_params,
litellm_params={},
headers={},
)
assert result.get("output_config") == {"effort": "high"}
def test_output_config_preserved_for_opus_4_5_on_bedrock_invoke():
"""
Bedrock Invoke /completion forwards ``output_config`` on Claude Opus 4.5
(gated behind the ``effort-2025-11-24`` beta header on Bedrock).
"""
config = AmazonAnthropicClaudeConfig()
messages = [{"role": "user", "content": "test"}]
optional_params = {
"max_tokens": 100,
"output_config": {"effort": "low"},
}
result = config.transform_request(
model="anthropic.claude-opus-4-5-20251101-v1:0",
messages=messages,
optional_params=optional_params,
litellm_params={},
headers={},
)
assert result.get("output_config") == {"effort": "low"}
def test_output_format_removed_from_bedrock_invoke_request():
"""
Test that output_format parameter is removed from Bedrock Invoke requests.

View file

@ -3284,6 +3284,98 @@ def test_transform_request_with_output_config():
)
def test_converse_folds_output_config_effort_into_thinking_for_46():
"""
Bedrock Converse exposes the Anthropic ``effort`` parameter through
``additionalModelRequestFields.thinking.effort`` (per AWS adaptive thinking
docs), not as a top-level ``output_config`` block. The transform must
fold ``output_config.effort`` into ``thinking`` for adaptive-thinking
models so the value survives onto the wire body.
Customer regression: Sonnet 4.6 ``output_config.effort`` was being dropped
on Bedrock Converse and never reached the request payload.
"""
config = AmazonConverseConfig()
messages = [{"role": "user", "content": "hello"}]
result = config._transform_request(
model="us.anthropic.claude-sonnet-4-6-v1:0",
messages=messages,
optional_params={
"maxTokens": 64,
"output_config": {"effort": "low"},
},
litellm_params={},
headers={},
)
assert "outputConfig" not in result
additional_fields = result.get("additionalModelRequestFields", {})
assert "output_config" not in additional_fields
thinking = additional_fields.get("thinking")
assert isinstance(thinking, dict)
assert thinking.get("type") == "adaptive"
assert thinking.get("effort") == "low"
def test_converse_folds_output_config_effort_preserves_existing_thinking():
"""
When the caller already supplied a ``thinking`` block, fold ``effort``
into it without overriding any explicit ``thinking.effort``.
"""
config = AmazonConverseConfig()
messages = [{"role": "user", "content": "hello"}]
# Caller already set thinking with no effort — fold into it.
result = config._transform_request(
model="us.anthropic.claude-opus-4-7-v1",
messages=messages,
optional_params={
"maxTokens": 64,
"thinking": {"type": "adaptive"},
"output_config": {"effort": "high"},
},
litellm_params={},
headers={},
)
thinking = result["additionalModelRequestFields"].get("thinking")
assert thinking == {"type": "adaptive", "effort": "high"}
# Caller set thinking.effort explicitly — must not be overridden.
result = config._transform_request(
model="us.anthropic.claude-opus-4-7-v1",
messages=messages,
optional_params={
"maxTokens": 64,
"thinking": {"type": "adaptive", "effort": "max"},
"output_config": {"effort": "low"},
},
litellm_params={},
headers={},
)
thinking = result["additionalModelRequestFields"].get("thinking")
assert thinking.get("effort") == "max"
def test_converse_reasoning_effort_sets_thinking_effort_for_46():
"""
On adaptive-thinking models, OpenAI ``reasoning_effort`` should produce
``thinking.type=adaptive`` plus ``thinking.effort`` so Converse forwards
the value through ``additionalModelRequestFields``.
"""
config = AmazonConverseConfig()
optional_params: dict = {}
config._handle_reasoning_effort_parameter(
model="us.anthropic.claude-sonnet-4-6-v1:0",
reasoning_effort="medium",
optional_params=optional_params,
)
thinking = optional_params.get("thinking")
assert isinstance(thinking, dict)
assert thinking.get("type") == "adaptive"
assert thinking.get("effort") == "medium"
def test_transform_request_strips_anthropic_output_config():
"""
output_config is Anthropic-specific and must never be forwarded to Bedrock.

View file

@ -625,6 +625,79 @@ def test_bedrock_messages_strips_output_config():
assert result.get("max_tokens") == 4096
def test_bedrock_messages_preserves_output_config_for_adaptive_thinking_models():
"""
Bedrock Invoke accepts ``output_config.effort`` natively on Claude 4.6/4.7
adaptive-thinking models (no beta header required). The transform must
forward the field instead of stripping it.
Customer regression: Sonnet 4.6 ``output_config.effort`` was being dropped
on Bedrock Invoke and never appeared in the wire body.
"""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
for model in (
"anthropic.claude-sonnet-4-6-v1:0",
"global.anthropic.claude-sonnet-4-6-v1:0",
"anthropic.claude-opus-4-6-v1",
"us.anthropic.claude-opus-4-7-v1",
):
optional_params = {
"max_tokens": 4096,
"output_config": {"effort": "low"},
}
result = cfg.transform_anthropic_messages_request(
model=model,
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert result.get("output_config") == {
"effort": "low"
}, f"output_config should be forwarded for {model}, got {result!r}"
# Adaptive-thinking models don't need the effort beta header.
betas = result.get("anthropic_beta") or []
assert "effort-2025-11-24" not in betas
def test_bedrock_messages_preserves_output_config_for_opus_4_5_with_beta(
monkeypatch,
):
"""
Bedrock Invoke accepts ``output_config.effort`` on Claude Opus 4.5 only
when the ``effort-2025-11-24`` beta header is attached. The transform must
forward the field and auto-attach that beta.
"""
from litellm.types.router import GenericLiteLLMParams
import litellm.anthropic_beta_headers_manager as beta_mgr
# Force local beta-header config — the remote fetch may lag the in-tree
# mapping update that allows ``effort-2025-11-24`` for ``bedrock``.
monkeypatch.setenv("LITELLM_LOCAL_ANTHROPIC_BETA_HEADERS", "True")
monkeypatch.setattr(beta_mgr, "_BETA_HEADERS_CONFIG", None)
cfg = AmazonAnthropicClaudeMessagesConfig()
messages = [{"role": "user", "content": [{"type": "text", "text": "Hello"}]}]
optional_params = {
"max_tokens": 4096,
"output_config": {"effort": "medium"},
}
result = cfg.transform_anthropic_messages_request(
model="anthropic.claude-opus-4-5-20251101-v1:0",
messages=messages,
anthropic_messages_optional_request_params=optional_params,
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert result.get("output_config") == {"effort": "medium"}
betas = result.get("anthropic_beta") or []
assert "effort-2025-11-24" in betas
def test_bedrock_messages_strips_output_config_with_output_format():
"""
When both output_config and output_format are present, both should be
@ -882,7 +955,10 @@ async def test_promote_message_start_cache_when_message_stop_omits_cache_fields(
"delta": {"stop_reason": "end_turn", "stop_sequence": None},
"usage": {"input_tokens": 10, "output_tokens": 181},
}
yield {"type": "message_stop", "usage": {"input_tokens": 10, "output_tokens": 181}}
yield {
"type": "message_stop",
"usage": {"input_tokens": 10, "output_tokens": 181},
}
merged: list[dict] = []
async for chunk in cfg._promote_message_stop_usage(_stream()):
@ -936,7 +1012,11 @@ async def test_unified_bedrock_messages_cache_on_start_only_never_negative_cost(
},
},
}
yield {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}
yield {
"type": "content_block_start",
"index": 0,
"content_block": {"type": "text", "text": ""},
}
yield {
"type": "content_block_delta",
"index": 0,
@ -948,7 +1028,10 @@ async def test_unified_bedrock_messages_cache_on_start_only_never_negative_cost(
"delta": {"stop_reason": "end_turn", "stop_sequence": None},
"usage": {"output_tokens": 181, "input_tokens": 10},
}
yield {"type": "message_stop", "usage": {"input_tokens": 10, "output_tokens": 181}}
yield {
"type": "message_stop",
"usage": {"input_tokens": 10, "output_tokens": 181},
}
logging_obj = LiteLLMLoggingObj(
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",