fix(anthropic): reject explicit unsupported effort instead of rewriting it

An explicit output_config.effort is the caller's native choice, so it wins
over the reasoning_effort alias and a tier the model rejects returns a 400
on both the chat and /v1/messages paths, matching the existing chat
contract. Only the alias is lowered to a tier the model is known to accept,
through one shared helper.

Bedrock invoke clamps an explicit effort sent alongside the alias to its
effort ceiling before the shared gate, as it already did for the alias.

Tests register a real Router deployment without effort metadata to prove it
keeps the requested tier, replacing get_model_info mocks.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
hx 2026-09-27 18:08:26 +08:00
parent f9209497cb
commit 372fc14486
7 changed files with 176 additions and 160 deletions

View file

@ -415,6 +415,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
return f"effort='xhigh' is not supported by this model. Got model: {model}"
return None
@staticmethod
def degrade_alias_effort_for_model(model: str, effort: str, custom_llm_provider: str) -> str:
"""Keep an alias-derived effort the gate accepts, else lower it to a tier the model is known to accept.
Explicit ``output_config.effort`` must not be routed here: a caller naming a native tier gets a 400.
"""
if AnthropicConfig._validate_effort_for_model(model, effort, custom_llm_provider) is None:
return effort
return normalize_reasoning_effort_value(effort, model, custom_llm_provider)
@staticmethod
def _model_supports_effort_param(model: str, custom_llm_provider: str) -> bool:
"""Whether the model accepts ``output_config.effort`` at all.
@ -1265,33 +1275,33 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
type="adaptive",
display="summarized",
)
reasoning_effort = normalize_reasoning_effort_value(str(reasoning_effort), model, custom_llm_provider)
if reasoning_effort == "low":
resolved_effort: Final = normalize_reasoning_effort_value(reasoning_effort, model, custom_llm_provider)
if resolved_effort == "low":
return AnthropicThinkingParam(
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET,
)
elif reasoning_effort == "medium":
elif resolved_effort == "medium":
return AnthropicThinkingParam(
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
)
elif reasoning_effort == "high":
elif resolved_effort == "high":
return AnthropicThinkingParam(
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
)
elif reasoning_effort == "xhigh":
elif resolved_effort == "xhigh":
return AnthropicThinkingParam(
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
)
elif reasoning_effort == "max":
elif resolved_effort == "max":
return AnthropicThinkingParam(
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET,
)
elif reasoning_effort == "minimal":
elif resolved_effort == "minimal":
return AnthropicThinkingParam(
type="enabled",
budget_tokens=max(
@ -1624,7 +1634,11 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
value=effort_value,
llm_provider=self._resolved_provider,
)
optional_params["output_config"] = {"effort": mapped_effort}
optional_params["output_config"] = {
"effort": AnthropicConfig.degrade_alias_effort_for_model(
model, mapped_effort, self._resolved_provider
)
}
elif param == "web_search_options" and isinstance(value, dict):
hosted_web_search_tool = self.map_web_search_tool(cast(OpenAIWebSearchOptions, value))
self._add_tools_to_optional_params(optional_params=optional_params, tools=[hosted_web_search_tool])
@ -2124,20 +2138,11 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
)
gate_error: Final = self._validate_effort_for_model(model, effort, self._resolved_provider)
if gate_error is not None:
if not isinstance(effort, str):
raise litellm.exceptions.BadRequestError(
message=gate_error,
model=model,
llm_provider=self._resolved_provider,
)
normalized_effort: Final = normalize_reasoning_effort_value(effort, model, self._resolved_provider)
if normalized_effort == effort:
raise litellm.exceptions.BadRequestError(
message=gate_error,
model=model,
llm_provider=self._resolved_provider,
)
output_config["effort"] = normalized_effort
raise litellm.exceptions.BadRequestError(
message=gate_error,
model=model,
llm_provider=self._resolved_provider,
)
data["output_config"] = output_config
def _resolve_json_mode_non_streaming(

View file

@ -28,7 +28,6 @@ from ...common_utils import (
strip_advisor_blocks_from_messages,
strip_encrypted_reasoning_blocks_from_anthropic_messages,
)
from ..utils import normalize_reasoning_effort_value
from .mid_conversation_system import (
as_system_content_blocks,
convert_mid_conversation_system_turns,
@ -324,8 +323,8 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
optional_params.setdefault("thinking", fitted_thinking)
if AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider):
requested_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort)
if requested_effort is None:
mapped_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort)
if mapped_effort is None:
raise AnthropicError(
message=(
f"Invalid reasoning_effort: {reasoning_effort!r}. "
@ -335,19 +334,28 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
status_code=400,
)
existing_output_config: Final = optional_params.get("output_config")
existing_mapping: Final = existing_output_config if isinstance(existing_output_config, dict) else {}
raw_explicit_effort: Final = existing_mapping.get("effort")
explicit_effort: Final = raw_explicit_effort if isinstance(raw_explicit_effort, str) else None
candidate_effort: Final = explicit_effort if explicit_effort is not None else requested_effort
gate_error: Final = AnthropicConfig._validate_effort_for_model(model, candidate_effort, custom_llm_provider)
explicit_effort: Final = AnthropicMessagesConfig._explicit_output_config_effort(existing_output_config)
resolved_effort: Final = (
candidate_effort
if gate_error is None
else normalize_reasoning_effort_value(candidate_effort, model, custom_llm_provider)
explicit_effort
if explicit_effort is not None
else AnthropicConfig.degrade_alias_effort_for_model(model, mapped_effort, custom_llm_provider)
)
if gate_error is not None and resolved_effort == candidate_effort:
gate_error: Final = AnthropicConfig._validate_effort_for_model(model, resolved_effort, custom_llm_provider)
if gate_error is not None:
raise AnthropicError(message=gate_error, status_code=400)
optional_params["output_config"] = {**existing_mapping, "effort": resolved_effort}
optional_params["output_config"] = (
{**existing_output_config, "effort": resolved_effort}
if isinstance(existing_output_config, dict)
else {"effort": resolved_effort}
)
@staticmethod
def _explicit_output_config_effort(output_config: object) -> str | None:
match output_config:
case {"effort": str() as effort}:
return effort
case _:
return None
@staticmethod
def _translate_adaptive_effort_for_non_adaptive_model(

View file

@ -74,10 +74,8 @@ def normalize_reasoning_effort_value(
The accepted set is resolved by the same owner that answers ``/model_group/info``, so a level
the proxy advertises is a level this path forwards.
Degradation only happens when the capability set is known and the requested
tier is not in it. A model the map does not describe, or a mapped entry that
declares no effort metadata, keeps the requested value so third-party
Anthropic-compatible deployments are not silently downgraded.
Only a known capability set can refuse a tier: a model the map does not describe, or an entry
declaring no effort metadata, keeps the requested tier instead of being silently downgraded.
A deployment that refuses every step of a chain falls back to an accepted level read off that
same set rather than to an assumed one, since an entry naming its levels outright can exclude

View file

@ -631,7 +631,8 @@ class AmazonAnthropicClaudeMessagesConfig(
@staticmethod
def _clamp_adaptive_reasoning_effort_for_bedrock(model: str, optional_params: dict) -> None:
"""Lower ``reasoning_effort`` to the Bedrock effort ceiling before validation.
"""Lower ``reasoning_effort`` and an explicit ``output_config.effort`` to the Bedrock effort ceiling
before validation.
The shared ``/v1/messages`` effort gate rejects tiers a model does not
natively support (e.g. ``xhigh`` on Opus 4.6). Bedrock's chat paths instead
@ -648,6 +649,14 @@ class AmazonAnthropicClaudeMessagesConfig(
clamped: Final = {"effort": effort}
normalize_bedrock_opus_output_config_effort(model=model, output_config=clamped)
optional_params["reasoning_effort"] = clamped["effort"]
explicit_effort: Final = AnthropicMessagesConfig._explicit_output_config_effort(
optional_params.get("output_config")
)
if explicit_effort is None:
return
clamped_explicit: Final = {"effort": explicit_effort}
normalize_bedrock_opus_output_config_effort(model=model, output_config=clamped_explicit)
optional_params["output_config"] = {**optional_params["output_config"], **clamped_explicit}
def transform_anthropic_messages_request(
self,

View file

@ -158,6 +158,27 @@ def test_bedrock_invoke_messages_clamps_effort_to_ceiling(local_model_cost_map,
assert result["thinking"]["type"] == "adaptive"
def test_bedrock_invoke_messages_clamps_explicit_effort_sent_with_alias(local_model_cost_map):
config = AmazonAnthropicClaudeMessagesConfig()
explicit_output_config = {"effort": "xhigh"}
optional_params = {
"max_tokens": 1024,
"reasoning_effort": "xhigh",
"output_config": explicit_output_config,
}
result = config.transform_anthropic_messages_request(
model="invoke/us.anthropic.claude-opus-4-6-v1",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert result["output_config"]["effort"] == "max"
assert explicit_output_config == {"effort": "xhigh"}
def test_bedrock_invoke_messages_degrades_xhigh_without_ceiling(local_model_cost_map):
config = AmazonAnthropicClaudeMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": "xhigh"}
@ -196,39 +217,44 @@ def test_reasoning_effort_max_accepted_on_sonnet_46_messages(local_model_cost_ma
assert isinstance(output_config, dict) and output_config.get("effort") == "max"
def test_conflicting_unsupported_output_config_effort_is_not_forwarded():
with (
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model",
return_value=True,
),
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model",
side_effect=lambda model, effort, provider: (
None if effort == "high" else f"effort={effort!r} is not supported by this model. Got model: {model}"
),
),
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.utils.get_model_info",
return_value={
"supports_reasoning": True,
"supports_max_reasoning_effort": False,
"supports_xhigh_reasoning_effort": False,
},
),
):
optional_params = {
"reasoning_effort": "high",
"output_config": {"effort": "max"},
}
AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic(
def test_explicit_unsupported_output_config_effort_is_rejected_not_rewritten(local_model_cost_map):
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
"reasoning_effort": "high",
"output_config": {"effort": "xhigh"},
}
with pytest.raises(AnthropicError) as exc_info:
config.transform_anthropic_messages_request(
model="claude-sonnet-4-6",
optional_params=optional_params,
max_tokens=1024,
custom_llm_provider="anthropic",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert optional_params["output_config"]["effort"] != "max"
assert optional_params["output_config"]["effort"] == "high"
assert exc_info.value.status_code == 400
assert "xhigh" in str(exc_info.value)
def test_explicit_supported_output_config_effort_wins_over_unsupported_alias(local_model_cost_map):
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
"reasoning_effort": "xhigh",
"output_config": {"effort": "low"},
}
result = config.transform_anthropic_messages_request(
model="claude-sonnet-4-6",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert result["output_config"] == {"effort": "low"}
def test_explicit_output_config_wins_over_reasoning_effort():

View file

@ -10,7 +10,6 @@ Covers:
import json
import os
from typing import Any
from unittest.mock import patch
import pytest
@ -146,26 +145,39 @@ class TestNormalizeReasoningEffortValue:
def test_a_model_the_map_does_not_describe_keeps_the_requested_tier(self, local_model_cost_map, effort):
assert normalize_reasoning_effort_value(effort, "totally-made-up-model-xyz", "openai") == effort
@staticmethod
def _register_deployment(model_info: dict[str, object]) -> str:
router = litellm.Router(
model_list=[
{
"model_name": "compat",
"litellm_params": {"model": "anthropic/compat-reasoner-1", "api_key": "fake-key"},
"model_info": model_info,
}
]
)
return router.model_list[0]["model_info"]["id"]
@pytest.mark.parametrize("effort", ["max", "xhigh", "minimal"])
def test_a_registered_deployment_without_effort_metadata_keeps_the_requested_tier(
self, local_model_cost_map, effort
):
deployment_id = self._register_deployment({})
assert normalize_reasoning_effort_value(effort, deployment_id, "anthropic") == effort
@pytest.mark.parametrize(
"model_info, effort",
"model_info, effort, expected",
[
({}, "max"),
({}, "xhigh"),
({"supports_reasoning": None}, "max"),
({"supports_reasoning": None}, "xhigh"),
({"supports_reasoning": False}, "max", "high"),
({"supports_reasoning": True, "supports_xhigh_reasoning_effort": False}, "xhigh", "high"),
({"supports_reasoning": True, "supports_minimal_reasoning_effort": False}, "minimal", "low"),
],
)
def test_unknown_effort_metadata_keeps_the_requested_tier(self, model_info, effort):
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.utils.get_model_info", return_value=model_info
):
assert normalize_reasoning_effort_value(effort, "custom-registered-model", "anthropic") == effort
def test_explicit_non_reasoning_still_degrades_to_the_chain_floor(self):
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.utils.get_model_info", return_value={"supports_reasoning": False}
):
assert normalize_reasoning_effort_value("max", "custom-registered-model", "anthropic") == "high"
def test_a_registered_deployment_declaring_a_tier_unsupported_degrades(
self, local_model_cost_map, model_info, effort, expected
):
deployment_id = self._register_deployment(model_info)
assert normalize_reasoning_effort_value(effort, deployment_id, "anthropic") == expected
# ---------------------------------------------------------------------------

View file

@ -290,78 +290,36 @@ class TestMapReasoningEffortDegradation:
assert result["type"] == "adaptive"
class TestApplyOutputConfigDegradation:
def test_max_degrades_to_high_when_unsupported(self):
with (
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model",
return_value="effort='max' is not supported by this model. Got model: test",
),
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model",
return_value=True,
),
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.utils.get_model_info",
return_value=_mock_model_info(
supports_reasoning=True,
supports_max_reasoning_effort=False,
supports_xhigh_reasoning_effort=False,
),
),
):
cfg = AnthropicConfig()
data: dict = {}
optional_params = {"output_config": {"effort": "max"}}
cfg._apply_output_config(data, "test-model", optional_params)
assert data["output_config"]["effort"] == "high"
class TestReasoningEffortAliasOutputConfig:
@staticmethod
def _transform_alias(reasoning_effort: str) -> dict:
config = AnthropicConfig()
optional_params = config.map_openai_params(
non_default_params={"reasoning_effort": reasoning_effort},
optional_params={},
model="claude-sonnet-4-6",
drop_params=False,
)
return config.transform_request(
model="claude-sonnet-4-6",
messages=[{"role": "user", "content": "Hello"}],
optional_params={**optional_params, "max_tokens": 1024},
litellm_params={},
headers={},
)
def test_xhigh_degrades_to_high_when_unsupported(self):
with (
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model",
return_value="effort='xhigh' is not supported by this model. Got model: test",
),
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model",
return_value=True,
),
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.utils.get_model_info",
return_value=_mock_model_info(
supports_reasoning=True,
supports_xhigh_reasoning_effort=False,
),
),
):
cfg = AnthropicConfig()
data: dict = {}
optional_params = {"output_config": {"effort": "xhigh"}}
cfg._apply_output_config(data, "test-model", optional_params)
assert data["output_config"]["effort"] == "high"
def test_unsupported_alias_tier_degrades_to_an_accepted_one(self, local_model_cost_map):
assert self._transform_alias("xhigh")["output_config"] == {"effort": "high"}
def test_max_stays_max_when_supported(self):
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model",
return_value=None,
):
cfg = AnthropicConfig()
data: dict = {}
optional_params = {"output_config": {"effort": "max"}}
cfg._apply_output_config(data, "test-model", optional_params)
assert data["output_config"]["effort"] == "max"
def test_supported_alias_tier_is_kept(self, local_model_cost_map):
assert self._transform_alias("max")["output_config"] == {"effort": "max"}
def test_no_output_config_is_noop(self):
cfg = AnthropicConfig()
data: dict = {}
cfg._apply_output_config(data, "test-model", {})
assert "output_config" not in data
def test_invalid_effort_value_still_raises(self):
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model",
return_value=True,
):
cfg = AnthropicConfig()
with pytest.raises(litellm.exceptions.BadRequestError, match="Invalid effort value"):
cfg._apply_output_config({}, "test-model", {"output_config": {"effort": "bogus"}})
def test_explicit_unsupported_output_config_effort_is_rejected_not_rewritten(self, local_model_cost_map):
with pytest.raises(litellm.exceptions.BadRequestError, match="xhigh"):
AnthropicConfig().transform_request(
model="claude-sonnet-4-6",
messages=[{"role": "user", "content": "Hello"}],
optional_params={"max_tokens": 1024, "output_config": {"effort": "xhigh"}},
litellm_params={},
headers={},
)