fix(anthropic): stop class-attr leak; gate xhigh/max on every route

The reasoning-effort mapping dict was a public class attribute on
AnthropicConfig, so BaseConfig.get_config returned it as a request
parameter and every Anthropic-backed call (Anthropic / Azure / Vertex /
Bedrock Invoke) hit a 400 'REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT:
Extra inputs are not permitted' from the provider. Move the mapping
to a module-level constant.

_supports_effort_level only looked the model up under
custom_llm_provider='anthropic', so bedrock-prefixed model ids
(e.g. bedrock/invoke/us.anthropic.claude-opus-4-7) returned False
for both 'max' and 'xhigh' even when the underlying model entry has
the flag set. Strip known provider prefixes and retry the lookup
against litellm.model_cost directly so per-model gating works on
every route.

Mirror the per-model xhigh/max gate from
AnthropicConfig._apply_output_config in
AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic so
the /v1/messages route also raises a clean 400 instead of forwarding
the unsupported tier.
This commit is contained in:
mateo-berri 2026-05-03 02:29:29 -07:00
parent 47031c08d2
commit 82d7405c6f
5 changed files with 163 additions and 27 deletions

View file

@ -93,6 +93,16 @@ else:
LoggingClass = Any
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT: Dict[str, str] = {
"low": "low",
"minimal": "low",
"medium": "medium",
"high": "high",
"xhigh": "xhigh",
"max": "max",
}
class AnthropicConfig(AnthropicModelInfo, BaseConfig):
"""
Reference: https://docs.anthropic.com/claude/reference/messages_post
@ -108,18 +118,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
metadata: Optional[dict] = None
system: Optional[str] = None
# Shared mapping from OpenAI ``reasoning_effort`` values to Anthropic
# ``output_config.effort`` tier values. Used by both the direct Anthropic
# path and the Bedrock Converse path so the two routes cannot drift.
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT: Dict[str, str] = {
"low": "low",
"minimal": "low",
"medium": "medium",
"high": "high",
"xhigh": "xhigh",
"max": "max",
}
def __init__(
self,
max_tokens: Optional[int] = None,
@ -217,15 +215,54 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
Mirrors the pattern used in ``openai/chat/gpt_5_transformation.py`` so
that adding support for a new effort level is a pure model-map change.
Handles bedrock-prefixed and vertex-prefixed model ids by stripping
the prefix and re-checking against ``litellm.model_cost`` directly,
so a Bedrock-routed Claude 4.6/4.7 keeps its model-map flag.
"""
key = f"supports_{level}_reasoning_effort"
try:
return _supports_factory(
if _supports_factory(
model=model,
custom_llm_provider="anthropic",
key=f"supports_{level}_reasoning_effort",
)
key=key,
):
return True
except Exception:
return False
pass
# Bedrock and Vertex route the model id with a provider-prefix
# (e.g. ``bedrock/invoke/us.anthropic.claude-opus-4-7``). Strip
# known prefixes and look the resulting Anthropic-flavoured key
# up directly in ``litellm.model_cost`` so the lookup keeps
# working regardless of which route the request arrived on.
candidates = [model]
for prefix in (
"bedrock/converse/",
"bedrock/invoke/",
"bedrock/",
"vertex_ai/",
):
if model.startswith(prefix):
candidates.append(model[len(prefix) :])
try:
from litellm.llms.bedrock.common_utils import BedrockModelInfo
base = BedrockModelInfo.get_base_model(model)
if base:
candidates.append(base)
candidates.append(f"bedrock/{base}")
except Exception:
pass
try:
import litellm
for cand in candidates:
if cand in litellm.model_cost and (
litellm.model_cost[cand].get(key) is True
):
return True
except Exception:
pass
return False
def get_supported_openai_params(self, model: str):
params = [
@ -1141,7 +1178,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
# validation with the mapping prevents garbage from
# leaking into ``optional_params`` if ``map_openai_params``
# is ever called without a subsequent ``transform_request``.
mapped_effort = AnthropicConfig.REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
value
)
if mapped_effort is None:

View file

@ -192,7 +192,10 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
- Invalid efforts raise ``BadRequestError`` (clean 400) instead of
surfacing as 500s downstream.
"""
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.chat.transformation import (
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT,
AnthropicConfig,
)
reasoning_effort = optional_params.pop("reasoning_effort", None)
if not isinstance(reasoning_effort, str):
@ -212,10 +215,8 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
optional_params.setdefault("thinking", mapped_thinking)
if AnthropicModelInfo._is_adaptive_thinking_model(model):
mapped_effort = (
AnthropicConfig.REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
reasoning_effort
)
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
reasoning_effort
)
# ``_map_reasoning_effort`` returns ``type=adaptive`` for any
# string on adaptive models without checking the value. The
@ -232,6 +233,33 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
),
status_code=400,
)
# Per-model gating: ``xhigh`` and ``max`` are only valid on
# specific tiers (Opus 4.6/4.7 for max; data-driven for xhigh).
# The chat completion path enforces this via
# ``_apply_output_config``; mirror it here so /v1/messages
# callers see a clean 400 instead of a provider-side error.
if mapped_effort == "max" and not (
AnthropicConfig._is_opus_4_6_model(model)
or AnthropicConfig._is_opus_4_7_model(model)
or AnthropicConfig._supports_effort_level(model, "max")
):
raise AnthropicError(
message=(
f"effort='max' is not supported by this model. "
f"Got model: {model}"
),
status_code=400,
)
if mapped_effort == "xhigh" and not AnthropicConfig._supports_effort_level(
model, "xhigh"
):
raise AnthropicError(
message=(
f"effort='xhigh' is not supported by this model. "
f"Got model: {model}"
),
status_code=400,
)
existing_output_config = optional_params.get("output_config")
if not isinstance(existing_output_config, dict):
existing_output_config = {}

View file

@ -31,7 +31,10 @@ from litellm.litellm_core_utils.prompt_templates.factory import (
_bedrock_converse_messages_pt,
_bedrock_tools_pt,
)
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.chat.transformation import (
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT,
AnthropicConfig,
)
from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException
from litellm.types.llms.bedrock import *
from litellm.types.llms.openai import (
@ -492,10 +495,8 @@ class AmazonConverseConfig(BaseConfig):
# catch it, but only because validation happens to run).
# Matches the /v1/messages pattern where validation is
# co-located with the mapping.
mapped_effort = (
AnthropicConfig.REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
reasoning_effort
)
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
reasoning_effort
)
if mapped_effort is None:
raise litellm.exceptions.BadRequestError(

View file

@ -1959,6 +1959,45 @@ def test_get_config_without_model_uses_fallback():
assert config["max_tokens"] == 4096
def test_get_config_does_not_leak_module_constants():
"""``BaseConfig.get_config`` returns class attributes; the
reasoning-effort mapping must not be one of them or it ends up
serialised onto the wire as an extra request key.
"""
cfg = AnthropicConfig.get_config(model="claude-opus-4-7")
for forbidden in (
"REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT",
"_REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT",
):
assert forbidden not in cfg
@pytest.mark.parametrize(
"model,level,expected",
[
("claude-opus-4-7", "max", True),
("claude-opus-4-7", "xhigh", True),
("claude-opus-4-6", "max", True),
("claude-opus-4-6", "xhigh", False),
("claude-sonnet-4-6", "max", False),
("claude-sonnet-4-6", "xhigh", False),
("bedrock/invoke/us.anthropic.claude-opus-4-7", "max", True),
("bedrock/invoke/us.anthropic.claude-opus-4-7", "xhigh", True),
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "max", True),
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh", False),
("bedrock/invoke/us.anthropic.claude-sonnet-4-6", "max", False),
("vertex_ai/claude-opus-4-7", "xhigh", True),
("azure_ai/claude-opus-4-7", "xhigh", True),
],
)
def test_supports_effort_level_handles_provider_prefixes(model, level, expected):
"""``_supports_effort_level`` must handle bedrock/ vertex_ai/ azure_ai/
prefixed model ids so per-model gating works on every route, not just
the bare-Anthropic chat completion path.
"""
assert AnthropicConfig._supports_effort_level(model, level) is expected
def test_transform_request_uses_dynamic_max_tokens():
"""
Test that transform_request uses dynamic max_tokens based on model

View file

@ -129,6 +129,37 @@ def test_invalid_reasoning_effort_raises_400(bad_effort):
assert exc_info.value.status_code == 400
@pytest.mark.parametrize(
"model,bad_effort",
[
("claude-opus-4-6", "xhigh"),
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh"),
("claude-sonnet-4-6", "xhigh"),
("claude-sonnet-4-6", "max"),
("bedrock/invoke/us.anthropic.claude-sonnet-4-6", "max"),
],
)
def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort):
"""``xhigh`` and ``max`` are gated per-model. The /v1/messages route must
surface a clean 400 client-side instead of forwarding the unsupported
tier and letting the provider 500/400 it.
"""
config = AnthropicMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort}
with pytest.raises(AnthropicError) as exc_info:
config.transform_anthropic_messages_request(
model=model,
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=optional_params,
litellm_params={},
headers={},
)
assert exc_info.value.status_code == 400
assert "not supported by this model" in str(exc_info.value)
def test_explicit_output_config_wins_over_reasoning_effort():
"""
Explicit native ``output_config.effort`` is never overridden by the