mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(anthropic): stop class-attr leak; gate xhigh/max on every route
The reasoning-effort mapping dict was a public class attribute on AnthropicConfig, so BaseConfig.get_config returned it as a request parameter and every Anthropic-backed call (Anthropic / Azure / Vertex / Bedrock Invoke) hit a 400 'REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT: Extra inputs are not permitted' from the provider. Move the mapping to a module-level constant. _supports_effort_level only looked the model up under custom_llm_provider='anthropic', so bedrock-prefixed model ids (e.g. bedrock/invoke/us.anthropic.claude-opus-4-7) returned False for both 'max' and 'xhigh' even when the underlying model entry has the flag set. Strip known provider prefixes and retry the lookup against litellm.model_cost directly so per-model gating works on every route. Mirror the per-model xhigh/max gate from AnthropicConfig._apply_output_config in AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic so the /v1/messages route also raises a clean 400 instead of forwarding the unsupported tier.
This commit is contained in:
parent
47031c08d2
commit
82d7405c6f
5 changed files with 163 additions and 27 deletions
|
|
@ -93,6 +93,16 @@ else:
|
|||
LoggingClass = Any
|
||||
|
||||
|
||||
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT: Dict[str, str] = {
|
||||
"low": "low",
|
||||
"minimal": "low",
|
||||
"medium": "medium",
|
||||
"high": "high",
|
||||
"xhigh": "xhigh",
|
||||
"max": "max",
|
||||
}
|
||||
|
||||
|
||||
class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
||||
"""
|
||||
Reference: https://docs.anthropic.com/claude/reference/messages_post
|
||||
|
|
@ -108,18 +118,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
metadata: Optional[dict] = None
|
||||
system: Optional[str] = None
|
||||
|
||||
# Shared mapping from OpenAI ``reasoning_effort`` values to Anthropic
|
||||
# ``output_config.effort`` tier values. Used by both the direct Anthropic
|
||||
# path and the Bedrock Converse path so the two routes cannot drift.
|
||||
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT: Dict[str, str] = {
|
||||
"low": "low",
|
||||
"minimal": "low",
|
||||
"medium": "medium",
|
||||
"high": "high",
|
||||
"xhigh": "xhigh",
|
||||
"max": "max",
|
||||
}
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
max_tokens: Optional[int] = None,
|
||||
|
|
@ -217,15 +215,54 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
|
||||
Mirrors the pattern used in ``openai/chat/gpt_5_transformation.py`` so
|
||||
that adding support for a new effort level is a pure model-map change.
|
||||
Handles bedrock-prefixed and vertex-prefixed model ids by stripping
|
||||
the prefix and re-checking against ``litellm.model_cost`` directly,
|
||||
so a Bedrock-routed Claude 4.6/4.7 keeps its model-map flag.
|
||||
"""
|
||||
key = f"supports_{level}_reasoning_effort"
|
||||
try:
|
||||
return _supports_factory(
|
||||
if _supports_factory(
|
||||
model=model,
|
||||
custom_llm_provider="anthropic",
|
||||
key=f"supports_{level}_reasoning_effort",
|
||||
)
|
||||
key=key,
|
||||
):
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
pass
|
||||
# Bedrock and Vertex route the model id with a provider-prefix
|
||||
# (e.g. ``bedrock/invoke/us.anthropic.claude-opus-4-7``). Strip
|
||||
# known prefixes and look the resulting Anthropic-flavoured key
|
||||
# up directly in ``litellm.model_cost`` so the lookup keeps
|
||||
# working regardless of which route the request arrived on.
|
||||
candidates = [model]
|
||||
for prefix in (
|
||||
"bedrock/converse/",
|
||||
"bedrock/invoke/",
|
||||
"bedrock/",
|
||||
"vertex_ai/",
|
||||
):
|
||||
if model.startswith(prefix):
|
||||
candidates.append(model[len(prefix) :])
|
||||
try:
|
||||
from litellm.llms.bedrock.common_utils import BedrockModelInfo
|
||||
|
||||
base = BedrockModelInfo.get_base_model(model)
|
||||
if base:
|
||||
candidates.append(base)
|
||||
candidates.append(f"bedrock/{base}")
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
import litellm
|
||||
|
||||
for cand in candidates:
|
||||
if cand in litellm.model_cost and (
|
||||
litellm.model_cost[cand].get(key) is True
|
||||
):
|
||||
return True
|
||||
except Exception:
|
||||
pass
|
||||
return False
|
||||
|
||||
def get_supported_openai_params(self, model: str):
|
||||
params = [
|
||||
|
|
@ -1141,7 +1178,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
# validation with the mapping prevents garbage from
|
||||
# leaking into ``optional_params`` if ``map_openai_params``
|
||||
# is ever called without a subsequent ``transform_request``.
|
||||
mapped_effort = AnthropicConfig.REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
|
||||
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
|
||||
value
|
||||
)
|
||||
if mapped_effort is None:
|
||||
|
|
|
|||
|
|
@ -192,7 +192,10 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
- Invalid efforts raise ``BadRequestError`` (clean 400) instead of
|
||||
surfacing as 500s downstream.
|
||||
"""
|
||||
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
||||
from litellm.llms.anthropic.chat.transformation import (
|
||||
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT,
|
||||
AnthropicConfig,
|
||||
)
|
||||
|
||||
reasoning_effort = optional_params.pop("reasoning_effort", None)
|
||||
if not isinstance(reasoning_effort, str):
|
||||
|
|
@ -212,10 +215,8 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
|
||||
optional_params.setdefault("thinking", mapped_thinking)
|
||||
if AnthropicModelInfo._is_adaptive_thinking_model(model):
|
||||
mapped_effort = (
|
||||
AnthropicConfig.REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
|
||||
reasoning_effort
|
||||
)
|
||||
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
|
||||
reasoning_effort
|
||||
)
|
||||
# ``_map_reasoning_effort`` returns ``type=adaptive`` for any
|
||||
# string on adaptive models without checking the value. The
|
||||
|
|
@ -232,6 +233,33 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
),
|
||||
status_code=400,
|
||||
)
|
||||
# Per-model gating: ``xhigh`` and ``max`` are only valid on
|
||||
# specific tiers (Opus 4.6/4.7 for max; data-driven for xhigh).
|
||||
# The chat completion path enforces this via
|
||||
# ``_apply_output_config``; mirror it here so /v1/messages
|
||||
# callers see a clean 400 instead of a provider-side error.
|
||||
if mapped_effort == "max" and not (
|
||||
AnthropicConfig._is_opus_4_6_model(model)
|
||||
or AnthropicConfig._is_opus_4_7_model(model)
|
||||
or AnthropicConfig._supports_effort_level(model, "max")
|
||||
):
|
||||
raise AnthropicError(
|
||||
message=(
|
||||
f"effort='max' is not supported by this model. "
|
||||
f"Got model: {model}"
|
||||
),
|
||||
status_code=400,
|
||||
)
|
||||
if mapped_effort == "xhigh" and not AnthropicConfig._supports_effort_level(
|
||||
model, "xhigh"
|
||||
):
|
||||
raise AnthropicError(
|
||||
message=(
|
||||
f"effort='xhigh' is not supported by this model. "
|
||||
f"Got model: {model}"
|
||||
),
|
||||
status_code=400,
|
||||
)
|
||||
existing_output_config = optional_params.get("output_config")
|
||||
if not isinstance(existing_output_config, dict):
|
||||
existing_output_config = {}
|
||||
|
|
|
|||
|
|
@ -31,7 +31,10 @@ from litellm.litellm_core_utils.prompt_templates.factory import (
|
|||
_bedrock_converse_messages_pt,
|
||||
_bedrock_tools_pt,
|
||||
)
|
||||
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
||||
from litellm.llms.anthropic.chat.transformation import (
|
||||
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT,
|
||||
AnthropicConfig,
|
||||
)
|
||||
from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException
|
||||
from litellm.types.llms.bedrock import *
|
||||
from litellm.types.llms.openai import (
|
||||
|
|
@ -492,10 +495,8 @@ class AmazonConverseConfig(BaseConfig):
|
|||
# catch it, but only because validation happens to run).
|
||||
# Matches the /v1/messages pattern where validation is
|
||||
# co-located with the mapping.
|
||||
mapped_effort = (
|
||||
AnthropicConfig.REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
|
||||
reasoning_effort
|
||||
)
|
||||
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
|
||||
reasoning_effort
|
||||
)
|
||||
if mapped_effort is None:
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
|
|
|
|||
|
|
@ -1959,6 +1959,45 @@ def test_get_config_without_model_uses_fallback():
|
|||
assert config["max_tokens"] == 4096
|
||||
|
||||
|
||||
def test_get_config_does_not_leak_module_constants():
|
||||
"""``BaseConfig.get_config`` returns class attributes; the
|
||||
reasoning-effort mapping must not be one of them or it ends up
|
||||
serialised onto the wire as an extra request key.
|
||||
"""
|
||||
cfg = AnthropicConfig.get_config(model="claude-opus-4-7")
|
||||
for forbidden in (
|
||||
"REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT",
|
||||
"_REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT",
|
||||
):
|
||||
assert forbidden not in cfg
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,level,expected",
|
||||
[
|
||||
("claude-opus-4-7", "max", True),
|
||||
("claude-opus-4-7", "xhigh", True),
|
||||
("claude-opus-4-6", "max", True),
|
||||
("claude-opus-4-6", "xhigh", False),
|
||||
("claude-sonnet-4-6", "max", False),
|
||||
("claude-sonnet-4-6", "xhigh", False),
|
||||
("bedrock/invoke/us.anthropic.claude-opus-4-7", "max", True),
|
||||
("bedrock/invoke/us.anthropic.claude-opus-4-7", "xhigh", True),
|
||||
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "max", True),
|
||||
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh", False),
|
||||
("bedrock/invoke/us.anthropic.claude-sonnet-4-6", "max", False),
|
||||
("vertex_ai/claude-opus-4-7", "xhigh", True),
|
||||
("azure_ai/claude-opus-4-7", "xhigh", True),
|
||||
],
|
||||
)
|
||||
def test_supports_effort_level_handles_provider_prefixes(model, level, expected):
|
||||
"""``_supports_effort_level`` must handle bedrock/ vertex_ai/ azure_ai/
|
||||
prefixed model ids so per-model gating works on every route, not just
|
||||
the bare-Anthropic chat completion path.
|
||||
"""
|
||||
assert AnthropicConfig._supports_effort_level(model, level) is expected
|
||||
|
||||
|
||||
def test_transform_request_uses_dynamic_max_tokens():
|
||||
"""
|
||||
Test that transform_request uses dynamic max_tokens based on model
|
||||
|
|
|
|||
|
|
@ -129,6 +129,37 @@ def test_invalid_reasoning_effort_raises_400(bad_effort):
|
|||
assert exc_info.value.status_code == 400
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,bad_effort",
|
||||
[
|
||||
("claude-opus-4-6", "xhigh"),
|
||||
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh"),
|
||||
("claude-sonnet-4-6", "xhigh"),
|
||||
("claude-sonnet-4-6", "max"),
|
||||
("bedrock/invoke/us.anthropic.claude-sonnet-4-6", "max"),
|
||||
],
|
||||
)
|
||||
def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort):
|
||||
"""``xhigh`` and ``max`` are gated per-model. The /v1/messages route must
|
||||
surface a clean 400 client-side instead of forwarding the unsupported
|
||||
tier and letting the provider 500/400 it.
|
||||
"""
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort}
|
||||
|
||||
with pytest.raises(AnthropicError) as exc_info:
|
||||
config.transform_anthropic_messages_request(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "not supported by this model" in str(exc_info.value)
|
||||
|
||||
|
||||
def test_explicit_output_config_wins_over_reasoning_effort():
|
||||
"""
|
||||
Explicit native ``output_config.effort`` is never overridden by the
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue