diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 49f0543b7e1..d800a865832 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -93,6 +93,16 @@ else: LoggingClass = Any +REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT: Dict[str, str] = { + "low": "low", + "minimal": "low", + "medium": "medium", + "high": "high", + "xhigh": "xhigh", + "max": "max", +} + + class AnthropicConfig(AnthropicModelInfo, BaseConfig): """ Reference: https://docs.anthropic.com/claude/reference/messages_post @@ -108,18 +118,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): metadata: Optional[dict] = None system: Optional[str] = None - # Shared mapping from OpenAI ``reasoning_effort`` values to Anthropic - # ``output_config.effort`` tier values. Used by both the direct Anthropic - # path and the Bedrock Converse path so the two routes cannot drift. - REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT: Dict[str, str] = { - "low": "low", - "minimal": "low", - "medium": "medium", - "high": "high", - "xhigh": "xhigh", - "max": "max", - } - def __init__( self, max_tokens: Optional[int] = None, @@ -217,15 +215,54 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): Mirrors the pattern used in ``openai/chat/gpt_5_transformation.py`` so that adding support for a new effort level is a pure model-map change. + Handles bedrock-prefixed and vertex-prefixed model ids by stripping + the prefix and re-checking against ``litellm.model_cost`` directly, + so a Bedrock-routed Claude 4.6/4.7 keeps its model-map flag. """ + key = f"supports_{level}_reasoning_effort" try: - return _supports_factory( + if _supports_factory( model=model, custom_llm_provider="anthropic", - key=f"supports_{level}_reasoning_effort", - ) + key=key, + ): + return True except Exception: - return False + pass + # Bedrock and Vertex route the model id with a provider-prefix + # (e.g. ``bedrock/invoke/us.anthropic.claude-opus-4-7``). Strip + # known prefixes and look the resulting Anthropic-flavoured key + # up directly in ``litellm.model_cost`` so the lookup keeps + # working regardless of which route the request arrived on. + candidates = [model] + for prefix in ( + "bedrock/converse/", + "bedrock/invoke/", + "bedrock/", + "vertex_ai/", + ): + if model.startswith(prefix): + candidates.append(model[len(prefix) :]) + try: + from litellm.llms.bedrock.common_utils import BedrockModelInfo + + base = BedrockModelInfo.get_base_model(model) + if base: + candidates.append(base) + candidates.append(f"bedrock/{base}") + except Exception: + pass + try: + import litellm + + for cand in candidates: + if cand in litellm.model_cost and ( + litellm.model_cost[cand].get(key) is True + ): + return True + except Exception: + pass + return False def get_supported_openai_params(self, model: str): params = [ @@ -1141,7 +1178,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): # validation with the mapping prevents garbage from # leaking into ``optional_params`` if ``map_openai_params`` # is ever called without a subsequent ``transform_request``. - mapped_effort = AnthropicConfig.REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( + mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( value ) if mapped_effort is None: diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index b737bf9aaff..98004f647ef 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -192,7 +192,10 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): - Invalid efforts raise ``BadRequestError`` (clean 400) instead of surfacing as 500s downstream. """ - from litellm.llms.anthropic.chat.transformation import AnthropicConfig + from litellm.llms.anthropic.chat.transformation import ( + REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT, + AnthropicConfig, + ) reasoning_effort = optional_params.pop("reasoning_effort", None) if not isinstance(reasoning_effort, str): @@ -212,10 +215,8 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): optional_params.setdefault("thinking", mapped_thinking) if AnthropicModelInfo._is_adaptive_thinking_model(model): - mapped_effort = ( - AnthropicConfig.REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( - reasoning_effort - ) + mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( + reasoning_effort ) # ``_map_reasoning_effort`` returns ``type=adaptive`` for any # string on adaptive models without checking the value. The @@ -232,6 +233,33 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): ), status_code=400, ) + # Per-model gating: ``xhigh`` and ``max`` are only valid on + # specific tiers (Opus 4.6/4.7 for max; data-driven for xhigh). + # The chat completion path enforces this via + # ``_apply_output_config``; mirror it here so /v1/messages + # callers see a clean 400 instead of a provider-side error. + if mapped_effort == "max" and not ( + AnthropicConfig._is_opus_4_6_model(model) + or AnthropicConfig._is_opus_4_7_model(model) + or AnthropicConfig._supports_effort_level(model, "max") + ): + raise AnthropicError( + message=( + f"effort='max' is not supported by this model. " + f"Got model: {model}" + ), + status_code=400, + ) + if mapped_effort == "xhigh" and not AnthropicConfig._supports_effort_level( + model, "xhigh" + ): + raise AnthropicError( + message=( + f"effort='xhigh' is not supported by this model. " + f"Got model: {model}" + ), + status_code=400, + ) existing_output_config = optional_params.get("output_config") if not isinstance(existing_output_config, dict): existing_output_config = {} diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 6f9e8de62e5..92fef67a601 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -31,7 +31,10 @@ from litellm.litellm_core_utils.prompt_templates.factory import ( _bedrock_converse_messages_pt, _bedrock_tools_pt, ) -from litellm.llms.anthropic.chat.transformation import AnthropicConfig +from litellm.llms.anthropic.chat.transformation import ( + REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT, + AnthropicConfig, +) from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException from litellm.types.llms.bedrock import * from litellm.types.llms.openai import ( @@ -492,10 +495,8 @@ class AmazonConverseConfig(BaseConfig): # catch it, but only because validation happens to run). # Matches the /v1/messages pattern where validation is # co-located with the mapping. - mapped_effort = ( - AnthropicConfig.REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( - reasoning_effort - ) + mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( + reasoning_effort ) if mapped_effort is None: raise litellm.exceptions.BadRequestError( diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py index 0381b850d65..da4684ec9c9 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -1959,6 +1959,45 @@ def test_get_config_without_model_uses_fallback(): assert config["max_tokens"] == 4096 +def test_get_config_does_not_leak_module_constants(): + """``BaseConfig.get_config`` returns class attributes; the + reasoning-effort mapping must not be one of them or it ends up + serialised onto the wire as an extra request key. + """ + cfg = AnthropicConfig.get_config(model="claude-opus-4-7") + for forbidden in ( + "REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT", + "_REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT", + ): + assert forbidden not in cfg + + +@pytest.mark.parametrize( + "model,level,expected", + [ + ("claude-opus-4-7", "max", True), + ("claude-opus-4-7", "xhigh", True), + ("claude-opus-4-6", "max", True), + ("claude-opus-4-6", "xhigh", False), + ("claude-sonnet-4-6", "max", False), + ("claude-sonnet-4-6", "xhigh", False), + ("bedrock/invoke/us.anthropic.claude-opus-4-7", "max", True), + ("bedrock/invoke/us.anthropic.claude-opus-4-7", "xhigh", True), + ("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "max", True), + ("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh", False), + ("bedrock/invoke/us.anthropic.claude-sonnet-4-6", "max", False), + ("vertex_ai/claude-opus-4-7", "xhigh", True), + ("azure_ai/claude-opus-4-7", "xhigh", True), + ], +) +def test_supports_effort_level_handles_provider_prefixes(model, level, expected): + """``_supports_effort_level`` must handle bedrock/ vertex_ai/ azure_ai/ + prefixed model ids so per-model gating works on every route, not just + the bare-Anthropic chat completion path. + """ + assert AnthropicConfig._supports_effort_level(model, level) is expected + + def test_transform_request_uses_dynamic_max_tokens(): """ Test that transform_request uses dynamic max_tokens based on model diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_reasoning_effort_translation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_reasoning_effort_translation.py index a2fcf1595bc..900bb15545a 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_reasoning_effort_translation.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_reasoning_effort_translation.py @@ -129,6 +129,37 @@ def test_invalid_reasoning_effort_raises_400(bad_effort): assert exc_info.value.status_code == 400 +@pytest.mark.parametrize( + "model,bad_effort", + [ + ("claude-opus-4-6", "xhigh"), + ("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh"), + ("claude-sonnet-4-6", "xhigh"), + ("claude-sonnet-4-6", "max"), + ("bedrock/invoke/us.anthropic.claude-sonnet-4-6", "max"), + ], +) +def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort): + """``xhigh`` and ``max`` are gated per-model. The /v1/messages route must + surface a clean 400 client-side instead of forwarding the unsupported + tier and letting the provider 500/400 it. + """ + config = AnthropicMessagesConfig() + optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort} + + with pytest.raises(AnthropicError) as exc_info: + config.transform_anthropic_messages_request( + model=model, + messages=[{"role": "user", "content": "Hello"}], + anthropic_messages_optional_request_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert exc_info.value.status_code == 400 + assert "not supported by this model" in str(exc_info.value) + + def test_explicit_output_config_wins_over_reasoning_effort(): """ Explicit native ``output_config.effort`` is never overridden by the