diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index ecf400b9787..7860e0998dd 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -415,6 +415,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): return f"effort='xhigh' is not supported by this model. Got model: {model}" return None + @staticmethod + def degrade_alias_effort_for_model(model: str, effort: str, custom_llm_provider: str) -> str: + """Keep an alias-derived effort the gate accepts, else lower it to a tier the model is known to accept. + + Explicit ``output_config.effort`` must not be routed here: a caller naming a native tier gets a 400. + """ + if AnthropicConfig._validate_effort_for_model(model, effort, custom_llm_provider) is None: + return effort + return normalize_reasoning_effort_value(effort, model, custom_llm_provider) + @staticmethod def _model_supports_effort_param(model: str, custom_llm_provider: str) -> bool: """Whether the model accepts ``output_config.effort`` at all. @@ -1265,33 +1275,33 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): type="adaptive", display="summarized", ) - reasoning_effort = normalize_reasoning_effort_value(str(reasoning_effort), model, custom_llm_provider) - if reasoning_effort == "low": + resolved_effort: Final = normalize_reasoning_effort_value(reasoning_effort, model, custom_llm_provider) + if resolved_effort == "low": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET, ) - elif reasoning_effort == "medium": + elif resolved_effort == "medium": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, ) - elif reasoning_effort == "high": + elif resolved_effort == "high": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, ) - elif reasoning_effort == "xhigh": + elif resolved_effort == "xhigh": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, ) - elif reasoning_effort == "max": + elif resolved_effort == "max": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET, ) - elif reasoning_effort == "minimal": + elif resolved_effort == "minimal": return AnthropicThinkingParam( type="enabled", budget_tokens=max( @@ -1624,7 +1634,11 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): value=effort_value, llm_provider=self._resolved_provider, ) - optional_params["output_config"] = {"effort": mapped_effort} + optional_params["output_config"] = { + "effort": AnthropicConfig.degrade_alias_effort_for_model( + model, mapped_effort, self._resolved_provider + ) + } elif param == "web_search_options" and isinstance(value, dict): hosted_web_search_tool = self.map_web_search_tool(cast(OpenAIWebSearchOptions, value)) self._add_tools_to_optional_params(optional_params=optional_params, tools=[hosted_web_search_tool]) @@ -2124,20 +2138,11 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): ) gate_error: Final = self._validate_effort_for_model(model, effort, self._resolved_provider) if gate_error is not None: - if not isinstance(effort, str): - raise litellm.exceptions.BadRequestError( - message=gate_error, - model=model, - llm_provider=self._resolved_provider, - ) - normalized_effort: Final = normalize_reasoning_effort_value(effort, model, self._resolved_provider) - if normalized_effort == effort: - raise litellm.exceptions.BadRequestError( - message=gate_error, - model=model, - llm_provider=self._resolved_provider, - ) - output_config["effort"] = normalized_effort + raise litellm.exceptions.BadRequestError( + message=gate_error, + model=model, + llm_provider=self._resolved_provider, + ) data["output_config"] = output_config def _resolve_json_mode_non_streaming( diff --git a/litellm/llms/anthropic/pass_through/messages/transformation.py b/litellm/llms/anthropic/pass_through/messages/transformation.py index 5ad78e2eabb..8dd31eb32a9 100644 --- a/litellm/llms/anthropic/pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/pass_through/messages/transformation.py @@ -28,7 +28,6 @@ from ...common_utils import ( strip_advisor_blocks_from_messages, strip_encrypted_reasoning_blocks_from_anthropic_messages, ) -from ..utils import normalize_reasoning_effort_value from .mid_conversation_system import ( as_system_content_blocks, convert_mid_conversation_system_turns, @@ -324,8 +323,8 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): optional_params.setdefault("thinking", fitted_thinking) if AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider): - requested_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort) - if requested_effort is None: + mapped_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort) + if mapped_effort is None: raise AnthropicError( message=( f"Invalid reasoning_effort: {reasoning_effort!r}. " @@ -335,19 +334,28 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): status_code=400, ) existing_output_config: Final = optional_params.get("output_config") - existing_mapping: Final = existing_output_config if isinstance(existing_output_config, dict) else {} - raw_explicit_effort: Final = existing_mapping.get("effort") - explicit_effort: Final = raw_explicit_effort if isinstance(raw_explicit_effort, str) else None - candidate_effort: Final = explicit_effort if explicit_effort is not None else requested_effort - gate_error: Final = AnthropicConfig._validate_effort_for_model(model, candidate_effort, custom_llm_provider) + explicit_effort: Final = AnthropicMessagesConfig._explicit_output_config_effort(existing_output_config) resolved_effort: Final = ( - candidate_effort - if gate_error is None - else normalize_reasoning_effort_value(candidate_effort, model, custom_llm_provider) + explicit_effort + if explicit_effort is not None + else AnthropicConfig.degrade_alias_effort_for_model(model, mapped_effort, custom_llm_provider) ) - if gate_error is not None and resolved_effort == candidate_effort: + gate_error: Final = AnthropicConfig._validate_effort_for_model(model, resolved_effort, custom_llm_provider) + if gate_error is not None: raise AnthropicError(message=gate_error, status_code=400) - optional_params["output_config"] = {**existing_mapping, "effort": resolved_effort} + optional_params["output_config"] = ( + {**existing_output_config, "effort": resolved_effort} + if isinstance(existing_output_config, dict) + else {"effort": resolved_effort} + ) + + @staticmethod + def _explicit_output_config_effort(output_config: object) -> str | None: + match output_config: + case {"effort": str() as effort}: + return effort + case _: + return None @staticmethod def _translate_adaptive_effort_for_non_adaptive_model( diff --git a/litellm/llms/anthropic/pass_through/utils.py b/litellm/llms/anthropic/pass_through/utils.py index f14c6812acd..f64f411b69e 100644 --- a/litellm/llms/anthropic/pass_through/utils.py +++ b/litellm/llms/anthropic/pass_through/utils.py @@ -74,10 +74,8 @@ def normalize_reasoning_effort_value( The accepted set is resolved by the same owner that answers ``/model_group/info``, so a level the proxy advertises is a level this path forwards. - Degradation only happens when the capability set is known and the requested - tier is not in it. A model the map does not describe, or a mapped entry that - declares no effort metadata, keeps the requested value so third-party - Anthropic-compatible deployments are not silently downgraded. + Only a known capability set can refuse a tier: a model the map does not describe, or an entry + declaring no effort metadata, keeps the requested tier instead of being silently downgraded. A deployment that refuses every step of a chain falls back to an accepted level read off that same set rather than to an assumed one, since an entry naming its levels outright can exclude diff --git a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py index 94eb0c92e40..948f989fc3d 100644 --- a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py @@ -631,7 +631,8 @@ class AmazonAnthropicClaudeMessagesConfig( @staticmethod def _clamp_adaptive_reasoning_effort_for_bedrock(model: str, optional_params: dict) -> None: - """Lower ``reasoning_effort`` to the Bedrock effort ceiling before validation. + """Lower ``reasoning_effort`` and an explicit ``output_config.effort`` to the Bedrock effort ceiling + before validation. The shared ``/v1/messages`` effort gate rejects tiers a model does not natively support (e.g. ``xhigh`` on Opus 4.6). Bedrock's chat paths instead @@ -648,6 +649,14 @@ class AmazonAnthropicClaudeMessagesConfig( clamped: Final = {"effort": effort} normalize_bedrock_opus_output_config_effort(model=model, output_config=clamped) optional_params["reasoning_effort"] = clamped["effort"] + explicit_effort: Final = AnthropicMessagesConfig._explicit_output_config_effort( + optional_params.get("output_config") + ) + if explicit_effort is None: + return + clamped_explicit: Final = {"effort": explicit_effort} + normalize_bedrock_opus_output_config_effort(model=model, output_config=clamped_explicit) + optional_params["output_config"] = {**optional_params["output_config"], **clamped_explicit} def transform_anthropic_messages_request( self, diff --git a/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py b/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py index d6ed5ec0971..eee937bd83a 100644 --- a/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py +++ b/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py @@ -158,6 +158,27 @@ def test_bedrock_invoke_messages_clamps_effort_to_ceiling(local_model_cost_map, assert result["thinking"]["type"] == "adaptive" +def test_bedrock_invoke_messages_clamps_explicit_effort_sent_with_alias(local_model_cost_map): + config = AmazonAnthropicClaudeMessagesConfig() + explicit_output_config = {"effort": "xhigh"} + optional_params = { + "max_tokens": 1024, + "reasoning_effort": "xhigh", + "output_config": explicit_output_config, + } + + result = config.transform_anthropic_messages_request( + model="invoke/us.anthropic.claude-opus-4-6-v1", + messages=[{"role": "user", "content": "Hello"}], + anthropic_messages_optional_request_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert result["output_config"]["effort"] == "max" + assert explicit_output_config == {"effort": "xhigh"} + + def test_bedrock_invoke_messages_degrades_xhigh_without_ceiling(local_model_cost_map): config = AmazonAnthropicClaudeMessagesConfig() optional_params = {"max_tokens": 1024, "reasoning_effort": "xhigh"} @@ -196,39 +217,44 @@ def test_reasoning_effort_max_accepted_on_sonnet_46_messages(local_model_cost_ma assert isinstance(output_config, dict) and output_config.get("effort") == "max" -def test_conflicting_unsupported_output_config_effort_is_not_forwarded(): - with ( - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model", - return_value=True, - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", - side_effect=lambda model, effort, provider: ( - None if effort == "high" else f"effort={effort!r} is not supported by this model. Got model: {model}" - ), - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.utils.get_model_info", - return_value={ - "supports_reasoning": True, - "supports_max_reasoning_effort": False, - "supports_xhigh_reasoning_effort": False, - }, - ), - ): - optional_params = { - "reasoning_effort": "high", - "output_config": {"effort": "max"}, - } - AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic( +def test_explicit_unsupported_output_config_effort_is_rejected_not_rewritten(local_model_cost_map): + config = AnthropicMessagesConfig() + optional_params = { + "max_tokens": 1024, + "reasoning_effort": "high", + "output_config": {"effort": "xhigh"}, + } + + with pytest.raises(AnthropicError) as exc_info: + config.transform_anthropic_messages_request( model="claude-sonnet-4-6", - optional_params=optional_params, - max_tokens=1024, - custom_llm_provider="anthropic", + messages=[{"role": "user", "content": "Hello"}], + anthropic_messages_optional_request_params=optional_params, + litellm_params={}, + headers={}, ) - assert optional_params["output_config"]["effort"] != "max" - assert optional_params["output_config"]["effort"] == "high" + + assert exc_info.value.status_code == 400 + assert "xhigh" in str(exc_info.value) + + +def test_explicit_supported_output_config_effort_wins_over_unsupported_alias(local_model_cost_map): + config = AnthropicMessagesConfig() + optional_params = { + "max_tokens": 1024, + "reasoning_effort": "xhigh", + "output_config": {"effort": "low"}, + } + + result = config.transform_anthropic_messages_request( + model="claude-sonnet-4-6", + messages=[{"role": "user", "content": "Hello"}], + anthropic_messages_optional_request_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert result["output_config"] == {"effort": "low"} def test_explicit_output_config_wins_over_reasoning_effort(): diff --git a/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py b/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py index 8dfd20fcca4..068bf0edc3b 100644 --- a/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py +++ b/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py @@ -10,7 +10,6 @@ Covers: import json import os from typing import Any -from unittest.mock import patch import pytest @@ -146,26 +145,39 @@ class TestNormalizeReasoningEffortValue: def test_a_model_the_map_does_not_describe_keeps_the_requested_tier(self, local_model_cost_map, effort): assert normalize_reasoning_effort_value(effort, "totally-made-up-model-xyz", "openai") == effort + @staticmethod + def _register_deployment(model_info: dict[str, object]) -> str: + router = litellm.Router( + model_list=[ + { + "model_name": "compat", + "litellm_params": {"model": "anthropic/compat-reasoner-1", "api_key": "fake-key"}, + "model_info": model_info, + } + ] + ) + return router.model_list[0]["model_info"]["id"] + + @pytest.mark.parametrize("effort", ["max", "xhigh", "minimal"]) + def test_a_registered_deployment_without_effort_metadata_keeps_the_requested_tier( + self, local_model_cost_map, effort + ): + deployment_id = self._register_deployment({}) + assert normalize_reasoning_effort_value(effort, deployment_id, "anthropic") == effort + @pytest.mark.parametrize( - "model_info, effort", + "model_info, effort, expected", [ - ({}, "max"), - ({}, "xhigh"), - ({"supports_reasoning": None}, "max"), - ({"supports_reasoning": None}, "xhigh"), + ({"supports_reasoning": False}, "max", "high"), + ({"supports_reasoning": True, "supports_xhigh_reasoning_effort": False}, "xhigh", "high"), + ({"supports_reasoning": True, "supports_minimal_reasoning_effort": False}, "minimal", "low"), ], ) - def test_unknown_effort_metadata_keeps_the_requested_tier(self, model_info, effort): - with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.utils.get_model_info", return_value=model_info - ): - assert normalize_reasoning_effort_value(effort, "custom-registered-model", "anthropic") == effort - - def test_explicit_non_reasoning_still_degrades_to_the_chain_floor(self): - with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.utils.get_model_info", return_value={"supports_reasoning": False} - ): - assert normalize_reasoning_effort_value("max", "custom-registered-model", "anthropic") == "high" + def test_a_registered_deployment_declaring_a_tier_unsupported_degrades( + self, local_model_cost_map, model_info, effort, expected + ): + deployment_id = self._register_deployment(model_info) + assert normalize_reasoning_effort_value(effort, deployment_id, "anthropic") == expected # --------------------------------------------------------------------------- diff --git a/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py b/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py index 7b06899a013..9518c3c6c8e 100644 --- a/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py +++ b/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py @@ -290,78 +290,36 @@ class TestMapReasoningEffortDegradation: assert result["type"] == "adaptive" -class TestApplyOutputConfigDegradation: - def test_max_degrades_to_high_when_unsupported(self): - with ( - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", - return_value="effort='max' is not supported by this model. Got model: test", - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model", - return_value=True, - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.utils.get_model_info", - return_value=_mock_model_info( - supports_reasoning=True, - supports_max_reasoning_effort=False, - supports_xhigh_reasoning_effort=False, - ), - ), - ): - cfg = AnthropicConfig() - data: dict = {} - optional_params = {"output_config": {"effort": "max"}} - cfg._apply_output_config(data, "test-model", optional_params) - assert data["output_config"]["effort"] == "high" +class TestReasoningEffortAliasOutputConfig: + @staticmethod + def _transform_alias(reasoning_effort: str) -> dict: + config = AnthropicConfig() + optional_params = config.map_openai_params( + non_default_params={"reasoning_effort": reasoning_effort}, + optional_params={}, + model="claude-sonnet-4-6", + drop_params=False, + ) + return config.transform_request( + model="claude-sonnet-4-6", + messages=[{"role": "user", "content": "Hello"}], + optional_params={**optional_params, "max_tokens": 1024}, + litellm_params={}, + headers={}, + ) - def test_xhigh_degrades_to_high_when_unsupported(self): - with ( - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", - return_value="effort='xhigh' is not supported by this model. Got model: test", - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model", - return_value=True, - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.utils.get_model_info", - return_value=_mock_model_info( - supports_reasoning=True, - supports_xhigh_reasoning_effort=False, - ), - ), - ): - cfg = AnthropicConfig() - data: dict = {} - optional_params = {"output_config": {"effort": "xhigh"}} - cfg._apply_output_config(data, "test-model", optional_params) - assert data["output_config"]["effort"] == "high" + def test_unsupported_alias_tier_degrades_to_an_accepted_one(self, local_model_cost_map): + assert self._transform_alias("xhigh")["output_config"] == {"effort": "high"} - def test_max_stays_max_when_supported(self): - with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", - return_value=None, - ): - cfg = AnthropicConfig() - data: dict = {} - optional_params = {"output_config": {"effort": "max"}} - cfg._apply_output_config(data, "test-model", optional_params) - assert data["output_config"]["effort"] == "max" + def test_supported_alias_tier_is_kept(self, local_model_cost_map): + assert self._transform_alias("max")["output_config"] == {"effort": "max"} - def test_no_output_config_is_noop(self): - cfg = AnthropicConfig() - data: dict = {} - cfg._apply_output_config(data, "test-model", {}) - assert "output_config" not in data - - def test_invalid_effort_value_still_raises(self): - with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model", - return_value=True, - ): - cfg = AnthropicConfig() - with pytest.raises(litellm.exceptions.BadRequestError, match="Invalid effort value"): - cfg._apply_output_config({}, "test-model", {"output_config": {"effort": "bogus"}}) + def test_explicit_unsupported_output_config_effort_is_rejected_not_rewritten(self, local_model_cost_map): + with pytest.raises(litellm.exceptions.BadRequestError, match="xhigh"): + AnthropicConfig().transform_request( + model="claude-sonnet-4-6", + messages=[{"role": "user", "content": "Hello"}], + optional_params={"max_tokens": 1024, "output_config": {"effort": "xhigh"}}, + litellm_params={}, + headers={}, + )