From f9209497cbf2baca1314c7c207f1bb3810926bc6 Mon Sep 17 00:00:00 2001 From: hx <1367557521@qq.com> Date: Sun, 27 Sep 2026 17:48:40 +0800 Subject: [PATCH 1/2] fix(anthropic): degrade reasoning effort only when a capability is false Co-authored-by: Cursor --- litellm/llms/anthropic/chat/transformation.py | 23 +- .../pass_through/messages/transformation.py | 25 +- litellm/llms/anthropic/pass_through/utils.py | 11 +- .../test_reasoning_effort_translation.py | 173 ++++++++--- .../test_reasoning_effort_fields.py | 36 ++- .../test_anthropic_reasoning_effort.py | 291 ++++++++++++++++++ 6 files changed, 483 insertions(+), 76 deletions(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 3bffee48d6a..ecf400b9787 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -104,6 +104,7 @@ from ..common_utils import ( requires_native_compaction_beta, strip_advisor_blocks_from_messages, ) +from ..pass_through.utils import normalize_reasoning_effort_value if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj @@ -1264,7 +1265,8 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): type="adaptive", display="summarized", ) - elif reasoning_effort == "low": + reasoning_effort = normalize_reasoning_effort_value(str(reasoning_effort), model, custom_llm_provider) + if reasoning_effort == "low": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET, @@ -2122,11 +2124,20 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): ) gate_error: Final = self._validate_effort_for_model(model, effort, self._resolved_provider) if gate_error is not None: - raise litellm.exceptions.BadRequestError( - message=gate_error, - model=model, - llm_provider=self._resolved_provider, - ) + if not isinstance(effort, str): + raise litellm.exceptions.BadRequestError( + message=gate_error, + model=model, + llm_provider=self._resolved_provider, + ) + normalized_effort: Final = normalize_reasoning_effort_value(effort, model, self._resolved_provider) + if normalized_effort == effort: + raise litellm.exceptions.BadRequestError( + message=gate_error, + model=model, + llm_provider=self._resolved_provider, + ) + output_config["effort"] = normalized_effort data["output_config"] = output_config def _resolve_json_mode_non_streaming( diff --git a/litellm/llms/anthropic/pass_through/messages/transformation.py b/litellm/llms/anthropic/pass_through/messages/transformation.py index 1f604cfb8d7..5ad78e2eabb 100644 --- a/litellm/llms/anthropic/pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/pass_through/messages/transformation.py @@ -28,6 +28,7 @@ from ...common_utils import ( strip_advisor_blocks_from_messages, strip_encrypted_reasoning_blocks_from_anthropic_messages, ) +from ..utils import normalize_reasoning_effort_value from .mid_conversation_system import ( as_system_content_blocks, convert_mid_conversation_system_turns, @@ -323,8 +324,8 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): optional_params.setdefault("thinking", fitted_thinking) if AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider): - mapped_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort) - if mapped_effort is None: + requested_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort) + if requested_effort is None: raise AnthropicError( message=( f"Invalid reasoning_effort: {reasoning_effort!r}. " @@ -333,14 +334,20 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): ), status_code=400, ) - gate_error: Final = AnthropicConfig._validate_effort_for_model(model, mapped_effort, custom_llm_provider) - if gate_error is not None: + existing_output_config: Final = optional_params.get("output_config") + existing_mapping: Final = existing_output_config if isinstance(existing_output_config, dict) else {} + raw_explicit_effort: Final = existing_mapping.get("effort") + explicit_effort: Final = raw_explicit_effort if isinstance(raw_explicit_effort, str) else None + candidate_effort: Final = explicit_effort if explicit_effort is not None else requested_effort + gate_error: Final = AnthropicConfig._validate_effort_for_model(model, candidate_effort, custom_llm_provider) + resolved_effort: Final = ( + candidate_effort + if gate_error is None + else normalize_reasoning_effort_value(candidate_effort, model, custom_llm_provider) + ) + if gate_error is not None and resolved_effort == candidate_effort: raise AnthropicError(message=gate_error, status_code=400) - existing_output_config = optional_params.get("output_config") - if not isinstance(existing_output_config, dict): - existing_output_config = {} - existing_output_config.setdefault("effort", mapped_effort) - optional_params["output_config"] = existing_output_config + optional_params["output_config"] = {**existing_mapping, "effort": resolved_effort} @staticmethod def _translate_adaptive_effort_for_non_adaptive_model( diff --git a/litellm/llms/anthropic/pass_through/utils.py b/litellm/llms/anthropic/pass_through/utils.py index 335a0e5641d..f14c6812acd 100644 --- a/litellm/llms/anthropic/pass_through/utils.py +++ b/litellm/llms/anthropic/pass_through/utils.py @@ -74,6 +74,11 @@ def normalize_reasoning_effort_value( The accepted set is resolved by the same owner that answers ``/model_group/info``, so a level the proxy advertises is a level this path forwards. + Degradation only happens when the capability set is known and the requested + tier is not in it. A model the map does not describe, or a mapped entry that + declares no effort metadata, keeps the requested value so third-party + Anthropic-compatible deployments are not silently downgraded. + A deployment that refuses every step of a chain falls back to an accepted level read off that same set rather than to an assumed one, since an entry naming its levels outright can exclude the tiers the per-level flags treat as unconditional. ``none`` is never that fallback and is @@ -91,9 +96,11 @@ def normalize_reasoning_effort_value( try: model_info: Final[ModelInfo] = get_model_info(model=model, custom_llm_provider=custom_llm_provider) except Exception: - return chain[-1] + return effort - supported: Final = resolve_supported_reasoning_efforts(model_info, deployment_is_mapped=True) + supported: Final = resolve_supported_reasoning_efforts(model_info, deployment_is_mapped=False) + if supported is None: + return effort if not supported: return chain[-1] diff --git a/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py b/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py index c1295305c7a..d6ed5ec0971 100644 --- a/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py +++ b/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py @@ -1,5 +1,7 @@ """Tests for ``reasoning_effort`` translation on the Anthropic /v1/messages route.""" +from unittest.mock import patch + import pytest from litellm.constants import ( @@ -27,9 +29,7 @@ from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_tran ("max", "max"), ], ) -def test_reasoning_effort_maps_to_output_config_for_adaptive_model( - reasoning_effort, expected_effort -): +def test_reasoning_effort_maps_to_output_config_for_adaptive_model(reasoning_effort, expected_effort): config = AnthropicMessagesConfig() optional_params = {"max_tokens": 1024, "reasoning_effort": reasoning_effort} @@ -107,27 +107,25 @@ def test_invalid_reasoning_effort_raises_400(bad_effort): @pytest.mark.parametrize( - "model,bad_effort", + "model,requested_effort,expected_effort", [ - ("claude-opus-4-6", "xhigh"), - ("claude-sonnet-4-6", "xhigh"), + ("claude-opus-4-6", "xhigh", "high"), + ("claude-sonnet-4-6", "xhigh", "high"), ], ) -def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort): +def test_reasoning_effort_unsupported_tier_degrades_on_messages(model, requested_effort, expected_effort): config = AnthropicMessagesConfig() - optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort} + optional_params = {"max_tokens": 1024, "reasoning_effort": requested_effort} - with pytest.raises(AnthropicError) as exc_info: - config.transform_anthropic_messages_request( - model=model, - messages=[{"role": "user", "content": "Hello"}], - anthropic_messages_optional_request_params=optional_params, - litellm_params={}, - headers={}, - ) + result = config.transform_anthropic_messages_request( + model=model, + messages=[{"role": "user", "content": "Hello"}], + anthropic_messages_optional_request_params=optional_params, + litellm_params={}, + headers={}, + ) - assert exc_info.value.status_code == 400 - assert "not supported by this model" in str(exc_info.value) + assert result["output_config"]["effort"] == expected_effort @pytest.mark.parametrize( @@ -139,9 +137,7 @@ def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort ("invoke/us.anthropic.claude-opus-4-7", "xhigh", "xhigh"), ], ) -def test_bedrock_invoke_messages_clamps_effort_to_ceiling( - local_model_cost_map, model, effort, expected_effort -): +def test_bedrock_invoke_messages_clamps_effort_to_ceiling(local_model_cost_map, model, effort, expected_effort): """Bedrock Invoke /v1/messages degrades effort to the model's ceiling. Claude Code "goal mode" sends ``xhigh``; Opus 4.6 must clamp to ``max`` @@ -162,22 +158,19 @@ def test_bedrock_invoke_messages_clamps_effort_to_ceiling( assert result["thinking"]["type"] == "adaptive" -def test_bedrock_invoke_messages_rejects_xhigh_without_ceiling(local_model_cost_map): - """Sonnet 4.6 on Bedrock has no effort ceiling, so xhigh is still rejected.""" +def test_bedrock_invoke_messages_degrades_xhigh_without_ceiling(local_model_cost_map): config = AmazonAnthropicClaudeMessagesConfig() optional_params = {"max_tokens": 1024, "reasoning_effort": "xhigh"} - with pytest.raises(AnthropicError) as exc_info: - config.transform_anthropic_messages_request( - model="invoke/us.anthropic.claude-sonnet-4-6", - messages=[{"role": "user", "content": "Hello"}], - anthropic_messages_optional_request_params=optional_params, - litellm_params={}, - headers={}, - ) + result = config.transform_anthropic_messages_request( + model="invoke/us.anthropic.claude-sonnet-4-6", + messages=[{"role": "user", "content": "Hello"}], + anthropic_messages_optional_request_params=optional_params, + litellm_params={}, + headers={}, + ) - assert exc_info.value.status_code == 400 - assert "not supported by this model" in str(exc_info.value) + assert result["output_config"]["effort"] == "high" @pytest.mark.parametrize( @@ -187,9 +180,7 @@ def test_bedrock_invoke_messages_rejects_xhigh_without_ceiling(local_model_cost_ "bedrock/invoke/us.anthropic.claude-sonnet-4-6", ], ) -def test_reasoning_effort_max_accepted_on_sonnet_46_messages( - local_model_cost_map, model -): +def test_reasoning_effort_max_accepted_on_sonnet_46_messages(local_model_cost_map, model): config = AnthropicMessagesConfig() optional_params = {"max_tokens": 1024, "reasoning_effort": "max"} @@ -205,6 +196,41 @@ def test_reasoning_effort_max_accepted_on_sonnet_46_messages( assert isinstance(output_config, dict) and output_config.get("effort") == "max" +def test_conflicting_unsupported_output_config_effort_is_not_forwarded(): + with ( + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model", + return_value=True, + ), + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", + side_effect=lambda model, effort, provider: ( + None if effort == "high" else f"effort={effort!r} is not supported by this model. Got model: {model}" + ), + ), + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value={ + "supports_reasoning": True, + "supports_max_reasoning_effort": False, + "supports_xhigh_reasoning_effort": False, + }, + ), + ): + optional_params = { + "reasoning_effort": "high", + "output_config": {"effort": "max"}, + } + AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic( + model="claude-sonnet-4-6", + optional_params=optional_params, + max_tokens=1024, + custom_llm_provider="anthropic", + ) + assert optional_params["output_config"]["effort"] != "max" + assert optional_params["output_config"]["effort"] == "high" + + def test_explicit_output_config_wins_over_reasoning_effort(): config = AnthropicMessagesConfig() optional_params = { @@ -247,9 +273,7 @@ def test_explicit_thinking_wins_over_reasoning_effort(): def test_reasoning_effort_in_supported_params(): config = AnthropicMessagesConfig() - assert "reasoning_effort" in config.get_supported_anthropic_messages_params( - "claude-opus-4-7" - ) + assert "reasoning_effort" in config.get_supported_anthropic_messages_params("claude-opus-4-7") @pytest.mark.parametrize( @@ -318,9 +342,7 @@ def test_legacy_thinking_high_budget_keeps_xhigh_when_supported(): "bedrock/invoke/us.anthropic.claude-opus-4-8", ], ) -def test_legacy_thinking_translates_to_adaptive_for_opus_48( - model, local_model_cost_map -): +def test_legacy_thinking_translates_to_adaptive_for_opus_48(model, local_model_cost_map): """Regression for issue #29188: Opus 4.8 requires adaptive thinking, but the legacy ``thinking.type='enabled'`` shape was passed through unchanged for Bedrock 4.8 (its cost-map entry lacked ``supports_adaptive_thinking`` and the @@ -352,9 +374,7 @@ def test_legacy_thinking_translates_to_adaptive_for_opus_48( ("claude-newfamily-6", "high"), ], ) -def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models( - local_model_cost_map, model, expected_effort -): +def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models(local_model_cost_map, model, expected_effort): """The 5 families reject ``thinking.type=enabled``, so the adaptive translation stays the safe default for every adaptive model not flagged ``supports_legacy_thinking``, unmapped future ids included. An unmapped id @@ -389,9 +409,7 @@ def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models( (1, "low"), ], ) -def test_legacy_thinking_budget_buckets_on_opus_48( - local_model_cost_map, budget_tokens, expected_effort -): +def test_legacy_thinking_budget_buckets_on_opus_48(local_model_cost_map, budget_tokens, expected_effort): config = AnthropicMessagesConfig() optional_params = { "max_tokens": 1024, @@ -478,9 +496,7 @@ def test_legacy_thinking_left_untouched_on_non_adaptive_model(): ("claude-sonnet-4-5", False), ], ) -def test_disabled_thinking_omitted_for_always_on_models_messages( - local_model_cost_map, model, expected_dropped -): +def test_disabled_thinking_omitted_for_always_on_models_messages(local_model_cost_map, model, expected_dropped): """/v1/messages: ``thinking={"type": "disabled"}`` is omitted for always-on-thinking models and forwarded verbatim for models that accept it.""" config = AnthropicMessagesConfig() @@ -498,3 +514,58 @@ def test_disabled_thinking_omitted_for_always_on_models_messages( assert "thinking" not in result else: assert result["thinking"] == {"type": "disabled"} + + +def _mock_model_info(**flags): + return flags + + +def test_xhigh_degrades_to_high_for_non_adaptive_model(): + with ( + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model", + return_value=False, + ), + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_xhigh_reasoning_effort=False, + ), + ), + ): + optional_params = {"reasoning_effort": "xhigh"} + AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic( + model="unknown-glm-4.6", + optional_params=optional_params, + max_tokens=None, + custom_llm_provider="anthropic", + ) + assert optional_params["thinking"]["type"] == "enabled" + assert optional_params["thinking"]["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET + assert "output_config" not in optional_params + + +def test_max_degrades_to_high_for_non_adaptive_model(): + with ( + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model", + return_value=False, + ), + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_max_reasoning_effort=False, + supports_xhigh_reasoning_effort=False, + ), + ), + ): + optional_params = {"reasoning_effort": "max"} + AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic( + model="unknown-deepseek", + optional_params=optional_params, + max_tokens=None, + custom_llm_provider="anthropic", + ) + assert optional_params["thinking"]["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET diff --git a/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py b/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py index e9450d025a6..8dfd20fcca4 100644 --- a/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py +++ b/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py @@ -9,7 +9,8 @@ Covers: import json import os -from typing import Any, Dict +from typing import Any +from unittest.mock import patch import pytest @@ -23,7 +24,7 @@ from litellm.router_utils.reasoning_effort_capability import ( from litellm.utils import get_model_info -def _load_model_registry() -> Dict[str, Any]: +def _load_model_registry() -> dict[str, Any]: """Load the root model_prices_and_context_window.json.""" json_path = os.path.join( os.path.dirname(__file__), @@ -141,9 +142,30 @@ class TestNormalizeReasoningEffortValue: def test_a_tier_outside_any_chain_passes_through(self, local_model_cost_map, effort): assert normalize_reasoning_effort_value(effort, "claude-opus-4-7", "anthropic") == effort - @pytest.mark.parametrize("effort, expected", [("max", "high"), ("xhigh", "high"), ("minimal", "low")]) - def test_a_model_the_map_does_not_describe_keeps_the_floor(self, local_model_cost_map, effort, expected): - assert normalize_reasoning_effort_value(effort, "totally-made-up-model-xyz", "openai") == expected + @pytest.mark.parametrize("effort", ["max", "xhigh", "minimal"]) + def test_a_model_the_map_does_not_describe_keeps_the_requested_tier(self, local_model_cost_map, effort): + assert normalize_reasoning_effort_value(effort, "totally-made-up-model-xyz", "openai") == effort + + @pytest.mark.parametrize( + "model_info, effort", + [ + ({}, "max"), + ({}, "xhigh"), + ({"supports_reasoning": None}, "max"), + ({"supports_reasoning": None}, "xhigh"), + ], + ) + def test_unknown_effort_metadata_keeps_the_requested_tier(self, model_info, effort): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", return_value=model_info + ): + assert normalize_reasoning_effort_value(effort, "custom-registered-model", "anthropic") == effort + + def test_explicit_non_reasoning_still_degrades_to_the_chain_floor(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", return_value={"supports_reasoning": False} + ): + assert normalize_reasoning_effort_value("max", "custom-registered-model", "anthropic") == "high" # --------------------------------------------------------------------------- @@ -161,9 +183,7 @@ class TestAdapterAdaptiveThinking: ) adapter = LiteLLMAnthropicMessagesAdapter() - result = adapter.translate_anthropic_thinking_to_reasoning_effort( - {"type": "adaptive"} - ) + result = adapter.translate_anthropic_thinking_to_reasoning_effort({"type": "adaptive"}) assert result == "medium" def test_messages_adapter_adaptive_overridden_by_output_config(self): diff --git a/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py b/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py index 288817dff07..7b06899a013 100644 --- a/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py +++ b/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py @@ -5,8 +5,19 @@ Verifies that reasoning_effort=None returns None for all models, including Claude Opus 4.6. """ +from unittest.mock import patch + import pytest +import litellm.exceptions +from litellm.constants import ( + DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, + DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET, + DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET, + DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, + DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET, + DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, +) from litellm.llms.anthropic.chat.transformation import AnthropicConfig @@ -74,3 +85,283 @@ class TestMapReasoningEffort: reasoning_effort="none", model="claude-4-sonnet-20250514", custom_llm_provider="anthropic" ) assert result is None + + +def _mock_model_info(**flags): + return flags + + +class TestMapReasoningEffortDegradation: + def test_max_stays_max_when_supported(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_max_reasoning_effort=True, + supports_xhigh_reasoning_effort=True, + ), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="max", + model="test-model", + custom_llm_provider="anthropic", + ) + assert result["type"] == "enabled" + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET + + def test_max_degrades_to_xhigh_when_only_xhigh_supported(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_max_reasoning_effort=False, + supports_xhigh_reasoning_effort=True, + ), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="max", + model="test-model", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET + + def test_max_degrades_to_high_when_neither_max_nor_xhigh_supported(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_max_reasoning_effort=False, + supports_xhigh_reasoning_effort=False, + ), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="max", + model="test-model", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET + + def test_max_passthrough_for_unknown_model(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + side_effect=Exception("model not found"), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="max", + model="unknown-glm-4.6", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET + + def test_xhigh_stays_xhigh_when_supported(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_xhigh_reasoning_effort=True, + ), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="xhigh", + model="test-model", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET + + def test_xhigh_degrades_to_high_when_unsupported(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_xhigh_reasoning_effort=False, + ), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="xhigh", + model="test-model", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET + + def test_xhigh_passthrough_for_unknown_model(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + side_effect=Exception("model not found"), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="xhigh", + model="unknown-deepseek", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET + + def test_minimal_stays_minimal_when_supported(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_minimal_reasoning_effort=True, + ), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="minimal", + model="test-model", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == max(DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET, 1024) + + def test_minimal_degrades_to_low_when_unsupported(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_minimal_reasoning_effort=False, + ), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="minimal", + model="test-model", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET + + def test_high_unchanged(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + side_effect=Exception("model not found"), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="high", + model="unknown-model", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET + + def test_medium_unchanged(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + side_effect=Exception("model not found"), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="medium", + model="unknown-model", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET + + def test_low_unchanged(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + side_effect=Exception("model not found"), + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="low", + model="unknown-model", + custom_llm_provider="anthropic", + ) + assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET + + def test_none_returns_none(self): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="none", + model="any-model", + custom_llm_provider="anthropic", + ) + assert result is None + + def test_none_value_returns_none(self): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort=None, + model="any-model", + custom_llm_provider="anthropic", + ) + assert result is None + + def test_adaptive_model_short_circuits_before_degradation(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model", + return_value=True, + ): + result = AnthropicConfig._map_reasoning_effort( + reasoning_effort="max", + model="claude-opus-4-6", + custom_llm_provider="anthropic", + ) + assert result["type"] == "adaptive" + + +class TestApplyOutputConfigDegradation: + def test_max_degrades_to_high_when_unsupported(self): + with ( + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", + return_value="effort='max' is not supported by this model. Got model: test", + ), + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model", + return_value=True, + ), + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_max_reasoning_effort=False, + supports_xhigh_reasoning_effort=False, + ), + ), + ): + cfg = AnthropicConfig() + data: dict = {} + optional_params = {"output_config": {"effort": "max"}} + cfg._apply_output_config(data, "test-model", optional_params) + assert data["output_config"]["effort"] == "high" + + def test_xhigh_degrades_to_high_when_unsupported(self): + with ( + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", + return_value="effort='xhigh' is not supported by this model. Got model: test", + ), + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model", + return_value=True, + ), + patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.utils.get_model_info", + return_value=_mock_model_info( + supports_reasoning=True, + supports_xhigh_reasoning_effort=False, + ), + ), + ): + cfg = AnthropicConfig() + data: dict = {} + optional_params = {"output_config": {"effort": "xhigh"}} + cfg._apply_output_config(data, "test-model", optional_params) + assert data["output_config"]["effort"] == "high" + + def test_max_stays_max_when_supported(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", + return_value=None, + ): + cfg = AnthropicConfig() + data: dict = {} + optional_params = {"output_config": {"effort": "max"}} + cfg._apply_output_config(data, "test-model", optional_params) + assert data["output_config"]["effort"] == "max" + + def test_no_output_config_is_noop(self): + cfg = AnthropicConfig() + data: dict = {} + cfg._apply_output_config(data, "test-model", {}) + assert "output_config" not in data + + def test_invalid_effort_value_still_raises(self): + with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain + "litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model", + return_value=True, + ): + cfg = AnthropicConfig() + with pytest.raises(litellm.exceptions.BadRequestError, match="Invalid effort value"): + cfg._apply_output_config({}, "test-model", {"output_config": {"effort": "bogus"}}) From 372fc144860ed9325daecdb5d061cea25b541039 Mon Sep 17 00:00:00 2001 From: hx <1367557521@qq.com> Date: Sun, 27 Sep 2026 18:08:26 +0800 Subject: [PATCH 2/2] fix(anthropic): reject explicit unsupported effort instead of rewriting it An explicit output_config.effort is the caller's native choice, so it wins over the reasoning_effort alias and a tier the model rejects returns a 400 on both the chat and /v1/messages paths, matching the existing chat contract. Only the alias is lowered to a tier the model is known to accept, through one shared helper. Bedrock invoke clamps an explicit effort sent alongside the alias to its effort ceiling before the shared gate, as it already did for the alias. Tests register a real Router deployment without effort metadata to prove it keeps the requested tier, replacing get_model_info mocks. Co-authored-by: Cursor --- litellm/llms/anthropic/chat/transformation.py | 49 +++++---- .../pass_through/messages/transformation.py | 34 +++--- litellm/llms/anthropic/pass_through/utils.py | 6 +- .../anthropic_claude3_transformation.py | 11 +- .../test_reasoning_effort_translation.py | 88 +++++++++------ .../test_reasoning_effort_fields.py | 46 +++++--- .../test_anthropic_reasoning_effort.py | 102 ++++++------------ 7 files changed, 176 insertions(+), 160 deletions(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index ecf400b9787..7860e0998dd 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -415,6 +415,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): return f"effort='xhigh' is not supported by this model. Got model: {model}" return None + @staticmethod + def degrade_alias_effort_for_model(model: str, effort: str, custom_llm_provider: str) -> str: + """Keep an alias-derived effort the gate accepts, else lower it to a tier the model is known to accept. + + Explicit ``output_config.effort`` must not be routed here: a caller naming a native tier gets a 400. + """ + if AnthropicConfig._validate_effort_for_model(model, effort, custom_llm_provider) is None: + return effort + return normalize_reasoning_effort_value(effort, model, custom_llm_provider) + @staticmethod def _model_supports_effort_param(model: str, custom_llm_provider: str) -> bool: """Whether the model accepts ``output_config.effort`` at all. @@ -1265,33 +1275,33 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): type="adaptive", display="summarized", ) - reasoning_effort = normalize_reasoning_effort_value(str(reasoning_effort), model, custom_llm_provider) - if reasoning_effort == "low": + resolved_effort: Final = normalize_reasoning_effort_value(reasoning_effort, model, custom_llm_provider) + if resolved_effort == "low": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET, ) - elif reasoning_effort == "medium": + elif resolved_effort == "medium": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, ) - elif reasoning_effort == "high": + elif resolved_effort == "high": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, ) - elif reasoning_effort == "xhigh": + elif resolved_effort == "xhigh": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, ) - elif reasoning_effort == "max": + elif resolved_effort == "max": return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET, ) - elif reasoning_effort == "minimal": + elif resolved_effort == "minimal": return AnthropicThinkingParam( type="enabled", budget_tokens=max( @@ -1624,7 +1634,11 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): value=effort_value, llm_provider=self._resolved_provider, ) - optional_params["output_config"] = {"effort": mapped_effort} + optional_params["output_config"] = { + "effort": AnthropicConfig.degrade_alias_effort_for_model( + model, mapped_effort, self._resolved_provider + ) + } elif param == "web_search_options" and isinstance(value, dict): hosted_web_search_tool = self.map_web_search_tool(cast(OpenAIWebSearchOptions, value)) self._add_tools_to_optional_params(optional_params=optional_params, tools=[hosted_web_search_tool]) @@ -2124,20 +2138,11 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): ) gate_error: Final = self._validate_effort_for_model(model, effort, self._resolved_provider) if gate_error is not None: - if not isinstance(effort, str): - raise litellm.exceptions.BadRequestError( - message=gate_error, - model=model, - llm_provider=self._resolved_provider, - ) - normalized_effort: Final = normalize_reasoning_effort_value(effort, model, self._resolved_provider) - if normalized_effort == effort: - raise litellm.exceptions.BadRequestError( - message=gate_error, - model=model, - llm_provider=self._resolved_provider, - ) - output_config["effort"] = normalized_effort + raise litellm.exceptions.BadRequestError( + message=gate_error, + model=model, + llm_provider=self._resolved_provider, + ) data["output_config"] = output_config def _resolve_json_mode_non_streaming( diff --git a/litellm/llms/anthropic/pass_through/messages/transformation.py b/litellm/llms/anthropic/pass_through/messages/transformation.py index 5ad78e2eabb..8dd31eb32a9 100644 --- a/litellm/llms/anthropic/pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/pass_through/messages/transformation.py @@ -28,7 +28,6 @@ from ...common_utils import ( strip_advisor_blocks_from_messages, strip_encrypted_reasoning_blocks_from_anthropic_messages, ) -from ..utils import normalize_reasoning_effort_value from .mid_conversation_system import ( as_system_content_blocks, convert_mid_conversation_system_turns, @@ -324,8 +323,8 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): optional_params.setdefault("thinking", fitted_thinking) if AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider): - requested_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort) - if requested_effort is None: + mapped_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort) + if mapped_effort is None: raise AnthropicError( message=( f"Invalid reasoning_effort: {reasoning_effort!r}. " @@ -335,19 +334,28 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): status_code=400, ) existing_output_config: Final = optional_params.get("output_config") - existing_mapping: Final = existing_output_config if isinstance(existing_output_config, dict) else {} - raw_explicit_effort: Final = existing_mapping.get("effort") - explicit_effort: Final = raw_explicit_effort if isinstance(raw_explicit_effort, str) else None - candidate_effort: Final = explicit_effort if explicit_effort is not None else requested_effort - gate_error: Final = AnthropicConfig._validate_effort_for_model(model, candidate_effort, custom_llm_provider) + explicit_effort: Final = AnthropicMessagesConfig._explicit_output_config_effort(existing_output_config) resolved_effort: Final = ( - candidate_effort - if gate_error is None - else normalize_reasoning_effort_value(candidate_effort, model, custom_llm_provider) + explicit_effort + if explicit_effort is not None + else AnthropicConfig.degrade_alias_effort_for_model(model, mapped_effort, custom_llm_provider) ) - if gate_error is not None and resolved_effort == candidate_effort: + gate_error: Final = AnthropicConfig._validate_effort_for_model(model, resolved_effort, custom_llm_provider) + if gate_error is not None: raise AnthropicError(message=gate_error, status_code=400) - optional_params["output_config"] = {**existing_mapping, "effort": resolved_effort} + optional_params["output_config"] = ( + {**existing_output_config, "effort": resolved_effort} + if isinstance(existing_output_config, dict) + else {"effort": resolved_effort} + ) + + @staticmethod + def _explicit_output_config_effort(output_config: object) -> str | None: + match output_config: + case {"effort": str() as effort}: + return effort + case _: + return None @staticmethod def _translate_adaptive_effort_for_non_adaptive_model( diff --git a/litellm/llms/anthropic/pass_through/utils.py b/litellm/llms/anthropic/pass_through/utils.py index f14c6812acd..f64f411b69e 100644 --- a/litellm/llms/anthropic/pass_through/utils.py +++ b/litellm/llms/anthropic/pass_through/utils.py @@ -74,10 +74,8 @@ def normalize_reasoning_effort_value( The accepted set is resolved by the same owner that answers ``/model_group/info``, so a level the proxy advertises is a level this path forwards. - Degradation only happens when the capability set is known and the requested - tier is not in it. A model the map does not describe, or a mapped entry that - declares no effort metadata, keeps the requested value so third-party - Anthropic-compatible deployments are not silently downgraded. + Only a known capability set can refuse a tier: a model the map does not describe, or an entry + declaring no effort metadata, keeps the requested tier instead of being silently downgraded. A deployment that refuses every step of a chain falls back to an accepted level read off that same set rather than to an assumed one, since an entry naming its levels outright can exclude diff --git a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py index 94eb0c92e40..948f989fc3d 100644 --- a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py @@ -631,7 +631,8 @@ class AmazonAnthropicClaudeMessagesConfig( @staticmethod def _clamp_adaptive_reasoning_effort_for_bedrock(model: str, optional_params: dict) -> None: - """Lower ``reasoning_effort`` to the Bedrock effort ceiling before validation. + """Lower ``reasoning_effort`` and an explicit ``output_config.effort`` to the Bedrock effort ceiling + before validation. The shared ``/v1/messages`` effort gate rejects tiers a model does not natively support (e.g. ``xhigh`` on Opus 4.6). Bedrock's chat paths instead @@ -648,6 +649,14 @@ class AmazonAnthropicClaudeMessagesConfig( clamped: Final = {"effort": effort} normalize_bedrock_opus_output_config_effort(model=model, output_config=clamped) optional_params["reasoning_effort"] = clamped["effort"] + explicit_effort: Final = AnthropicMessagesConfig._explicit_output_config_effort( + optional_params.get("output_config") + ) + if explicit_effort is None: + return + clamped_explicit: Final = {"effort": explicit_effort} + normalize_bedrock_opus_output_config_effort(model=model, output_config=clamped_explicit) + optional_params["output_config"] = {**optional_params["output_config"], **clamped_explicit} def transform_anthropic_messages_request( self, diff --git a/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py b/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py index d6ed5ec0971..eee937bd83a 100644 --- a/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py +++ b/tests/unit/llms/anthropic/pass_through/messages/test_reasoning_effort_translation.py @@ -158,6 +158,27 @@ def test_bedrock_invoke_messages_clamps_effort_to_ceiling(local_model_cost_map, assert result["thinking"]["type"] == "adaptive" +def test_bedrock_invoke_messages_clamps_explicit_effort_sent_with_alias(local_model_cost_map): + config = AmazonAnthropicClaudeMessagesConfig() + explicit_output_config = {"effort": "xhigh"} + optional_params = { + "max_tokens": 1024, + "reasoning_effort": "xhigh", + "output_config": explicit_output_config, + } + + result = config.transform_anthropic_messages_request( + model="invoke/us.anthropic.claude-opus-4-6-v1", + messages=[{"role": "user", "content": "Hello"}], + anthropic_messages_optional_request_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert result["output_config"]["effort"] == "max" + assert explicit_output_config == {"effort": "xhigh"} + + def test_bedrock_invoke_messages_degrades_xhigh_without_ceiling(local_model_cost_map): config = AmazonAnthropicClaudeMessagesConfig() optional_params = {"max_tokens": 1024, "reasoning_effort": "xhigh"} @@ -196,39 +217,44 @@ def test_reasoning_effort_max_accepted_on_sonnet_46_messages(local_model_cost_ma assert isinstance(output_config, dict) and output_config.get("effort") == "max" -def test_conflicting_unsupported_output_config_effort_is_not_forwarded(): - with ( - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model", - return_value=True, - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", - side_effect=lambda model, effort, provider: ( - None if effort == "high" else f"effort={effort!r} is not supported by this model. Got model: {model}" - ), - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.utils.get_model_info", - return_value={ - "supports_reasoning": True, - "supports_max_reasoning_effort": False, - "supports_xhigh_reasoning_effort": False, - }, - ), - ): - optional_params = { - "reasoning_effort": "high", - "output_config": {"effort": "max"}, - } - AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic( +def test_explicit_unsupported_output_config_effort_is_rejected_not_rewritten(local_model_cost_map): + config = AnthropicMessagesConfig() + optional_params = { + "max_tokens": 1024, + "reasoning_effort": "high", + "output_config": {"effort": "xhigh"}, + } + + with pytest.raises(AnthropicError) as exc_info: + config.transform_anthropic_messages_request( model="claude-sonnet-4-6", - optional_params=optional_params, - max_tokens=1024, - custom_llm_provider="anthropic", + messages=[{"role": "user", "content": "Hello"}], + anthropic_messages_optional_request_params=optional_params, + litellm_params={}, + headers={}, ) - assert optional_params["output_config"]["effort"] != "max" - assert optional_params["output_config"]["effort"] == "high" + + assert exc_info.value.status_code == 400 + assert "xhigh" in str(exc_info.value) + + +def test_explicit_supported_output_config_effort_wins_over_unsupported_alias(local_model_cost_map): + config = AnthropicMessagesConfig() + optional_params = { + "max_tokens": 1024, + "reasoning_effort": "xhigh", + "output_config": {"effort": "low"}, + } + + result = config.transform_anthropic_messages_request( + model="claude-sonnet-4-6", + messages=[{"role": "user", "content": "Hello"}], + anthropic_messages_optional_request_params=optional_params, + litellm_params={}, + headers={}, + ) + + assert result["output_config"] == {"effort": "low"} def test_explicit_output_config_wins_over_reasoning_effort(): diff --git a/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py b/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py index 8dfd20fcca4..068bf0edc3b 100644 --- a/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py +++ b/tests/unit/llms/anthropic/pass_through/test_reasoning_effort_fields.py @@ -10,7 +10,6 @@ Covers: import json import os from typing import Any -from unittest.mock import patch import pytest @@ -146,26 +145,39 @@ class TestNormalizeReasoningEffortValue: def test_a_model_the_map_does_not_describe_keeps_the_requested_tier(self, local_model_cost_map, effort): assert normalize_reasoning_effort_value(effort, "totally-made-up-model-xyz", "openai") == effort + @staticmethod + def _register_deployment(model_info: dict[str, object]) -> str: + router = litellm.Router( + model_list=[ + { + "model_name": "compat", + "litellm_params": {"model": "anthropic/compat-reasoner-1", "api_key": "fake-key"}, + "model_info": model_info, + } + ] + ) + return router.model_list[0]["model_info"]["id"] + + @pytest.mark.parametrize("effort", ["max", "xhigh", "minimal"]) + def test_a_registered_deployment_without_effort_metadata_keeps_the_requested_tier( + self, local_model_cost_map, effort + ): + deployment_id = self._register_deployment({}) + assert normalize_reasoning_effort_value(effort, deployment_id, "anthropic") == effort + @pytest.mark.parametrize( - "model_info, effort", + "model_info, effort, expected", [ - ({}, "max"), - ({}, "xhigh"), - ({"supports_reasoning": None}, "max"), - ({"supports_reasoning": None}, "xhigh"), + ({"supports_reasoning": False}, "max", "high"), + ({"supports_reasoning": True, "supports_xhigh_reasoning_effort": False}, "xhigh", "high"), + ({"supports_reasoning": True, "supports_minimal_reasoning_effort": False}, "minimal", "low"), ], ) - def test_unknown_effort_metadata_keeps_the_requested_tier(self, model_info, effort): - with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.utils.get_model_info", return_value=model_info - ): - assert normalize_reasoning_effort_value(effort, "custom-registered-model", "anthropic") == effort - - def test_explicit_non_reasoning_still_degrades_to_the_chain_floor(self): - with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.utils.get_model_info", return_value={"supports_reasoning": False} - ): - assert normalize_reasoning_effort_value("max", "custom-registered-model", "anthropic") == "high" + def test_a_registered_deployment_declaring_a_tier_unsupported_degrades( + self, local_model_cost_map, model_info, effort, expected + ): + deployment_id = self._register_deployment(model_info) + assert normalize_reasoning_effort_value(effort, deployment_id, "anthropic") == expected # --------------------------------------------------------------------------- diff --git a/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py b/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py index 7b06899a013..9518c3c6c8e 100644 --- a/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py +++ b/tests/unit/llms/anthropic/test_anthropic_reasoning_effort.py @@ -290,78 +290,36 @@ class TestMapReasoningEffortDegradation: assert result["type"] == "adaptive" -class TestApplyOutputConfigDegradation: - def test_max_degrades_to_high_when_unsupported(self): - with ( - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", - return_value="effort='max' is not supported by this model. Got model: test", - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model", - return_value=True, - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.utils.get_model_info", - return_value=_mock_model_info( - supports_reasoning=True, - supports_max_reasoning_effort=False, - supports_xhigh_reasoning_effort=False, - ), - ), - ): - cfg = AnthropicConfig() - data: dict = {} - optional_params = {"output_config": {"effort": "max"}} - cfg._apply_output_config(data, "test-model", optional_params) - assert data["output_config"]["effort"] == "high" +class TestReasoningEffortAliasOutputConfig: + @staticmethod + def _transform_alias(reasoning_effort: str) -> dict: + config = AnthropicConfig() + optional_params = config.map_openai_params( + non_default_params={"reasoning_effort": reasoning_effort}, + optional_params={}, + model="claude-sonnet-4-6", + drop_params=False, + ) + return config.transform_request( + model="claude-sonnet-4-6", + messages=[{"role": "user", "content": "Hello"}], + optional_params={**optional_params, "max_tokens": 1024}, + litellm_params={}, + headers={}, + ) - def test_xhigh_degrades_to_high_when_unsupported(self): - with ( - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", - return_value="effort='xhigh' is not supported by this model. Got model: test", - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model", - return_value=True, - ), - patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.utils.get_model_info", - return_value=_mock_model_info( - supports_reasoning=True, - supports_xhigh_reasoning_effort=False, - ), - ), - ): - cfg = AnthropicConfig() - data: dict = {} - optional_params = {"output_config": {"effort": "xhigh"}} - cfg._apply_output_config(data, "test-model", optional_params) - assert data["output_config"]["effort"] == "high" + def test_unsupported_alias_tier_degrades_to_an_accepted_one(self, local_model_cost_map): + assert self._transform_alias("xhigh")["output_config"] == {"effort": "high"} - def test_max_stays_max_when_supported(self): - with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model", - return_value=None, - ): - cfg = AnthropicConfig() - data: dict = {} - optional_params = {"output_config": {"effort": "max"}} - cfg._apply_output_config(data, "test-model", optional_params) - assert data["output_config"]["effort"] == "max" + def test_supported_alias_tier_is_kept(self, local_model_cost_map): + assert self._transform_alias("max")["output_config"] == {"effort": "max"} - def test_no_output_config_is_noop(self): - cfg = AnthropicConfig() - data: dict = {} - cfg._apply_output_config(data, "test-model", {}) - assert "output_config" not in data - - def test_invalid_effort_value_still_raises(self): - with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain - "litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model", - return_value=True, - ): - cfg = AnthropicConfig() - with pytest.raises(litellm.exceptions.BadRequestError, match="Invalid effort value"): - cfg._apply_output_config({}, "test-model", {"output_config": {"effort": "bogus"}}) + def test_explicit_unsupported_output_config_effort_is_rejected_not_rewritten(self, local_model_cost_map): + with pytest.raises(litellm.exceptions.BadRequestError, match="xhigh"): + AnthropicConfig().transform_request( + model="claude-sonnet-4-6", + messages=[{"role": "user", "content": "Hello"}], + optional_params={"max_tokens": 1024, "output_config": {"effort": "xhigh"}}, + litellm_params={}, + headers={}, + )