mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(anthropic): degrade reasoning effort only when a capability is false
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
f4308bc124
commit
f9209497cb
6 changed files with 483 additions and 76 deletions
|
|
@ -104,6 +104,7 @@ from ..common_utils import (
|
|||
requires_native_compaction_beta,
|
||||
strip_advisor_blocks_from_messages,
|
||||
)
|
||||
from ..pass_through.utils import normalize_reasoning_effort_value
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
|
@ -1264,7 +1265,8 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
type="adaptive",
|
||||
display="summarized",
|
||||
)
|
||||
elif reasoning_effort == "low":
|
||||
reasoning_effort = normalize_reasoning_effort_value(str(reasoning_effort), model, custom_llm_provider)
|
||||
if reasoning_effort == "low":
|
||||
return AnthropicThinkingParam(
|
||||
type="enabled",
|
||||
budget_tokens=DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET,
|
||||
|
|
@ -2122,11 +2124,20 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
)
|
||||
gate_error: Final = self._validate_effort_for_model(model, effort, self._resolved_provider)
|
||||
if gate_error is not None:
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message=gate_error,
|
||||
model=model,
|
||||
llm_provider=self._resolved_provider,
|
||||
)
|
||||
if not isinstance(effort, str):
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message=gate_error,
|
||||
model=model,
|
||||
llm_provider=self._resolved_provider,
|
||||
)
|
||||
normalized_effort: Final = normalize_reasoning_effort_value(effort, model, self._resolved_provider)
|
||||
if normalized_effort == effort:
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
message=gate_error,
|
||||
model=model,
|
||||
llm_provider=self._resolved_provider,
|
||||
)
|
||||
output_config["effort"] = normalized_effort
|
||||
data["output_config"] = output_config
|
||||
|
||||
def _resolve_json_mode_non_streaming(
|
||||
|
|
|
|||
|
|
@ -28,6 +28,7 @@ from ...common_utils import (
|
|||
strip_advisor_blocks_from_messages,
|
||||
strip_encrypted_reasoning_blocks_from_anthropic_messages,
|
||||
)
|
||||
from ..utils import normalize_reasoning_effort_value
|
||||
from .mid_conversation_system import (
|
||||
as_system_content_blocks,
|
||||
convert_mid_conversation_system_turns,
|
||||
|
|
@ -323,8 +324,8 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
|
||||
optional_params.setdefault("thinking", fitted_thinking)
|
||||
if AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider):
|
||||
mapped_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort)
|
||||
if mapped_effort is None:
|
||||
requested_effort: Final = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort)
|
||||
if requested_effort is None:
|
||||
raise AnthropicError(
|
||||
message=(
|
||||
f"Invalid reasoning_effort: {reasoning_effort!r}. "
|
||||
|
|
@ -333,14 +334,20 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
),
|
||||
status_code=400,
|
||||
)
|
||||
gate_error: Final = AnthropicConfig._validate_effort_for_model(model, mapped_effort, custom_llm_provider)
|
||||
if gate_error is not None:
|
||||
existing_output_config: Final = optional_params.get("output_config")
|
||||
existing_mapping: Final = existing_output_config if isinstance(existing_output_config, dict) else {}
|
||||
raw_explicit_effort: Final = existing_mapping.get("effort")
|
||||
explicit_effort: Final = raw_explicit_effort if isinstance(raw_explicit_effort, str) else None
|
||||
candidate_effort: Final = explicit_effort if explicit_effort is not None else requested_effort
|
||||
gate_error: Final = AnthropicConfig._validate_effort_for_model(model, candidate_effort, custom_llm_provider)
|
||||
resolved_effort: Final = (
|
||||
candidate_effort
|
||||
if gate_error is None
|
||||
else normalize_reasoning_effort_value(candidate_effort, model, custom_llm_provider)
|
||||
)
|
||||
if gate_error is not None and resolved_effort == candidate_effort:
|
||||
raise AnthropicError(message=gate_error, status_code=400)
|
||||
existing_output_config = optional_params.get("output_config")
|
||||
if not isinstance(existing_output_config, dict):
|
||||
existing_output_config = {}
|
||||
existing_output_config.setdefault("effort", mapped_effort)
|
||||
optional_params["output_config"] = existing_output_config
|
||||
optional_params["output_config"] = {**existing_mapping, "effort": resolved_effort}
|
||||
|
||||
@staticmethod
|
||||
def _translate_adaptive_effort_for_non_adaptive_model(
|
||||
|
|
|
|||
|
|
@ -74,6 +74,11 @@ def normalize_reasoning_effort_value(
|
|||
The accepted set is resolved by the same owner that answers ``/model_group/info``, so a level
|
||||
the proxy advertises is a level this path forwards.
|
||||
|
||||
Degradation only happens when the capability set is known and the requested
|
||||
tier is not in it. A model the map does not describe, or a mapped entry that
|
||||
declares no effort metadata, keeps the requested value so third-party
|
||||
Anthropic-compatible deployments are not silently downgraded.
|
||||
|
||||
A deployment that refuses every step of a chain falls back to an accepted level read off that
|
||||
same set rather than to an assumed one, since an entry naming its levels outright can exclude
|
||||
the tiers the per-level flags treat as unconditional. ``none`` is never that fallback and is
|
||||
|
|
@ -91,9 +96,11 @@ def normalize_reasoning_effort_value(
|
|||
try:
|
||||
model_info: Final[ModelInfo] = get_model_info(model=model, custom_llm_provider=custom_llm_provider)
|
||||
except Exception:
|
||||
return chain[-1]
|
||||
return effort
|
||||
|
||||
supported: Final = resolve_supported_reasoning_efforts(model_info, deployment_is_mapped=True)
|
||||
supported: Final = resolve_supported_reasoning_efforts(model_info, deployment_is_mapped=False)
|
||||
if supported is None:
|
||||
return effort
|
||||
if not supported:
|
||||
return chain[-1]
|
||||
|
||||
|
|
|
|||
|
|
@ -1,5 +1,7 @@
|
|||
"""Tests for ``reasoning_effort`` translation on the Anthropic /v1/messages route."""
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.constants import (
|
||||
|
|
@ -27,9 +29,7 @@ from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_tran
|
|||
("max", "max"),
|
||||
],
|
||||
)
|
||||
def test_reasoning_effort_maps_to_output_config_for_adaptive_model(
|
||||
reasoning_effort, expected_effort
|
||||
):
|
||||
def test_reasoning_effort_maps_to_output_config_for_adaptive_model(reasoning_effort, expected_effort):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": reasoning_effort}
|
||||
|
||||
|
|
@ -107,27 +107,25 @@ def test_invalid_reasoning_effort_raises_400(bad_effort):
|
|||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,bad_effort",
|
||||
"model,requested_effort,expected_effort",
|
||||
[
|
||||
("claude-opus-4-6", "xhigh"),
|
||||
("claude-sonnet-4-6", "xhigh"),
|
||||
("claude-opus-4-6", "xhigh", "high"),
|
||||
("claude-sonnet-4-6", "xhigh", "high"),
|
||||
],
|
||||
)
|
||||
def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort):
|
||||
def test_reasoning_effort_unsupported_tier_degrades_on_messages(model, requested_effort, expected_effort):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort}
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": requested_effort}
|
||||
|
||||
with pytest.raises(AnthropicError) as exc_info:
|
||||
config.transform_anthropic_messages_request(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
result = config.transform_anthropic_messages_request(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "not supported by this model" in str(exc_info.value)
|
||||
assert result["output_config"]["effort"] == expected_effort
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
@ -139,9 +137,7 @@ def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort
|
|||
("invoke/us.anthropic.claude-opus-4-7", "xhigh", "xhigh"),
|
||||
],
|
||||
)
|
||||
def test_bedrock_invoke_messages_clamps_effort_to_ceiling(
|
||||
local_model_cost_map, model, effort, expected_effort
|
||||
):
|
||||
def test_bedrock_invoke_messages_clamps_effort_to_ceiling(local_model_cost_map, model, effort, expected_effort):
|
||||
"""Bedrock Invoke /v1/messages degrades effort to the model's ceiling.
|
||||
|
||||
Claude Code "goal mode" sends ``xhigh``; Opus 4.6 must clamp to ``max``
|
||||
|
|
@ -162,22 +158,19 @@ def test_bedrock_invoke_messages_clamps_effort_to_ceiling(
|
|||
assert result["thinking"]["type"] == "adaptive"
|
||||
|
||||
|
||||
def test_bedrock_invoke_messages_rejects_xhigh_without_ceiling(local_model_cost_map):
|
||||
"""Sonnet 4.6 on Bedrock has no effort ceiling, so xhigh is still rejected."""
|
||||
def test_bedrock_invoke_messages_degrades_xhigh_without_ceiling(local_model_cost_map):
|
||||
config = AmazonAnthropicClaudeMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": "xhigh"}
|
||||
|
||||
with pytest.raises(AnthropicError) as exc_info:
|
||||
config.transform_anthropic_messages_request(
|
||||
model="invoke/us.anthropic.claude-sonnet-4-6",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
result = config.transform_anthropic_messages_request(
|
||||
model="invoke/us.anthropic.claude-sonnet-4-6",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "not supported by this model" in str(exc_info.value)
|
||||
assert result["output_config"]["effort"] == "high"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
@ -187,9 +180,7 @@ def test_bedrock_invoke_messages_rejects_xhigh_without_ceiling(local_model_cost_
|
|||
"bedrock/invoke/us.anthropic.claude-sonnet-4-6",
|
||||
],
|
||||
)
|
||||
def test_reasoning_effort_max_accepted_on_sonnet_46_messages(
|
||||
local_model_cost_map, model
|
||||
):
|
||||
def test_reasoning_effort_max_accepted_on_sonnet_46_messages(local_model_cost_map, model):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": "max"}
|
||||
|
||||
|
|
@ -205,6 +196,41 @@ def test_reasoning_effort_max_accepted_on_sonnet_46_messages(
|
|||
assert isinstance(output_config, dict) and output_config.get("effort") == "max"
|
||||
|
||||
|
||||
def test_conflicting_unsupported_output_config_effort_is_not_forwarded():
|
||||
with (
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model",
|
||||
return_value=True,
|
||||
),
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model",
|
||||
side_effect=lambda model, effort, provider: (
|
||||
None if effort == "high" else f"effort={effort!r} is not supported by this model. Got model: {model}"
|
||||
),
|
||||
),
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value={
|
||||
"supports_reasoning": True,
|
||||
"supports_max_reasoning_effort": False,
|
||||
"supports_xhigh_reasoning_effort": False,
|
||||
},
|
||||
),
|
||||
):
|
||||
optional_params = {
|
||||
"reasoning_effort": "high",
|
||||
"output_config": {"effort": "max"},
|
||||
}
|
||||
AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic(
|
||||
model="claude-sonnet-4-6",
|
||||
optional_params=optional_params,
|
||||
max_tokens=1024,
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert optional_params["output_config"]["effort"] != "max"
|
||||
assert optional_params["output_config"]["effort"] == "high"
|
||||
|
||||
|
||||
def test_explicit_output_config_wins_over_reasoning_effort():
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
|
|
@ -247,9 +273,7 @@ def test_explicit_thinking_wins_over_reasoning_effort():
|
|||
|
||||
def test_reasoning_effort_in_supported_params():
|
||||
config = AnthropicMessagesConfig()
|
||||
assert "reasoning_effort" in config.get_supported_anthropic_messages_params(
|
||||
"claude-opus-4-7"
|
||||
)
|
||||
assert "reasoning_effort" in config.get_supported_anthropic_messages_params("claude-opus-4-7")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
@ -318,9 +342,7 @@ def test_legacy_thinking_high_budget_keeps_xhigh_when_supported():
|
|||
"bedrock/invoke/us.anthropic.claude-opus-4-8",
|
||||
],
|
||||
)
|
||||
def test_legacy_thinking_translates_to_adaptive_for_opus_48(
|
||||
model, local_model_cost_map
|
||||
):
|
||||
def test_legacy_thinking_translates_to_adaptive_for_opus_48(model, local_model_cost_map):
|
||||
"""Regression for issue #29188: Opus 4.8 requires adaptive thinking, but the
|
||||
legacy ``thinking.type='enabled'`` shape was passed through unchanged for
|
||||
Bedrock 4.8 (its cost-map entry lacked ``supports_adaptive_thinking`` and the
|
||||
|
|
@ -352,9 +374,7 @@ def test_legacy_thinking_translates_to_adaptive_for_opus_48(
|
|||
("claude-newfamily-6", "high"),
|
||||
],
|
||||
)
|
||||
def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models(
|
||||
local_model_cost_map, model, expected_effort
|
||||
):
|
||||
def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models(local_model_cost_map, model, expected_effort):
|
||||
"""The 5 families reject ``thinking.type=enabled``, so the adaptive translation
|
||||
stays the safe default for every adaptive model not flagged
|
||||
``supports_legacy_thinking``, unmapped future ids included. An unmapped id
|
||||
|
|
@ -389,9 +409,7 @@ def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models(
|
|||
(1, "low"),
|
||||
],
|
||||
)
|
||||
def test_legacy_thinking_budget_buckets_on_opus_48(
|
||||
local_model_cost_map, budget_tokens, expected_effort
|
||||
):
|
||||
def test_legacy_thinking_budget_buckets_on_opus_48(local_model_cost_map, budget_tokens, expected_effort):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
|
|
@ -478,9 +496,7 @@ def test_legacy_thinking_left_untouched_on_non_adaptive_model():
|
|||
("claude-sonnet-4-5", False),
|
||||
],
|
||||
)
|
||||
def test_disabled_thinking_omitted_for_always_on_models_messages(
|
||||
local_model_cost_map, model, expected_dropped
|
||||
):
|
||||
def test_disabled_thinking_omitted_for_always_on_models_messages(local_model_cost_map, model, expected_dropped):
|
||||
"""/v1/messages: ``thinking={"type": "disabled"}`` is omitted for always-on-thinking
|
||||
models and forwarded verbatim for models that accept it."""
|
||||
config = AnthropicMessagesConfig()
|
||||
|
|
@ -498,3 +514,58 @@ def test_disabled_thinking_omitted_for_always_on_models_messages(
|
|||
assert "thinking" not in result
|
||||
else:
|
||||
assert result["thinking"] == {"type": "disabled"}
|
||||
|
||||
|
||||
def _mock_model_info(**flags):
|
||||
return flags
|
||||
|
||||
|
||||
def test_xhigh_degrades_to_high_for_non_adaptive_model():
|
||||
with (
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model",
|
||||
return_value=False,
|
||||
),
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
),
|
||||
):
|
||||
optional_params = {"reasoning_effort": "xhigh"}
|
||||
AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic(
|
||||
model="unknown-glm-4.6",
|
||||
optional_params=optional_params,
|
||||
max_tokens=None,
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert optional_params["thinking"]["type"] == "enabled"
|
||||
assert optional_params["thinking"]["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
assert "output_config" not in optional_params
|
||||
|
||||
|
||||
def test_max_degrades_to_high_for_non_adaptive_model():
|
||||
with (
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model",
|
||||
return_value=False,
|
||||
),
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_max_reasoning_effort=False,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
),
|
||||
):
|
||||
optional_params = {"reasoning_effort": "max"}
|
||||
AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic(
|
||||
model="unknown-deepseek",
|
||||
optional_params=optional_params,
|
||||
max_tokens=None,
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert optional_params["thinking"]["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
|
|
|
|||
|
|
@ -9,7 +9,8 @@ Covers:
|
|||
|
||||
import json
|
||||
import os
|
||||
from typing import Any, Dict
|
||||
from typing import Any
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -23,7 +24,7 @@ from litellm.router_utils.reasoning_effort_capability import (
|
|||
from litellm.utils import get_model_info
|
||||
|
||||
|
||||
def _load_model_registry() -> Dict[str, Any]:
|
||||
def _load_model_registry() -> dict[str, Any]:
|
||||
"""Load the root model_prices_and_context_window.json."""
|
||||
json_path = os.path.join(
|
||||
os.path.dirname(__file__),
|
||||
|
|
@ -141,9 +142,30 @@ class TestNormalizeReasoningEffortValue:
|
|||
def test_a_tier_outside_any_chain_passes_through(self, local_model_cost_map, effort):
|
||||
assert normalize_reasoning_effort_value(effort, "claude-opus-4-7", "anthropic") == effort
|
||||
|
||||
@pytest.mark.parametrize("effort, expected", [("max", "high"), ("xhigh", "high"), ("minimal", "low")])
|
||||
def test_a_model_the_map_does_not_describe_keeps_the_floor(self, local_model_cost_map, effort, expected):
|
||||
assert normalize_reasoning_effort_value(effort, "totally-made-up-model-xyz", "openai") == expected
|
||||
@pytest.mark.parametrize("effort", ["max", "xhigh", "minimal"])
|
||||
def test_a_model_the_map_does_not_describe_keeps_the_requested_tier(self, local_model_cost_map, effort):
|
||||
assert normalize_reasoning_effort_value(effort, "totally-made-up-model-xyz", "openai") == effort
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_info, effort",
|
||||
[
|
||||
({}, "max"),
|
||||
({}, "xhigh"),
|
||||
({"supports_reasoning": None}, "max"),
|
||||
({"supports_reasoning": None}, "xhigh"),
|
||||
],
|
||||
)
|
||||
def test_unknown_effort_metadata_keeps_the_requested_tier(self, model_info, effort):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info", return_value=model_info
|
||||
):
|
||||
assert normalize_reasoning_effort_value(effort, "custom-registered-model", "anthropic") == effort
|
||||
|
||||
def test_explicit_non_reasoning_still_degrades_to_the_chain_floor(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info", return_value={"supports_reasoning": False}
|
||||
):
|
||||
assert normalize_reasoning_effort_value("max", "custom-registered-model", "anthropic") == "high"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
|
@ -161,9 +183,7 @@ class TestAdapterAdaptiveThinking:
|
|||
)
|
||||
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
result = adapter.translate_anthropic_thinking_to_reasoning_effort(
|
||||
{"type": "adaptive"}
|
||||
)
|
||||
result = adapter.translate_anthropic_thinking_to_reasoning_effort({"type": "adaptive"})
|
||||
assert result == "medium"
|
||||
|
||||
def test_messages_adapter_adaptive_overridden_by_output_config(self):
|
||||
|
|
|
|||
|
|
@ -5,8 +5,19 @@ Verifies that reasoning_effort=None returns None for all models,
|
|||
including Claude Opus 4.6.
|
||||
"""
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm.exceptions
|
||||
from litellm.constants import (
|
||||
DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
|
||||
)
|
||||
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
||||
|
||||
|
||||
|
|
@ -74,3 +85,283 @@ class TestMapReasoningEffort:
|
|||
reasoning_effort="none", model="claude-4-sonnet-20250514", custom_llm_provider="anthropic"
|
||||
)
|
||||
assert result is None
|
||||
|
||||
|
||||
def _mock_model_info(**flags):
|
||||
return flags
|
||||
|
||||
|
||||
class TestMapReasoningEffortDegradation:
|
||||
def test_max_stays_max_when_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_max_reasoning_effort=True,
|
||||
supports_xhigh_reasoning_effort=True,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="max",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["type"] == "enabled"
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET
|
||||
|
||||
def test_max_degrades_to_xhigh_when_only_xhigh_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_max_reasoning_effort=False,
|
||||
supports_xhigh_reasoning_effort=True,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="max",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET
|
||||
|
||||
def test_max_degrades_to_high_when_neither_max_nor_xhigh_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_max_reasoning_effort=False,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="max",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
|
||||
def test_max_passthrough_for_unknown_model(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="max",
|
||||
model="unknown-glm-4.6",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET
|
||||
|
||||
def test_xhigh_stays_xhigh_when_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_xhigh_reasoning_effort=True,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="xhigh",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET
|
||||
|
||||
def test_xhigh_degrades_to_high_when_unsupported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="xhigh",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
|
||||
def test_xhigh_passthrough_for_unknown_model(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="xhigh",
|
||||
model="unknown-deepseek",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET
|
||||
|
||||
def test_minimal_stays_minimal_when_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_minimal_reasoning_effort=True,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="minimal",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == max(DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET, 1024)
|
||||
|
||||
def test_minimal_degrades_to_low_when_unsupported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_minimal_reasoning_effort=False,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="minimal",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET
|
||||
|
||||
def test_high_unchanged(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="high",
|
||||
model="unknown-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
|
||||
def test_medium_unchanged(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="medium",
|
||||
model="unknown-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET
|
||||
|
||||
def test_low_unchanged(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="low",
|
||||
model="unknown-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET
|
||||
|
||||
def test_none_returns_none(self):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="none",
|
||||
model="any-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result is None
|
||||
|
||||
def test_none_value_returns_none(self):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort=None,
|
||||
model="any-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result is None
|
||||
|
||||
def test_adaptive_model_short_circuits_before_degradation(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model",
|
||||
return_value=True,
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="max",
|
||||
model="claude-opus-4-6",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["type"] == "adaptive"
|
||||
|
||||
|
||||
class TestApplyOutputConfigDegradation:
|
||||
def test_max_degrades_to_high_when_unsupported(self):
|
||||
with (
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model",
|
||||
return_value="effort='max' is not supported by this model. Got model: test",
|
||||
),
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model",
|
||||
return_value=True,
|
||||
),
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_max_reasoning_effort=False,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
),
|
||||
):
|
||||
cfg = AnthropicConfig()
|
||||
data: dict = {}
|
||||
optional_params = {"output_config": {"effort": "max"}}
|
||||
cfg._apply_output_config(data, "test-model", optional_params)
|
||||
assert data["output_config"]["effort"] == "high"
|
||||
|
||||
def test_xhigh_degrades_to_high_when_unsupported(self):
|
||||
with (
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model",
|
||||
return_value="effort='xhigh' is not supported by this model. Got model: test",
|
||||
),
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model",
|
||||
return_value=True,
|
||||
),
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
),
|
||||
):
|
||||
cfg = AnthropicConfig()
|
||||
data: dict = {}
|
||||
optional_params = {"output_config": {"effort": "xhigh"}}
|
||||
cfg._apply_output_config(data, "test-model", optional_params)
|
||||
assert data["output_config"]["effort"] == "high"
|
||||
|
||||
def test_max_stays_max_when_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._validate_effort_for_model",
|
||||
return_value=None,
|
||||
):
|
||||
cfg = AnthropicConfig()
|
||||
data: dict = {}
|
||||
optional_params = {"output_config": {"effort": "max"}}
|
||||
cfg._apply_output_config(data, "test-model", optional_params)
|
||||
assert data["output_config"]["effort"] == "max"
|
||||
|
||||
def test_no_output_config_is_noop(self):
|
||||
cfg = AnthropicConfig()
|
||||
data: dict = {}
|
||||
cfg._apply_output_config(data, "test-model", {})
|
||||
assert "output_config" not in data
|
||||
|
||||
def test_invalid_effort_value_still_raises(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model",
|
||||
return_value=True,
|
||||
):
|
||||
cfg = AnthropicConfig()
|
||||
with pytest.raises(litellm.exceptions.BadRequestError, match="Invalid effort value"):
|
||||
cfg._apply_output_config({}, "test-model", {"output_config": {"effort": "bogus"}})
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue