mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
Merge 372fc14486 into f4308bc124
This commit is contained in:
commit
7547b33e77
7 changed files with 498 additions and 75 deletions
|
|
@ -104,6 +104,7 @@ from ..common_utils import (
|
|||
requires_native_compaction_beta,
|
||||
strip_advisor_blocks_from_messages,
|
||||
)
|
||||
from ..pass_through.utils import normalize_reasoning_effort_value
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
|
@ -414,6 +415,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
return f"effort='xhigh' is not supported by this model. Got model: {model}"
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def degrade_alias_effort_for_model(model: str, effort: str, custom_llm_provider: str) -> str:
|
||||
"""Keep an alias-derived effort the gate accepts, else lower it to a tier the model is known to accept.
|
||||
|
||||
Explicit ``output_config.effort`` must not be routed here: a caller naming a native tier gets a 400.
|
||||
"""
|
||||
if AnthropicConfig._validate_effort_for_model(model, effort, custom_llm_provider) is None:
|
||||
return effort
|
||||
return normalize_reasoning_effort_value(effort, model, custom_llm_provider)
|
||||
|
||||
@staticmethod
|
||||
def _model_supports_effort_param(model: str, custom_llm_provider: str) -> bool:
|
||||
"""Whether the model accepts ``output_config.effort`` at all.
|
||||
|
|
@ -1264,32 +1275,33 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
type="adaptive",
|
||||
display="summarized",
|
||||
)
|
||||
elif reasoning_effort == "low":
|
||||
resolved_effort: Final = normalize_reasoning_effort_value(reasoning_effort, model, custom_llm_provider)
|
||||
if resolved_effort == "low":
|
||||
return AnthropicThinkingParam(
|
||||
type="enabled",
|
||||
budget_tokens=DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET,
|
||||
)
|
||||
elif reasoning_effort == "medium":
|
||||
elif resolved_effort == "medium":
|
||||
return AnthropicThinkingParam(
|
||||
type="enabled",
|
||||
budget_tokens=DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
|
||||
)
|
||||
elif reasoning_effort == "high":
|
||||
elif resolved_effort == "high":
|
||||
return AnthropicThinkingParam(
|
||||
type="enabled",
|
||||
budget_tokens=DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
|
||||
)
|
||||
elif reasoning_effort == "xhigh":
|
||||
elif resolved_effort == "xhigh":
|
||||
return AnthropicThinkingParam(
|
||||
type="enabled",
|
||||
budget_tokens=DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
|
||||
)
|
||||
elif reasoning_effort == "max":
|
||||
elif resolved_effort == "max":
|
||||
return AnthropicThinkingParam(
|
||||
type="enabled",
|
||||
budget_tokens=DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET,
|
||||
)
|
||||
elif reasoning_effort == "minimal":
|
||||
elif resolved_effort == "minimal":
|
||||
return AnthropicThinkingParam(
|
||||
type="enabled",
|
||||
budget_tokens=max(
|
||||
|
|
@ -1622,7 +1634,11 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
value=effort_value,
|
||||
llm_provider=self._resolved_provider,
|
||||
)
|
||||
optional_params["output_config"] = {"effort": mapped_effort}
|
||||
optional_params["output_config"] = {
|
||||
"effort": AnthropicConfig.degrade_alias_effort_for_model(
|
||||
model, mapped_effort, self._resolved_provider
|
||||
)
|
||||
}
|
||||
elif param == "web_search_options" and isinstance(value, dict):
|
||||
hosted_web_search_tool = self.map_web_search_tool(cast(OpenAIWebSearchOptions, value))
|
||||
self._add_tools_to_optional_params(optional_params=optional_params, tools=[hosted_web_search_tool])
|
||||
|
|
|
|||
|
|
@ -333,14 +333,29 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
),
|
||||
status_code=400,
|
||||
)
|
||||
gate_error: Final = AnthropicConfig._validate_effort_for_model(model, mapped_effort, custom_llm_provider)
|
||||
existing_output_config: Final = optional_params.get("output_config")
|
||||
explicit_effort: Final = AnthropicMessagesConfig._explicit_output_config_effort(existing_output_config)
|
||||
resolved_effort: Final = (
|
||||
explicit_effort
|
||||
if explicit_effort is not None
|
||||
else AnthropicConfig.degrade_alias_effort_for_model(model, mapped_effort, custom_llm_provider)
|
||||
)
|
||||
gate_error: Final = AnthropicConfig._validate_effort_for_model(model, resolved_effort, custom_llm_provider)
|
||||
if gate_error is not None:
|
||||
raise AnthropicError(message=gate_error, status_code=400)
|
||||
existing_output_config = optional_params.get("output_config")
|
||||
if not isinstance(existing_output_config, dict):
|
||||
existing_output_config = {}
|
||||
existing_output_config.setdefault("effort", mapped_effort)
|
||||
optional_params["output_config"] = existing_output_config
|
||||
optional_params["output_config"] = (
|
||||
{**existing_output_config, "effort": resolved_effort}
|
||||
if isinstance(existing_output_config, dict)
|
||||
else {"effort": resolved_effort}
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _explicit_output_config_effort(output_config: object) -> str | None:
|
||||
match output_config:
|
||||
case {"effort": str() as effort}:
|
||||
return effort
|
||||
case _:
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _translate_adaptive_effort_for_non_adaptive_model(
|
||||
|
|
|
|||
|
|
@ -74,6 +74,9 @@ def normalize_reasoning_effort_value(
|
|||
The accepted set is resolved by the same owner that answers ``/model_group/info``, so a level
|
||||
the proxy advertises is a level this path forwards.
|
||||
|
||||
Only a known capability set can refuse a tier: a model the map does not describe, or an entry
|
||||
declaring no effort metadata, keeps the requested tier instead of being silently downgraded.
|
||||
|
||||
A deployment that refuses every step of a chain falls back to an accepted level read off that
|
||||
same set rather than to an assumed one, since an entry naming its levels outright can exclude
|
||||
the tiers the per-level flags treat as unconditional. ``none`` is never that fallback and is
|
||||
|
|
@ -91,9 +94,11 @@ def normalize_reasoning_effort_value(
|
|||
try:
|
||||
model_info: Final[ModelInfo] = get_model_info(model=model, custom_llm_provider=custom_llm_provider)
|
||||
except Exception:
|
||||
return chain[-1]
|
||||
return effort
|
||||
|
||||
supported: Final = resolve_supported_reasoning_efforts(model_info, deployment_is_mapped=True)
|
||||
supported: Final = resolve_supported_reasoning_efforts(model_info, deployment_is_mapped=False)
|
||||
if supported is None:
|
||||
return effort
|
||||
if not supported:
|
||||
return chain[-1]
|
||||
|
||||
|
|
|
|||
|
|
@ -631,7 +631,8 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
|
||||
@staticmethod
|
||||
def _clamp_adaptive_reasoning_effort_for_bedrock(model: str, optional_params: dict) -> None:
|
||||
"""Lower ``reasoning_effort`` to the Bedrock effort ceiling before validation.
|
||||
"""Lower ``reasoning_effort`` and an explicit ``output_config.effort`` to the Bedrock effort ceiling
|
||||
before validation.
|
||||
|
||||
The shared ``/v1/messages`` effort gate rejects tiers a model does not
|
||||
natively support (e.g. ``xhigh`` on Opus 4.6). Bedrock's chat paths instead
|
||||
|
|
@ -648,6 +649,14 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
clamped: Final = {"effort": effort}
|
||||
normalize_bedrock_opus_output_config_effort(model=model, output_config=clamped)
|
||||
optional_params["reasoning_effort"] = clamped["effort"]
|
||||
explicit_effort: Final = AnthropicMessagesConfig._explicit_output_config_effort(
|
||||
optional_params.get("output_config")
|
||||
)
|
||||
if explicit_effort is None:
|
||||
return
|
||||
clamped_explicit: Final = {"effort": explicit_effort}
|
||||
normalize_bedrock_opus_output_config_effort(model=model, output_config=clamped_explicit)
|
||||
optional_params["output_config"] = {**optional_params["output_config"], **clamped_explicit}
|
||||
|
||||
def transform_anthropic_messages_request(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -1,5 +1,7 @@
|
|||
"""Tests for ``reasoning_effort`` translation on the Anthropic /v1/messages route."""
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.constants import (
|
||||
|
|
@ -27,9 +29,7 @@ from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_tran
|
|||
("max", "max"),
|
||||
],
|
||||
)
|
||||
def test_reasoning_effort_maps_to_output_config_for_adaptive_model(
|
||||
reasoning_effort, expected_effort
|
||||
):
|
||||
def test_reasoning_effort_maps_to_output_config_for_adaptive_model(reasoning_effort, expected_effort):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": reasoning_effort}
|
||||
|
||||
|
|
@ -107,27 +107,25 @@ def test_invalid_reasoning_effort_raises_400(bad_effort):
|
|||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,bad_effort",
|
||||
"model,requested_effort,expected_effort",
|
||||
[
|
||||
("claude-opus-4-6", "xhigh"),
|
||||
("claude-sonnet-4-6", "xhigh"),
|
||||
("claude-opus-4-6", "xhigh", "high"),
|
||||
("claude-sonnet-4-6", "xhigh", "high"),
|
||||
],
|
||||
)
|
||||
def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort):
|
||||
def test_reasoning_effort_unsupported_tier_degrades_on_messages(model, requested_effort, expected_effort):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort}
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": requested_effort}
|
||||
|
||||
with pytest.raises(AnthropicError) as exc_info:
|
||||
config.transform_anthropic_messages_request(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
result = config.transform_anthropic_messages_request(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "not supported by this model" in str(exc_info.value)
|
||||
assert result["output_config"]["effort"] == expected_effort
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
@ -139,9 +137,7 @@ def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort
|
|||
("invoke/us.anthropic.claude-opus-4-7", "xhigh", "xhigh"),
|
||||
],
|
||||
)
|
||||
def test_bedrock_invoke_messages_clamps_effort_to_ceiling(
|
||||
local_model_cost_map, model, effort, expected_effort
|
||||
):
|
||||
def test_bedrock_invoke_messages_clamps_effort_to_ceiling(local_model_cost_map, model, effort, expected_effort):
|
||||
"""Bedrock Invoke /v1/messages degrades effort to the model's ceiling.
|
||||
|
||||
Claude Code "goal mode" sends ``xhigh``; Opus 4.6 must clamp to ``max``
|
||||
|
|
@ -162,22 +158,40 @@ def test_bedrock_invoke_messages_clamps_effort_to_ceiling(
|
|||
assert result["thinking"]["type"] == "adaptive"
|
||||
|
||||
|
||||
def test_bedrock_invoke_messages_rejects_xhigh_without_ceiling(local_model_cost_map):
|
||||
"""Sonnet 4.6 on Bedrock has no effort ceiling, so xhigh is still rejected."""
|
||||
def test_bedrock_invoke_messages_clamps_explicit_effort_sent_with_alias(local_model_cost_map):
|
||||
config = AmazonAnthropicClaudeMessagesConfig()
|
||||
explicit_output_config = {"effort": "xhigh"}
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
"reasoning_effort": "xhigh",
|
||||
"output_config": explicit_output_config,
|
||||
}
|
||||
|
||||
result = config.transform_anthropic_messages_request(
|
||||
model="invoke/us.anthropic.claude-opus-4-6-v1",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["output_config"]["effort"] == "max"
|
||||
assert explicit_output_config == {"effort": "xhigh"}
|
||||
|
||||
|
||||
def test_bedrock_invoke_messages_degrades_xhigh_without_ceiling(local_model_cost_map):
|
||||
config = AmazonAnthropicClaudeMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": "xhigh"}
|
||||
|
||||
with pytest.raises(AnthropicError) as exc_info:
|
||||
config.transform_anthropic_messages_request(
|
||||
model="invoke/us.anthropic.claude-sonnet-4-6",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
result = config.transform_anthropic_messages_request(
|
||||
model="invoke/us.anthropic.claude-sonnet-4-6",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "not supported by this model" in str(exc_info.value)
|
||||
assert result["output_config"]["effort"] == "high"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
@ -187,9 +201,7 @@ def test_bedrock_invoke_messages_rejects_xhigh_without_ceiling(local_model_cost_
|
|||
"bedrock/invoke/us.anthropic.claude-sonnet-4-6",
|
||||
],
|
||||
)
|
||||
def test_reasoning_effort_max_accepted_on_sonnet_46_messages(
|
||||
local_model_cost_map, model
|
||||
):
|
||||
def test_reasoning_effort_max_accepted_on_sonnet_46_messages(local_model_cost_map, model):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": "max"}
|
||||
|
||||
|
|
@ -205,6 +217,46 @@ def test_reasoning_effort_max_accepted_on_sonnet_46_messages(
|
|||
assert isinstance(output_config, dict) and output_config.get("effort") == "max"
|
||||
|
||||
|
||||
def test_explicit_unsupported_output_config_effort_is_rejected_not_rewritten(local_model_cost_map):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
"reasoning_effort": "high",
|
||||
"output_config": {"effort": "xhigh"},
|
||||
}
|
||||
|
||||
with pytest.raises(AnthropicError) as exc_info:
|
||||
config.transform_anthropic_messages_request(
|
||||
model="claude-sonnet-4-6",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 400
|
||||
assert "xhigh" in str(exc_info.value)
|
||||
|
||||
|
||||
def test_explicit_supported_output_config_effort_wins_over_unsupported_alias(local_model_cost_map):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
"reasoning_effort": "xhigh",
|
||||
"output_config": {"effort": "low"},
|
||||
}
|
||||
|
||||
result = config.transform_anthropic_messages_request(
|
||||
model="claude-sonnet-4-6",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["output_config"] == {"effort": "low"}
|
||||
|
||||
|
||||
def test_explicit_output_config_wins_over_reasoning_effort():
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
|
|
@ -247,9 +299,7 @@ def test_explicit_thinking_wins_over_reasoning_effort():
|
|||
|
||||
def test_reasoning_effort_in_supported_params():
|
||||
config = AnthropicMessagesConfig()
|
||||
assert "reasoning_effort" in config.get_supported_anthropic_messages_params(
|
||||
"claude-opus-4-7"
|
||||
)
|
||||
assert "reasoning_effort" in config.get_supported_anthropic_messages_params("claude-opus-4-7")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
@ -318,9 +368,7 @@ def test_legacy_thinking_high_budget_keeps_xhigh_when_supported():
|
|||
"bedrock/invoke/us.anthropic.claude-opus-4-8",
|
||||
],
|
||||
)
|
||||
def test_legacy_thinking_translates_to_adaptive_for_opus_48(
|
||||
model, local_model_cost_map
|
||||
):
|
||||
def test_legacy_thinking_translates_to_adaptive_for_opus_48(model, local_model_cost_map):
|
||||
"""Regression for issue #29188: Opus 4.8 requires adaptive thinking, but the
|
||||
legacy ``thinking.type='enabled'`` shape was passed through unchanged for
|
||||
Bedrock 4.8 (its cost-map entry lacked ``supports_adaptive_thinking`` and the
|
||||
|
|
@ -352,9 +400,7 @@ def test_legacy_thinking_translates_to_adaptive_for_opus_48(
|
|||
("claude-newfamily-6", "high"),
|
||||
],
|
||||
)
|
||||
def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models(
|
||||
local_model_cost_map, model, expected_effort
|
||||
):
|
||||
def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models(local_model_cost_map, model, expected_effort):
|
||||
"""The 5 families reject ``thinking.type=enabled``, so the adaptive translation
|
||||
stays the safe default for every adaptive model not flagged
|
||||
``supports_legacy_thinking``, unmapped future ids included. An unmapped id
|
||||
|
|
@ -389,9 +435,7 @@ def test_legacy_thinking_translates_to_adaptive_for_5_and_future_models(
|
|||
(1, "low"),
|
||||
],
|
||||
)
|
||||
def test_legacy_thinking_budget_buckets_on_opus_48(
|
||||
local_model_cost_map, budget_tokens, expected_effort
|
||||
):
|
||||
def test_legacy_thinking_budget_buckets_on_opus_48(local_model_cost_map, budget_tokens, expected_effort):
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
|
|
@ -478,9 +522,7 @@ def test_legacy_thinking_left_untouched_on_non_adaptive_model():
|
|||
("claude-sonnet-4-5", False),
|
||||
],
|
||||
)
|
||||
def test_disabled_thinking_omitted_for_always_on_models_messages(
|
||||
local_model_cost_map, model, expected_dropped
|
||||
):
|
||||
def test_disabled_thinking_omitted_for_always_on_models_messages(local_model_cost_map, model, expected_dropped):
|
||||
"""/v1/messages: ``thinking={"type": "disabled"}`` is omitted for always-on-thinking
|
||||
models and forwarded verbatim for models that accept it."""
|
||||
config = AnthropicMessagesConfig()
|
||||
|
|
@ -498,3 +540,58 @@ def test_disabled_thinking_omitted_for_always_on_models_messages(
|
|||
assert "thinking" not in result
|
||||
else:
|
||||
assert result["thinking"] == {"type": "disabled"}
|
||||
|
||||
|
||||
def _mock_model_info(**flags):
|
||||
return flags
|
||||
|
||||
|
||||
def test_xhigh_degrades_to_high_for_non_adaptive_model():
|
||||
with (
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model",
|
||||
return_value=False,
|
||||
),
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
),
|
||||
):
|
||||
optional_params = {"reasoning_effort": "xhigh"}
|
||||
AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic(
|
||||
model="unknown-glm-4.6",
|
||||
optional_params=optional_params,
|
||||
max_tokens=None,
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert optional_params["thinking"]["type"] == "enabled"
|
||||
assert optional_params["thinking"]["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
assert "output_config" not in optional_params
|
||||
|
||||
|
||||
def test_max_degrades_to_high_for_non_adaptive_model():
|
||||
with (
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.common_utils.AnthropicModelInfo._is_adaptive_thinking_model",
|
||||
return_value=False,
|
||||
),
|
||||
patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_max_reasoning_effort=False,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
),
|
||||
):
|
||||
optional_params = {"reasoning_effort": "max"}
|
||||
AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic(
|
||||
model="unknown-deepseek",
|
||||
optional_params=optional_params,
|
||||
max_tokens=None,
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert optional_params["thinking"]["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ Covers:
|
|||
|
||||
import json
|
||||
import os
|
||||
from typing import Any, Dict
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -23,7 +23,7 @@ from litellm.router_utils.reasoning_effort_capability import (
|
|||
from litellm.utils import get_model_info
|
||||
|
||||
|
||||
def _load_model_registry() -> Dict[str, Any]:
|
||||
def _load_model_registry() -> dict[str, Any]:
|
||||
"""Load the root model_prices_and_context_window.json."""
|
||||
json_path = os.path.join(
|
||||
os.path.dirname(__file__),
|
||||
|
|
@ -141,9 +141,43 @@ class TestNormalizeReasoningEffortValue:
|
|||
def test_a_tier_outside_any_chain_passes_through(self, local_model_cost_map, effort):
|
||||
assert normalize_reasoning_effort_value(effort, "claude-opus-4-7", "anthropic") == effort
|
||||
|
||||
@pytest.mark.parametrize("effort, expected", [("max", "high"), ("xhigh", "high"), ("minimal", "low")])
|
||||
def test_a_model_the_map_does_not_describe_keeps_the_floor(self, local_model_cost_map, effort, expected):
|
||||
assert normalize_reasoning_effort_value(effort, "totally-made-up-model-xyz", "openai") == expected
|
||||
@pytest.mark.parametrize("effort", ["max", "xhigh", "minimal"])
|
||||
def test_a_model_the_map_does_not_describe_keeps_the_requested_tier(self, local_model_cost_map, effort):
|
||||
assert normalize_reasoning_effort_value(effort, "totally-made-up-model-xyz", "openai") == effort
|
||||
|
||||
@staticmethod
|
||||
def _register_deployment(model_info: dict[str, object]) -> str:
|
||||
router = litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "compat",
|
||||
"litellm_params": {"model": "anthropic/compat-reasoner-1", "api_key": "fake-key"},
|
||||
"model_info": model_info,
|
||||
}
|
||||
]
|
||||
)
|
||||
return router.model_list[0]["model_info"]["id"]
|
||||
|
||||
@pytest.mark.parametrize("effort", ["max", "xhigh", "minimal"])
|
||||
def test_a_registered_deployment_without_effort_metadata_keeps_the_requested_tier(
|
||||
self, local_model_cost_map, effort
|
||||
):
|
||||
deployment_id = self._register_deployment({})
|
||||
assert normalize_reasoning_effort_value(effort, deployment_id, "anthropic") == effort
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_info, effort, expected",
|
||||
[
|
||||
({"supports_reasoning": False}, "max", "high"),
|
||||
({"supports_reasoning": True, "supports_xhigh_reasoning_effort": False}, "xhigh", "high"),
|
||||
({"supports_reasoning": True, "supports_minimal_reasoning_effort": False}, "minimal", "low"),
|
||||
],
|
||||
)
|
||||
def test_a_registered_deployment_declaring_a_tier_unsupported_degrades(
|
||||
self, local_model_cost_map, model_info, effort, expected
|
||||
):
|
||||
deployment_id = self._register_deployment(model_info)
|
||||
assert normalize_reasoning_effort_value(effort, deployment_id, "anthropic") == expected
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
|
@ -161,9 +195,7 @@ class TestAdapterAdaptiveThinking:
|
|||
)
|
||||
|
||||
adapter = LiteLLMAnthropicMessagesAdapter()
|
||||
result = adapter.translate_anthropic_thinking_to_reasoning_effort(
|
||||
{"type": "adaptive"}
|
||||
)
|
||||
result = adapter.translate_anthropic_thinking_to_reasoning_effort({"type": "adaptive"})
|
||||
assert result == "medium"
|
||||
|
||||
def test_messages_adapter_adaptive_overridden_by_output_config(self):
|
||||
|
|
|
|||
|
|
@ -5,8 +5,19 @@ Verifies that reasoning_effort=None returns None for all models,
|
|||
including Claude Opus 4.6.
|
||||
"""
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm.exceptions
|
||||
from litellm.constants import (
|
||||
DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET,
|
||||
DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
|
||||
)
|
||||
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
||||
|
||||
|
||||
|
|
@ -74,3 +85,241 @@ class TestMapReasoningEffort:
|
|||
reasoning_effort="none", model="claude-4-sonnet-20250514", custom_llm_provider="anthropic"
|
||||
)
|
||||
assert result is None
|
||||
|
||||
|
||||
def _mock_model_info(**flags):
|
||||
return flags
|
||||
|
||||
|
||||
class TestMapReasoningEffortDegradation:
|
||||
def test_max_stays_max_when_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_max_reasoning_effort=True,
|
||||
supports_xhigh_reasoning_effort=True,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="max",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["type"] == "enabled"
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET
|
||||
|
||||
def test_max_degrades_to_xhigh_when_only_xhigh_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_max_reasoning_effort=False,
|
||||
supports_xhigh_reasoning_effort=True,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="max",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET
|
||||
|
||||
def test_max_degrades_to_high_when_neither_max_nor_xhigh_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_max_reasoning_effort=False,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="max",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
|
||||
def test_max_passthrough_for_unknown_model(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="max",
|
||||
model="unknown-glm-4.6",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET
|
||||
|
||||
def test_xhigh_stays_xhigh_when_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_xhigh_reasoning_effort=True,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="xhigh",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET
|
||||
|
||||
def test_xhigh_degrades_to_high_when_unsupported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_xhigh_reasoning_effort=False,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="xhigh",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
|
||||
def test_xhigh_passthrough_for_unknown_model(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="xhigh",
|
||||
model="unknown-deepseek",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET
|
||||
|
||||
def test_minimal_stays_minimal_when_supported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_minimal_reasoning_effort=True,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="minimal",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == max(DEFAULT_REASONING_EFFORT_MINIMAL_THINKING_BUDGET, 1024)
|
||||
|
||||
def test_minimal_degrades_to_low_when_unsupported(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
return_value=_mock_model_info(
|
||||
supports_reasoning=True,
|
||||
supports_minimal_reasoning_effort=False,
|
||||
),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="minimal",
|
||||
model="test-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET
|
||||
|
||||
def test_high_unchanged(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="high",
|
||||
model="unknown-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
|
||||
def test_medium_unchanged(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="medium",
|
||||
model="unknown-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET
|
||||
|
||||
def test_low_unchanged(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.utils.get_model_info",
|
||||
side_effect=Exception("model not found"),
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="low",
|
||||
model="unknown-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["budget_tokens"] == DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET
|
||||
|
||||
def test_none_returns_none(self):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="none",
|
||||
model="any-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result is None
|
||||
|
||||
def test_none_value_returns_none(self):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort=None,
|
||||
model="any-model",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result is None
|
||||
|
||||
def test_adaptive_model_short_circuits_before_degradation(self):
|
||||
with patch( # test-quality-ok: capability flags live on get_model_info; HTTP cannot isolate the degrade chain
|
||||
"litellm.llms.anthropic.chat.transformation.AnthropicConfig._is_adaptive_thinking_model",
|
||||
return_value=True,
|
||||
):
|
||||
result = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort="max",
|
||||
model="claude-opus-4-6",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result["type"] == "adaptive"
|
||||
|
||||
|
||||
class TestReasoningEffortAliasOutputConfig:
|
||||
@staticmethod
|
||||
def _transform_alias(reasoning_effort: str) -> dict:
|
||||
config = AnthropicConfig()
|
||||
optional_params = config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": reasoning_effort},
|
||||
optional_params={},
|
||||
model="claude-sonnet-4-6",
|
||||
drop_params=False,
|
||||
)
|
||||
return config.transform_request(
|
||||
model="claude-sonnet-4-6",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={**optional_params, "max_tokens": 1024},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
def test_unsupported_alias_tier_degrades_to_an_accepted_one(self, local_model_cost_map):
|
||||
assert self._transform_alias("xhigh")["output_config"] == {"effort": "high"}
|
||||
|
||||
def test_supported_alias_tier_is_kept(self, local_model_cost_map):
|
||||
assert self._transform_alias("max")["output_config"] == {"effort": "max"}
|
||||
|
||||
def test_explicit_unsupported_output_config_effort_is_rejected_not_rewritten(self, local_model_cost_map):
|
||||
with pytest.raises(litellm.exceptions.BadRequestError, match="xhigh"):
|
||||
AnthropicConfig().transform_request(
|
||||
model="claude-sonnet-4-6",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
optional_params={"max_tokens": 1024, "output_config": {"effort": "xhigh"}},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue