fix(anthropic): thread real provider through capability probes instead of pinning anthropic

This commit is contained in:
mateo-berri 2026-07-10 21:13:20 -07:00
parent f604034c17
commit 41b599b2d8
16 changed files with 221 additions and 79 deletions

View file

@ -335,23 +335,26 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
return any(v in model_lower for v in ("opus-4-7", "opus_4_7", "opus-4.7", "opus_4.7"))
@staticmethod
def _supports_effort_level(model: str, level: str) -> bool:
def _supports_effort_level(model: str, level: str, custom_llm_provider: str) -> bool:
"""Check ``supports_{level}_reasoning_effort`` in the model map."""
return AnthropicConfig._supports_model_capability(model, f"supports_{level}_reasoning_effort")
return AnthropicConfig._supports_model_capability(
model, f"supports_{level}_reasoning_effort", custom_llm_provider
)
@staticmethod
def _validate_effort_for_model(model: str, effort: Optional[str]) -> Optional[str]:
def _validate_effort_for_model(model: str, effort: Optional[str], custom_llm_provider: str) -> Optional[str]:
"""Return ``None`` if ``effort`` is allowed on ``model``, else an error message."""
if effort == "max" and not (
AnthropicConfig._is_adaptive_thinking_model(model) or AnthropicConfig._supports_effort_level(model, "max")
AnthropicConfig._is_adaptive_thinking_model(model, custom_llm_provider)
or AnthropicConfig._supports_effort_level(model, "max", custom_llm_provider)
):
return f"effort='max' is not supported by this model. Got model: {model}"
if effort == "xhigh" and not AnthropicConfig._supports_effort_level(model, "xhigh"):
if effort == "xhigh" and not AnthropicConfig._supports_effort_level(model, "xhigh", custom_llm_provider):
return f"effort='xhigh' is not supported by this model. Got model: {model}"
return None
@staticmethod
def _model_supports_effort_param(model: str) -> bool:
def _model_supports_effort_param(model: str, custom_llm_provider: str) -> bool:
"""Whether the model accepts ``output_config.effort`` at all.
A model qualifies if its map entry advertises ``supports_output_config``
@ -359,10 +362,10 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
signals: e.g. Claude Opus 4.5 supports ``output_config`` without
advertising a non-default (max/xhigh) effort level.
"""
if AnthropicConfig._supports_model_capability(model, "supports_output_config"):
if AnthropicConfig._supports_model_capability(model, "supports_output_config", custom_llm_provider):
return True
return any(
AnthropicConfig._supports_effort_level(model, level)
AnthropicConfig._supports_effort_level(model, level, custom_llm_provider)
for level in ("low", "minimal", "medium", "high", "xhigh", "max")
)
@ -451,7 +454,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
if (
"claude-3-7-sonnet" in model
or AnthropicConfig._is_adaptive_thinking_model(model)
or AnthropicConfig._is_adaptive_thinking_model(model, self.custom_llm_provider or "anthropic")
or supports_reasoning(
model=model,
custom_llm_provider=self.custom_llm_provider,
@ -1159,11 +1162,12 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
def _map_reasoning_effort(
reasoning_effort: Optional[Union[REASONING_EFFORT, str]],
model: str,
custom_llm_provider: str,
llm_provider: str = "anthropic",
) -> Optional[AnthropicThinkingParam]:
if reasoning_effort is None or reasoning_effort == "none":
return None
if AnthropicConfig._is_adaptive_thinking_model(model):
if AnthropicConfig._is_adaptive_thinking_model(model, custom_llm_provider):
return AnthropicThinkingParam(
type="adaptive",
)
@ -1471,6 +1475,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=effort_value,
model=model,
custom_llm_provider=self.custom_llm_provider or "anthropic",
llm_provider=self.custom_llm_provider or "anthropic",
)
if mapped_thinking is None:
@ -1478,7 +1483,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
optional_params.pop("output_config", None)
else:
optional_params["thinking"] = mapped_thinking
if AnthropicConfig._is_adaptive_thinking_model(model):
if AnthropicConfig._is_adaptive_thinking_model(model, self.custom_llm_provider or "anthropic"):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(effort_value)
if mapped_effort is None:
AnthropicConfig._raise_invalid_reasoning_effort(
@ -1902,7 +1907,9 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
output_config = optional_params.get("output_config")
if not output_config or not isinstance(output_config, dict):
return
if litellm.drop_params is True and not self._model_supports_effort_param(model):
if litellm.drop_params is True and not self._model_supports_effort_param(
model, self.custom_llm_provider or "anthropic"
):
litellm.verbose_logger.warning(
DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING,
model,
@ -1918,7 +1925,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
model=model,
llm_provider=self.custom_llm_provider or "anthropic",
)
gate_error = self._validate_effort_for_model(model, effort)
gate_error = self._validate_effort_for_model(model, effort, self.custom_llm_provider or "anthropic")
if gate_error is not None:
raise litellm.exceptions.BadRequestError(
message=gate_error,

View file

@ -360,18 +360,19 @@ class AnthropicModelInfo(BaseLLMModelInfo):
return value if isinstance(value, bool) else None
@staticmethod
def _supports_model_capability(model: str, key: str) -> bool:
"""Check a boolean capability ``key`` in the model map.
def _supports_model_capability(model: str, key: str, custom_llm_provider: str) -> bool:
"""Check a boolean capability ``key`` in the model map under the caller's provider.
Strips bedrock/vertex prefixes so a provider-routed Claude still
resolves to the Anthropic model-map entry.
The provider-aware lookup makes exact provider-namespaced entries (e.g. the
Bedrock ``global.anthropic.*`` ids) authoritative; the raw model-map walk
remains as a provider-less backstop for alias forms the lookup misses.
"""
from litellm.utils import _supports_factory
try:
if _supports_factory(
model=model,
custom_llm_provider="anthropic",
custom_llm_provider=custom_llm_provider,
key=key,
):
return True
@ -380,17 +381,24 @@ class AnthropicModelInfo(BaseLLMModelInfo):
return AnthropicModelInfo._get_model_capability(model, key) is True
@staticmethod
def _is_adaptive_thinking_model(model: str) -> bool:
def _is_adaptive_thinking_model(model: str, custom_llm_provider: str) -> bool:
"""Whether ``model`` uses adaptive thinking (``output_config.effort``).
The model cost map is authoritative: an explicit ``supports_adaptive_thinking``
entry, or a ``fallback_generalizations`` rule for unknown Claude models. The
version gate (>= 4.6, including provider-prefixed Bedrock/Vertex ids that map to
no exact entry) lives entirely in that declarative rule, not here.
entry resolved under ``custom_llm_provider``, or a ``fallback_generalizations``
rule for unknown Claude models. The version gate (>= 4.6, including
provider-prefixed Bedrock/Vertex ids that map to no exact entry) lives entirely
in that declarative rule, not here.
"""
return AnthropicModelInfo._supports_model_capability(model, "supports_adaptive_thinking")
return AnthropicModelInfo._supports_model_capability(model, "supports_adaptive_thinking", custom_llm_provider)
def is_effort_used(self, optional_params: Optional[dict], model: Optional[str] = None) -> bool:
def is_effort_used(
self,
optional_params: Optional[dict],
model: Optional[str] = None,
*,
custom_llm_provider: str,
) -> bool:
"""
Check if effort parameter is being used and requires a beta header.
@ -402,7 +410,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
return False
# Claude 4.6+ models use output_config as a stable API feature — no beta header needed
if model and self._is_adaptive_thinking_model(model):
if model and self._is_adaptive_thinking_model(model, custom_llm_provider):
return False
# Check if reasoning_effort is provided for Claude Opus 4.5
@ -483,6 +491,8 @@ class AnthropicModelInfo(BaseLLMModelInfo):
prompt_caching_set: bool = False,
file_id_used: bool = False,
mcp_server_used: bool = False,
*,
custom_llm_provider: str,
) -> List[str]:
"""
Get list of common beta headers based on the features that are active.
@ -495,7 +505,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
betas = []
# Detect features
effort_used = self.is_effort_used(optional_params, model)
effort_used = self.is_effort_used(optional_params, model, custom_llm_provider=custom_llm_provider)
if effort_used:
betas.append(ANTHROPIC_EFFORT_BETA_HEADER) # effort-2025-11-24
@ -651,7 +661,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
tool_search_used = self.is_tool_search_used(tools=tools)
programmatic_tool_calling_used = self.is_programmatic_tool_calling_used(tools=tools)
input_examples_used = self.is_input_examples_used(tools=tools)
effort_used = self.is_effort_used(optional_params=optional_params, model=model)
effort_used = self.is_effort_used(optional_params=optional_params, model=model, custom_llm_provider="anthropic")
code_execution_tool_used = self.is_code_execution_tool_used(tools=tools)
container_with_skills_used = self.is_container_with_skills_used(optional_params=optional_params)
user_anthropic_beta_headers = self._get_user_anthropic_beta_headers(

View file

@ -41,6 +41,10 @@ DROP_UNSUPPORTED_ADAPTIVE_EFFORT_WARNING = (
class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
@property
def custom_llm_provider(self) -> Optional[str]:
return "anthropic"
def get_supported_anthropic_messages_params(self, model: str) -> list:
return [
"messages",
@ -181,7 +185,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
return headers, api_base
@staticmethod
def _translate_reasoning_effort_to_anthropic(model: str, optional_params: Dict) -> None:
def _translate_reasoning_effort_to_anthropic(model: str, optional_params: Dict, custom_llm_provider: str) -> None:
"""Map OpenAI-style ``reasoning_effort`` to native Anthropic params.
Caller-supplied ``thinking`` / ``output_config`` win over the alias.
@ -198,7 +202,11 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
return
try:
mapped_thinking = AnthropicConfig._map_reasoning_effort(reasoning_effort=reasoning_effort, model=model)
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=reasoning_effort,
model=model,
custom_llm_provider=custom_llm_provider,
)
except _BadRequestError as e:
raise AnthropicError(message=str(e.message), status_code=400)
@ -208,7 +216,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
return
optional_params.setdefault("thinking", mapped_thinking)
if AnthropicModelInfo._is_adaptive_thinking_model(model):
if AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort)
if mapped_effort is None:
raise AnthropicError(
@ -219,7 +227,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
),
status_code=400,
)
gate_error = AnthropicConfig._validate_effort_for_model(model, mapped_effort)
gate_error = AnthropicConfig._validate_effort_for_model(model, mapped_effort, custom_llm_provider)
if gate_error is not None:
raise AnthropicError(message=gate_error, status_code=400)
existing_output_config = optional_params.get("output_config")
@ -229,13 +237,15 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
optional_params["output_config"] = existing_output_config
@staticmethod
def _translate_legacy_thinking_for_adaptive_model(model: str, optional_params: Dict) -> None:
def _translate_legacy_thinking_for_adaptive_model(
model: str, optional_params: Dict, custom_llm_provider: str
) -> None:
"""Translate legacy ``thinking.type=enabled`` to adaptive for 4.6/4.7.
Caller-provided ``output_config.effort`` is never overridden.
"""
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
if not AnthropicModelInfo._is_adaptive_thinking_model(model):
if not AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider):
return
thinking = optional_params.get("thinking")
if not isinstance(thinking, dict) or thinking.get("type") != "enabled":
@ -243,7 +253,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
budget = int(thinking.get("budget_tokens") or 0)
if budget >= DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET and (
AnthropicConfig._supports_effort_level(model, "xhigh")
AnthropicConfig._supports_effort_level(model, "xhigh", custom_llm_provider)
):
effort = "xhigh"
elif budget >= DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET:
@ -262,7 +272,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
@staticmethod
def _translate_adaptive_effort_for_non_adaptive_model(
model: str, optional_params: Dict, max_tokens: Optional[int]
model: str, optional_params: Dict, max_tokens: Optional[int], custom_llm_provider: str
) -> None:
"""Translate the 4.6+ adaptive-thinking interface (``thinking.type=adaptive``
and/or ``output_config.effort``) down to what an older Anthropic model
@ -305,7 +315,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
from litellm.exceptions import BadRequestError as _BadRequestError
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
if AnthropicConfig._is_adaptive_thinking_model(model):
if AnthropicConfig._is_adaptive_thinking_model(model, custom_llm_provider):
return
output_config = optional_params.get("output_config")
@ -315,17 +325,24 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
if effort is None and not adaptive_thinking:
return
if AnthropicConfig._model_supports_effort_param(model) and (
not adaptive_thinking or AnthropicConfig._validate_effort_for_model(model, effort) is None
if AnthropicConfig._model_supports_effort_param(model, custom_llm_provider) and (
not adaptive_thinking
or AnthropicConfig._validate_effort_for_model(model, effort, custom_llm_provider) is None
):
if adaptive_thinking:
optional_params.pop("thinking", None)
return
supports_thinking = AnthropicModelInfo._supports_model_capability(model, "supports_reasoning")
supports_thinking = AnthropicModelInfo._supports_model_capability(
model, "supports_reasoning", custom_llm_provider
)
try:
legacy_thinking = (
AnthropicConfig._map_reasoning_effort(reasoning_effort=effort or "medium", model=model)
AnthropicConfig._map_reasoning_effort(
reasoning_effort=effort or "medium",
model=model,
custom_llm_provider=custom_llm_provider,
)
if supports_thinking
else None
)
@ -389,17 +406,20 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
self._translate_reasoning_effort_to_anthropic(
model=model,
optional_params=anthropic_messages_optional_request_params,
custom_llm_provider=self.custom_llm_provider or "anthropic",
)
self._translate_legacy_thinking_for_adaptive_model(
model=model,
optional_params=anthropic_messages_optional_request_params,
custom_llm_provider=self.custom_llm_provider or "anthropic",
)
self._translate_adaptive_effort_for_non_adaptive_model(
model=model,
optional_params=anthropic_messages_optional_request_params,
max_tokens=max_tokens,
custom_llm_provider=self.custom_llm_provider or "anthropic",
)
system_param = anthropic_messages_optional_request_params.get("system")

View file

@ -423,6 +423,7 @@ class AmazonConverseConfig(BaseConfig):
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=reasoning_effort,
model=model,
custom_llm_provider="bedrock",
llm_provider="bedrock_converse",
)
if mapped_thinking is None:
@ -430,7 +431,7 @@ class AmazonConverseConfig(BaseConfig):
optional_params.pop("output_config", None)
else:
optional_params["thinking"] = mapped_thinking
if AnthropicConfig._is_adaptive_thinking_model(model):
if AnthropicConfig._is_adaptive_thinking_model(model, "bedrock"):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort)
if mapped_effort is None:
AnthropicConfig._raise_invalid_reasoning_effort(
@ -465,7 +466,7 @@ class AmazonConverseConfig(BaseConfig):
model=model,
llm_provider="bedrock_converse",
)
error = AnthropicConfig._validate_effort_for_model(model=model, effort=effort)
error = AnthropicConfig._validate_effort_for_model(model=model, effort=effort, custom_llm_provider="bedrock")
if error is not None:
raise litellm.exceptions.BadRequestError(
message=error,
@ -1279,7 +1280,7 @@ class AmazonConverseConfig(BaseConfig):
if anthropic_output_config is not None and isinstance(anthropic_output_config, dict):
if base_model.startswith("anthropic"):
if litellm.drop_params is True and not AnthropicConfig._model_supports_effort_param(model):
if litellm.drop_params is True and not AnthropicConfig._model_supports_effort_param(model, "bedrock"):
litellm.verbose_logger.warning(
DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING,
model,
@ -1422,7 +1423,7 @@ class AmazonConverseConfig(BaseConfig):
if (
isinstance(output_config, dict)
and output_config.get("effort") is not None
and not AnthropicConfig._is_adaptive_thinking_model(model)
and not AnthropicConfig._is_adaptive_thinking_model(model, "bedrock")
):
from litellm.types.llms.anthropic import (
ANTHROPIC_EFFORT_BETA_HEADER,

View file

@ -115,7 +115,7 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
keeps working. Non-adaptive models and models without a ceiling are
left untouched.
"""
if not AnthropicConfig._is_adaptive_thinking_model(model):
if not AnthropicConfig._is_adaptive_thinking_model(model, "bedrock"):
return
effort = params.get("reasoning_effort")
if not isinstance(effort, str):
@ -228,7 +228,7 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
custom_llm_provider="bedrock",
key="supports_output_config",
)
or AnthropicConfig._model_supports_effort_param(model)
or AnthropicConfig._model_supports_effort_param(model, "bedrock")
):
if anthropic_request.pop("output_config", None) is not None:
verbose_logger.warning(
@ -269,6 +269,7 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
prompt_caching_set=False,
file_id_used=self.is_file_id_used(messages),
mcp_server_used=self.is_mcp_server_used(optional_params.get("mcp_servers")),
custom_llm_provider="bedrock",
)
beta_set.update(auto_betas)

View file

@ -54,7 +54,9 @@ class BedrockClaudePlatformConfig(BedrockClaudePlatformMixin, AnthropicConfig):
tool_search_used=self.is_tool_search_used(tools=optional_params.get("tools")),
programmatic_tool_calling_used=self.is_programmatic_tool_calling_used(tools=optional_params.get("tools")),
input_examples_used=self.is_input_examples_used(tools=optional_params.get("tools")),
effort_used=self.is_effort_used(optional_params=optional_params, model=model),
effort_used=self.is_effort_used(
optional_params=optional_params, model=model, custom_llm_provider="anthropic"
),
user_anthropic_beta_headers=self._get_user_anthropic_beta_headers(
anthropic_beta_header=headers.get("anthropic-beta")
),

View file

@ -77,6 +77,10 @@ class AmazonAnthropicClaudeMessagesConfig(
DEFAULT_BEDROCK_ANTHROPIC_API_VERSION = "bedrock-2023-05-31"
@property
def custom_llm_provider(self) -> Optional[str]:
return "bedrock"
BEDROCK_INVOKE_ALLOWED_TOP_LEVEL_FIELDS = frozenset(BedrockInvokeAnthropicMessagesRequest.__annotations__.keys())
def __init__(self, **kwargs):
@ -269,7 +273,7 @@ class AmazonAnthropicClaudeMessagesConfig(
Returns:
True if the model supports extended thinking on Bedrock
"""
if AnthropicModelInfo._is_adaptive_thinking_model(model):
if AnthropicModelInfo._is_adaptive_thinking_model(model, "bedrock"):
return True
model_lower = model.lower()
@ -319,7 +323,7 @@ class AmazonAnthropicClaudeMessagesConfig(
if not self._supports_extended_thinking_on_bedrock(model):
return False
is_adaptive_thinking_model = AnthropicModelInfo._is_adaptive_thinking_model(model)
is_adaptive_thinking_model = AnthropicModelInfo._is_adaptive_thinking_model(model, "bedrock")
thinking = anthropic_messages_request.get("thinking")
if isinstance(thinking, dict):
@ -596,6 +600,7 @@ class AmazonAnthropicClaudeMessagesConfig(
mcp_server_used=anthropic_model_info.is_mcp_server_used(
anthropic_messages_optional_request_params.get("mcp_servers")
),
custom_llm_provider="bedrock",
)
beta_set.update(auto_betas)
@ -662,7 +667,7 @@ class AmazonAnthropicClaudeMessagesConfig(
path degrades ``xhigh`` -> ``max`` rather than 400-ing. Non-adaptive models
and models without a ceiling are left untouched.
"""
if not AnthropicModelInfo._is_adaptive_thinking_model(model):
if not AnthropicModelInfo._is_adaptive_thinking_model(model, "bedrock"):
return
effort = optional_params.get("reasoning_effort")
if not isinstance(effort, str):
@ -750,7 +755,7 @@ class AmazonAnthropicClaudeMessagesConfig(
custom_llm_provider="bedrock",
key="supports_output_config",
)
or AnthropicConfig._model_supports_effort_param(model)
or AnthropicConfig._model_supports_effort_param(model, "bedrock")
):
if anthropic_messages_request.pop("output_config", None) is not None:
verbose_logger.warning(
@ -787,7 +792,7 @@ class AmazonAnthropicClaudeMessagesConfig(
if (
litellm.drop_params is True
and "output_config" in anthropic_messages_request
and not AnthropicConfig._model_supports_effort_param(model)
and not AnthropicConfig._model_supports_effort_param(model, "bedrock")
):
verbose_logger.warning(
DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING,

View file

@ -372,6 +372,7 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=reasoning_effort_value,
model=model,
custom_llm_provider="databricks",
llm_provider="databricks",
)
if mapped_thinking is None:
@ -379,7 +380,7 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
optional_params.pop("output_config", None)
else:
optional_params["thinking"] = mapped_thinking
if AnthropicConfig._is_adaptive_thinking_model(model):
if AnthropicConfig._is_adaptive_thinking_model(model, "databricks"):
mapped_effort: Optional[str] = None
if isinstance(reasoning_effort_value, str):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort_value)

View file

@ -26,7 +26,7 @@ def _model_accepts_output_config_effort(model: str) -> bool:
"""
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
return AnthropicConfig._model_supports_effort_param(model)
return AnthropicConfig._model_supports_effort_param(model, "vertex_ai")
def sanitize_vertex_anthropic_output_params(data: dict, model: str) -> None:

View file

@ -112,6 +112,7 @@ class VertexAIAnthropicConfig(AnthropicConfig):
prompt_caching_set=self.is_cache_control_set(messages),
file_id_used=self.is_file_id_used(messages),
mcp_server_used=self.is_mcp_server_used(optional_params.get("mcp_servers")),
custom_llm_provider="vertex_ai",
)
beta_set = set(auto_betas)

View file

@ -12,39 +12,39 @@ class TestMapReasoningEffort:
def test_none_returns_none_for_opus_4_6(self):
"""reasoning_effort=None should return None for Opus 4.6, not adaptive."""
result = AnthropicConfig._map_reasoning_effort(
reasoning_effort=None, model="claude-opus-4-6"
reasoning_effort=None, model="claude-opus-4-6", custom_llm_provider="anthropic"
)
assert result is None
def test_none_returns_none_for_other_models(self):
"""reasoning_effort=None should return None for non-Opus models."""
result = AnthropicConfig._map_reasoning_effort(
reasoning_effort=None, model="claude-4-sonnet-20250514"
reasoning_effort=None, model="claude-4-sonnet-20250514", custom_llm_provider="anthropic"
)
assert result is None
def test_opus_4_6_returns_adaptive_for_low(self):
result = AnthropicConfig._map_reasoning_effort(
reasoning_effort="low", model="claude-opus-4-6"
reasoning_effort="low", model="claude-opus-4-6", custom_llm_provider="anthropic"
)
assert result["type"] == "adaptive"
def test_opus_4_6_returns_adaptive_for_high(self):
result = AnthropicConfig._map_reasoning_effort(
reasoning_effort="high", model="claude-opus-4-6"
reasoning_effort="high", model="claude-opus-4-6", custom_llm_provider="anthropic"
)
assert result["type"] == "adaptive"
def test_other_model_low_returns_enabled_with_budget(self):
result = AnthropicConfig._map_reasoning_effort(
reasoning_effort="low", model="claude-4-sonnet-20250514"
reasoning_effort="low", model="claude-4-sonnet-20250514", custom_llm_provider="anthropic"
)
assert result["type"] == "enabled"
assert "budget_tokens" in result
def test_other_model_high_returns_enabled_with_budget(self):
result = AnthropicConfig._map_reasoning_effort(
reasoning_effort="high", model="claude-4-sonnet-20250514"
reasoning_effort="high", model="claude-4-sonnet-20250514", custom_llm_provider="anthropic"
)
assert result["type"] == "enabled"
assert "budget_tokens" in result
@ -52,13 +52,13 @@ class TestMapReasoningEffort:
def test_none_string_returns_none_for_opus_4_6(self):
"""reasoning_effort='none' should return None for Opus 4.6."""
result = AnthropicConfig._map_reasoning_effort(
reasoning_effort="none", model="claude-opus-4-6"
reasoning_effort="none", model="claude-opus-4-6", custom_llm_provider="anthropic"
)
assert result is None
def test_none_string_returns_none_for_other_models(self):
"""reasoning_effort='none' should return None for non-Opus models."""
result = AnthropicConfig._map_reasoning_effort(
reasoning_effort="none", model="claude-4-sonnet-20250514"
reasoning_effort="none", model="claude-4-sonnet-20250514", custom_llm_provider="anthropic"
)
assert result is None

View file

@ -520,8 +520,8 @@ def test_shipped_adaptive_rule_gates_on_version_not_pricing(shipped_cost_map):
non_adaptive = "us.anthropic.claude-opus-4-20250514"
assert adaptive not in litellm.model_cost
assert non_adaptive not in litellm.model_cost
assert AnthropicModelInfo._is_adaptive_thinking_model(adaptive) is True
assert AnthropicModelInfo._is_adaptive_thinking_model(non_adaptive) is False
assert AnthropicModelInfo._is_adaptive_thinking_model(adaptive, "anthropic") is True
assert AnthropicModelInfo._is_adaptive_thinking_model(non_adaptive, "anthropic") is False
def test_shipped_rules_resolve_unmapped_future_bedrock_claude_with_both_flags(shipped_cost_map):

View file

@ -1661,7 +1661,7 @@ def test_effort_beta_header_injection():
# Test with effort parameter
optional_params = {"output_config": {"effort": "low"}}
effort_used = model_info.is_effort_used(optional_params=optional_params)
effort_used = model_info.is_effort_used(optional_params=optional_params, custom_llm_provider="anthropic")
assert effort_used is True
headers = model_info.get_anthropic_headers(
@ -1877,7 +1877,7 @@ def test_anthropic_drop_params_false_forwards_to_unsupported_model():
],
)
def test_anthropic_model_supports_effort_param_recognizes_supporting_models(model):
assert AnthropicConfig._model_supports_effort_param(model) is True
assert AnthropicConfig._model_supports_effort_param(model, "anthropic") is True
@pytest.mark.parametrize(
@ -1890,7 +1890,7 @@ def test_anthropic_model_supports_effort_param_recognizes_supporting_models(mode
],
)
def test_anthropic_model_supports_effort_param_rejects_non_supporting_models(model):
assert AnthropicConfig._model_supports_effort_param(model) is False
assert AnthropicConfig._model_supports_effort_param(model, "anthropic") is False
@pytest.mark.parametrize(
@ -2217,7 +2217,7 @@ def test_get_config_does_not_leak_module_constants():
)
def test_supports_effort_level_handles_provider_prefixes(model, level, expected):
"""``_supports_effort_level`` resolves bedrock/vertex/azure-prefixed model ids."""
assert AnthropicConfig._supports_effort_level(model, level) is expected
assert AnthropicConfig._supports_effort_level(model, level, "anthropic") is expected
@pytest.mark.parametrize(
@ -2239,7 +2239,7 @@ def test_supports_effort_level_handles_provider_prefixes(model, level, expected)
def test_validate_effort_for_model_centralises_per_model_gating(
model, effort, expect_error
):
err = AnthropicConfig._validate_effort_for_model(model, effort)
err = AnthropicConfig._validate_effort_for_model(model, effort, "anthropic")
if expect_error:
assert err is not None
assert effort in err
@ -2490,7 +2490,7 @@ def test_is_adaptive_thinking_model_is_sourced_from_cost_map(
fallback for ids the cost map cannot resolve. The dated Claude 4.0 names stay
non-adaptive because the date suffix is not read as a minor version, while 4.8/4.9/5.x
are covered without a code change."""
assert AnthropicConfig._is_adaptive_thinking_model(model) is expected
assert AnthropicConfig._is_adaptive_thinking_model(model, "anthropic") is expected
def test_get_supported_params_includes_reasoning_for_sonnet_4_6_alias(
@ -2836,6 +2836,7 @@ def test_effort_beta_header_not_injected_for_46_models():
result = model_info.is_effort_used(
optional_params={"output_config": {"effort": "high"}},
model=model,
custom_llm_provider="anthropic",
)
assert result is False, f"is_effort_used should return False for {model}"
@ -2947,6 +2948,7 @@ def test_effort_beta_header_still_injected_for_older_models():
result = model_info.is_effort_used(
optional_params={"output_config": {"effort": "low"}},
model="claude-opus-4-5-20251101",
custom_llm_provider="anthropic",
)
assert result is True

View file

@ -1578,7 +1578,7 @@ class TestClaudeOpus48AdaptiveThinking:
def test_adaptive_thinking_detected_for_opus_4_8(self, local_model_cost_map, model):
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
assert AnthropicModelInfo._is_adaptive_thinking_model(model) is True
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is True
def test_resolver_reads_flag_through_bedrock_invoke_prefix(
self, local_model_cost_map
@ -1592,6 +1592,7 @@ class TestClaudeOpus48AdaptiveThinking:
AnthropicModelInfo._supports_model_capability(
"bedrock/invoke/us.anthropic.claude-opus-4-8",
"supports_adaptive_thinking",
"anthropic",
)
is True
)
@ -1609,7 +1610,7 @@ class TestClaudeOpus48AdaptiveThinking:
def test_adaptive_thinking_detected_for_fable_5(self, local_model_cost_map, model):
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
assert AnthropicModelInfo._is_adaptive_thinking_model(model) is True
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is True
@pytest.mark.parametrize(
"model",
@ -1644,7 +1645,7 @@ class TestClaudeOpus48AdaptiveThinking:
version (``4.6`` -> ``4-6``)."""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
assert AnthropicModelInfo._is_adaptive_thinking_model(model) is True
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is True
@pytest.mark.parametrize(
"model",
@ -1665,7 +1666,7 @@ class TestClaudeOpus48AdaptiveThinking:
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
assert model not in litellm.model_cost
assert AnthropicModelInfo._is_adaptive_thinking_model(model) is False
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is False
@pytest.mark.parametrize(
"model",
@ -1693,7 +1694,7 @@ class TestClaudeOpus48AdaptiveThinking:
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
assert model not in litellm.model_cost
assert AnthropicModelInfo._is_adaptive_thinking_model(model) is True
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is True
@pytest.mark.parametrize(
"model",
@ -1716,7 +1717,7 @@ class TestClaudeOpus48AdaptiveThinking:
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
assert model not in litellm.model_cost
assert AnthropicModelInfo._is_adaptive_thinking_model(model) is False
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is False
@pytest.mark.parametrize(
"model",
@ -1725,7 +1726,7 @@ class TestClaudeOpus48AdaptiveThinking:
def test_non_adaptive_models_not_detected(self, local_model_cost_map, model):
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
assert AnthropicModelInfo._is_adaptive_thinking_model(model) is False
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is False
class TestDefaultSuffixAdaptiveThinking:
@ -1750,7 +1751,7 @@ class TestDefaultSuffixAdaptiveThinking:
) -> None:
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
assert AnthropicModelInfo._is_adaptive_thinking_model(model) is True, (
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is True, (
f"{model} not classified as adaptive thinking. "
"Check _model_map_lookup_candidates strips @default suffix."
)
@ -1771,3 +1772,51 @@ class TestDefaultSuffixAdaptiveThinking:
assert expected_bare in candidates, (
f"Expected '{expected_bare}' in candidates for '{model}', got: {candidates}"
)
class TestCapabilityProbeUsesCallerProvider:
"""``_supports_model_capability`` must probe under the caller's real provider
namespace instead of a pinned ``"anthropic"``. With the pin, the exact Bedrock
cost-map entry for ``global.anthropic.claude-opus-4-8`` was rejected by the
provider match and the anthropic-scoped fallback rule answered instead, so
flipping ``supports_adaptive_thinking`` on the exact entry changed nothing and
the documented "exact entry beats rule" precedence was silently violated."""
BEDROCK_MODEL = "global.anthropic.claude-opus-4-8"
def test_exact_bedrock_entry_flag_is_authoritative_for_bedrock_caller(
self, local_model_cost_map, monkeypatch
):
import litellm
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
assert (
AnthropicModelInfo._is_adaptive_thinking_model(self.BEDROCK_MODEL, "bedrock")
is True
)
monkeypatch.setitem(
litellm.model_cost[self.BEDROCK_MODEL], "supports_adaptive_thinking", False
)
litellm.get_model_info.cache_clear()
assert (
AnthropicModelInfo._is_adaptive_thinking_model(self.BEDROCK_MODEL, "bedrock")
is False
)
def test_native_anthropic_probe_still_reads_anthropic_entry(
self, local_model_cost_map, monkeypatch
):
import litellm
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
monkeypatch.setitem(
litellm.model_cost[self.BEDROCK_MODEL], "supports_adaptive_thinking", False
)
litellm.get_model_info.cache_clear()
assert (
AnthropicModelInfo._is_adaptive_thinking_model("claude-opus-4-8", "anthropic")
is True
)

View file

@ -2472,3 +2472,46 @@ def test_filter_and_transform_beta_headers_passes_context_management_for_bedrock
)
assert out_converse == []
def test_bedrock_messages_thinking_shape_follows_exact_bedrock_entry_flag(
local_model_cost_map, monkeypatch
):
"""The outbound thinking payload must follow the exact Bedrock cost-map entry.
Before threading the caller's provider through the capability probes, the probe
was pinned to ``"anthropic"``: the exact ``global.anthropic.claude-opus-4-8``
entry was rejected by the provider match and the anthropic-scoped fallback rule
forced ``thinking.type='adaptive'`` even with ``supports_adaptive_thinking``
explicitly set to ``false`` on the entry."""
import litellm
from litellm.types.router import GenericLiteLLMParams
model = "global.anthropic.claude-opus-4-8"
cfg = AmazonAnthropicClaudeMessagesConfig()
def transform():
return cfg.transform_anthropic_messages_request(
model=model,
messages=[{"role": "user", "content": [{"type": "text", "text": "Hello"}]}],
anthropic_messages_optional_request_params={
"max_tokens": 4096,
"reasoning_effort": "medium",
},
litellm_params=GenericLiteLLMParams(),
headers={},
)
result = transform()
assert result.get("thinking") == {"type": "adaptive"}
assert result.get("output_config") == {"effort": "medium"}
monkeypatch.setitem(litellm.model_cost[model], "supports_adaptive_thinking", False)
litellm.get_model_info.cache_clear()
flipped = transform()
thinking = flipped.get("thinking")
assert isinstance(thinking, dict)
assert thinking.get("type") == "enabled"
assert isinstance(thinking.get("budget_tokens"), int)
assert "output_config" not in flipped

View file

@ -204,7 +204,7 @@ def test_adaptive_thinking_detected_for_fable_5(local_model_cost_map, model):
maps to ``thinking.type='adaptive'`` + ``output_config.effort``."""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
assert AnthropicModelInfo._is_adaptive_thinking_model(model) is True
assert AnthropicModelInfo._is_adaptive_thinking_model(model, "anthropic") is True
@pytest.mark.parametrize(