This commit is contained in:
kalavt 2026-09-23 14:40:52 +00:00 • committed by GitHub
commit 6749c0c54d
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 118 additions and 25 deletions

View file

@ -630,13 +630,22 @@ class AnthropicModelInfo(BaseLLMModelInfo):
return value if isinstance(value, bool) else None
@staticmethod
def _supports_model_capability(model: str, key: str, custom_llm_provider: str) -> bool:
def _supports_model_capability(
model: str, key: str, custom_llm_provider: str, base_model: str | None = None
) -> bool:
"""Check a boolean capability ``key`` in the model map under the caller's provider.
The provider-aware lookup is authoritative when it resolves an explicit flag,
so ``key: false`` on the provider-namespaced entry wins over every fallback.
Otherwise ``_supports_factory``'s provider-level fallbacks and the raw
model-map walk remain as backstops for alias forms the lookup misses.
``base_model`` is an opaque-``model``-string fallback (the same
``litellm_params.base_model`` Azure deployments use for model-type
detection): a Bedrock application inference profile ARN carries no
version substring, so every lookup above resolves nothing for it. When
the direct lookups are inconclusive and ``base_model`` names the actual
underlying model (e.g. ``claude-sonnet-5``), retry them against it.
"""
from litellm.utils import _supports_factory
@ -652,52 +661,67 @@ class AnthropicModelInfo(BaseLLMModelInfo):
return True
except Exception:
pass
return AnthropicModelInfo._get_model_capability(model, key) is True
if AnthropicModelInfo._get_model_capability(model, key) is True:
return True
if base_model is not None and base_model != model:
return AnthropicModelInfo._supports_model_capability(base_model, key, custom_llm_provider)
return False
@staticmethod
def _is_adaptive_thinking_model(model: str, custom_llm_provider: str) -> bool:
def _is_adaptive_thinking_model(model: str, custom_llm_provider: str, base_model: str | None = None) -> bool:
"""Whether ``model`` uses adaptive thinking (``output_config.effort``).
The model cost map is authoritative: an explicit ``supports_adaptive_thinking``
entry resolved under ``custom_llm_provider``, or a ``fallback_generalizations``
rule for unknown Claude models. The version gate (>= 4.6, including
provider-prefixed Bedrock/Vertex ids that map to no exact entry) lives entirely
in that declarative rule, not here.
in that declarative rule, not here. ``base_model`` is the opaque-id fallback —
see ``_supports_model_capability``.
"""
return AnthropicModelInfo._supports_model_capability(model, "supports_adaptive_thinking", custom_llm_provider)
return AnthropicModelInfo._supports_model_capability(
model, "supports_adaptive_thinking", custom_llm_provider, base_model=base_model
)
@staticmethod
def _is_always_on_thinking_model(model: str, custom_llm_provider: str) -> bool:
def _is_always_on_thinking_model(model: str, custom_llm_provider: str, base_model: str | None = None) -> bool:
"""Whether ``model`` always thinks and rejects ``thinking.type=disabled``
(Fable 5 / Mythos 5 generation). The model cost map is authoritative: an
explicit ``thinking_always_on`` entry resolved under ``custom_llm_provider``,
or a ``fallback_generalizations`` rule for unmapped ids of those families.
``base_model`` is the opaque-id fallback — see ``_supports_model_capability``.
"""
return AnthropicModelInfo._supports_model_capability(model, "thinking_always_on", custom_llm_provider)
return AnthropicModelInfo._supports_model_capability(
model, "thinking_always_on", custom_llm_provider, base_model=base_model
)
@staticmethod
def _supports_legacy_thinking(model: str, custom_llm_provider: str) -> bool:
def _supports_legacy_thinking(model: str, custom_llm_provider: str, base_model: str | None = None) -> bool:
"""Whether ``model`` is an adaptive-thinking model that still accepts legacy
``thinking.type=enabled`` with ``budget_tokens`` (the Claude 4.6 family).
The model cost map is authoritative: an explicit ``supports_legacy_thinking``
entry resolved under ``custom_llm_provider``, or a ``fallback_generalizations``
rule for unmapped 4.6 ids. Absent flag means the model rejects the legacy shape.
``base_model`` is the opaque-id fallback — see ``_supports_model_capability``.
"""
return AnthropicModelInfo._supports_model_capability(model, "supports_legacy_thinking", custom_llm_provider)
return AnthropicModelInfo._supports_model_capability(
model, "supports_legacy_thinking", custom_llm_provider, base_model=base_model
)
@staticmethod
def maybe_drop_disabled_thinking(
model: str,
optional_params: MutableMapping[str, object], # mutable-ok: in-place out-param, as in _maybe_drop_speed_param
custom_llm_provider: str,
base_model: str | None = None,
) -> None:
"""Omit ``thinking={'type': 'disabled'}`` for always-on-thinking models
(Fable 5 / Mythos 5), which 400 on it; omission is the API-documented
remedy and yields the model's default adaptive thinking."""
remedy and yields the model's default adaptive thinking. ``base_model``
is the opaque-id fallback — see ``_supports_model_capability``."""
thinking: Final = optional_params.get("thinking")
if not isinstance(thinking, dict) or thinking.get("type") != "disabled":
return
if not AnthropicModelInfo._is_always_on_thinking_model(model, custom_llm_provider):
if not AnthropicModelInfo._is_always_on_thinking_model(model, custom_llm_provider, base_model=base_model):
return
litellm.verbose_logger.warning(
DROP_DISABLED_THINKING_WARNING,
@ -710,17 +734,19 @@ class AnthropicModelInfo(BaseLLMModelInfo):
model: str,
optional_params: MutableMapping[str, object], # mutable-ok: in-place out-param like the sibling helpers
custom_llm_provider: str,
base_model: str | None = None,
) -> None:
"""Translate legacy ``thinking.type=enabled`` to adaptive for the
adaptive-thinking models that reject it (4.7+ and the 5 families).
Models flagged ``supports_legacy_thinking`` (the 4.6 family) accept the
legacy shape natively, so it is forwarded verbatim and the caller's
``budget_tokens`` cap keeps applying. Caller-provided
``output_config.effort`` is never overridden.
``output_config.effort`` is never overridden. ``base_model`` is the
opaque-id fallback — see ``_supports_model_capability``.
"""
if not AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider):
if not AnthropicModelInfo._is_adaptive_thinking_model(model, custom_llm_provider, base_model=base_model):
return
if AnthropicModelInfo._supports_legacy_thinking(model, custom_llm_provider):
if AnthropicModelInfo._supports_legacy_thinking(model, custom_llm_provider, base_model=base_model):
return
thinking: Final = optional_params.get("thinking")
if not isinstance(thinking, dict) or thinking.get("type") != "enabled":
@ -730,6 +756,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
model=model,
budget_tokens=int(thinking.get("budget_tokens") or 0),
custom_llm_provider=custom_llm_provider,
base_model=base_model,
)
existing_output_config: Final = optional_params.get("output_config")
optional_params["thinking"] = {"type": "adaptive"}
@ -739,9 +766,13 @@ class AnthropicModelInfo(BaseLLMModelInfo):
}
@staticmethod
def _legacy_budget_to_effort(model: str, budget_tokens: int, custom_llm_provider: str) -> str:
def _legacy_budget_to_effort(
model: str, budget_tokens: int, custom_llm_provider: str, base_model: str | None = None
) -> str:
if budget_tokens >= DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET and (
AnthropicModelInfo._supports_model_capability(model, "supports_xhigh_reasoning_effort", custom_llm_provider)
AnthropicModelInfo._supports_model_capability(
model, "supports_xhigh_reasoning_effort", custom_llm_provider, base_model=base_model
)
):
return "xhigh"
if budget_tokens >= DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET:

View file

@ -130,6 +130,12 @@ class AmazonConverseConfig(BaseConfig):
temperature: int | None
topP: int | None
topK: int | None
# Opaque-id fallback for adaptive-thinking capability resolution (an
# application inference profile ARN in `model` carries no version
# substring). Set by the caller (get_optional_params) before
# map_openai_params runs; not a constructor param, so map_openai_params's
# override stays signature-compatible with BaseConfig.
configured_base_model: str | None = None
def __init__(
self,
@ -508,7 +514,13 @@ class AmazonConverseConfig(BaseConfig):
}
}
def _handle_reasoning_effort_parameter(self, model: str, reasoning_effort: str, optional_params: dict) -> None:
def _handle_reasoning_effort_parameter(
self,
model: str,
reasoning_effort: str,
optional_params: dict,
base_model: str | None = None,
) -> None:
"""
Handle the reasoning_effort parameter based on the model type.
@ -516,7 +528,9 @@ class AmazonConverseConfig(BaseConfig):
- OpenAI GPT-5.x and GPT-6 models: mapped to ``reasoning.effort`` via additionalModelRequestFields.
- Nova 2 models: transformed to reasoningConfig.
- Anthropic models: mapped to ``thinking`` (and ``output_config.effort`` on
adaptive Claude 4.6 / 4.7).
adaptive Claude 4.6 / 4.7). ``base_model`` is the opaque-id fallback for
an application inference profile ARN — see
``AnthropicModelInfo._supports_model_capability``.
"""
if "gpt-oss" in model or "deepseek" in model:
optional_params["reasoning_effort"] = reasoning_effort
@ -538,7 +552,7 @@ class AmazonConverseConfig(BaseConfig):
optional_params.pop("output_config", None)
else:
optional_params["thinking"] = mapped_thinking
if AnthropicConfig._is_adaptive_thinking_model(model, "bedrock"):
if AnthropicConfig._is_adaptive_thinking_model(model, "bedrock", base_model=base_model):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(reasoning_effort)
if mapped_effort is None:
AnthropicConfig._raise_invalid_reasoning_effort(
@ -1005,6 +1019,14 @@ class AmazonConverseConfig(BaseConfig):
model: str,
drop_params: bool,
) -> dict:
# Application inference profile ARNs carry no version substring, so the
# adaptive-thinking capability lookup below resolves nothing for them.
# `configured_base_model` is the opaque-id fallback set by the caller
# (get_optional_params, from litellm_params.base_model) — same role as
# Azure's base_model, threaded via an instance attribute rather than a
# map_openai_params parameter so the override stays compatible with
# BaseConfig's shared signature.
base_model: Final = self.configured_base_model
is_thinking_enabled: Final = self.is_thinking_enabled(non_default_params)
base_model: Final = BedrockModelInfo.get_base_model(model)
drop_thinking_param: Final = self._is_deepseek_model(model=model, base_model=base_model)
@ -1084,7 +1106,7 @@ class AmazonConverseConfig(BaseConfig):
if (
isinstance(value, dict)
and value.get("type") == "adaptive"
and not AnthropicConfig._is_adaptive_thinking_model(model, "bedrock")
and not AnthropicConfig._is_adaptive_thinking_model(model, "bedrock", base_model=base_model)
):
max_tokens = non_default_params.get("max_completion_tokens") or non_default_params.get("max_tokens")
legacy_thinking = AnthropicConfig._map_reasoning_effort(
@ -1104,7 +1126,10 @@ class AmazonConverseConfig(BaseConfig):
else:
optional_params["thinking"] = value
AnthropicModelInfo.translate_legacy_thinking_for_adaptive_model(
model=model, optional_params=optional_params, custom_llm_provider="bedrock"
model=model,
optional_params=optional_params,
custom_llm_provider="bedrock",
base_model=base_model,
)
elif param == "reasoning_effort" and isinstance(value, str) and drop_reasoning_effort_param:
verbose_logger.debug(
@ -1113,7 +1138,7 @@ class AmazonConverseConfig(BaseConfig):
)
elif param == "reasoning_effort" and isinstance(value, str):
self._handle_reasoning_effort_parameter(
model=model, reasoning_effort=value, optional_params=optional_params
model=model, reasoning_effort=value, optional_params=optional_params, base_model=base_model
)
elif param == "output_config" and isinstance(value, dict):
mapped_output_config = dict(value)
@ -1552,6 +1577,7 @@ class AmazonConverseConfig(BaseConfig):
model: str,
headers: dict | None,
additional_request_params: dict,
configured_base_model: str | None = None,
) -> tuple[list[ToolBlock], list]:
"""Process tools and collect anthropic_beta values."""
bedrock_tools: list[ToolBlock] = []
@ -1668,7 +1694,7 @@ class AmazonConverseConfig(BaseConfig):
if (
isinstance(output_config, dict)
and output_config.get("effort") is not None
and not AnthropicConfig._is_adaptive_thinking_model(model, "bedrock")
and not AnthropicConfig._is_adaptive_thinking_model(model, "bedrock", base_model=configured_base_model)
):
from litellm.types.llms.anthropic import (
ANTHROPIC_EFFORT_BETA_HEADER,
@ -1750,10 +1776,20 @@ class AmazonConverseConfig(BaseConfig):
"has no thinking_blocks. The model won't use extended thinking for this turn."
)
# Application inference profile ARNs (litellm_params.model) carry no
# version substring, so the adaptive-thinking capability lookup below
# resolves nothing for them. litellm_params.base_model is the same
# opaque-id fallback Azure deployments use for model-type detection —
# thread it through so a chart/config pin (e.g. base_model:
# claude-sonnet-5) still gets the right adaptive-thinking behavior.
_raw_base_model: Final = litellm_params.get("base_model") if isinstance(litellm_params, Mapping) else None
_base_model: Final = _raw_base_model if isinstance(_raw_base_model, str) else None
AnthropicModelInfo.maybe_drop_disabled_thinking(
model=model,
optional_params=optional_params,
custom_llm_provider="bedrock",
base_model=_base_model,
)
# Prepare and separate parameters
@ -1768,7 +1804,7 @@ class AmazonConverseConfig(BaseConfig):
# Process tools and collect beta values
bedrock_tools, anthropic_beta_list = self._process_tools_and_beta(
original_tools, model, headers, additional_request_params
original_tools, model, headers, additional_request_params, configured_base_model=_base_model
)
# Append cachePoint to tools if cache_control_injection_points has tool_config

View file

@ -4650,7 +4650,12 @@ def get_optional_params(
bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model)
bedrock_base_model: Final = BedrockModelInfo.get_base_model(model)
if bedrock_route == "converse" or bedrock_route == "converse_like":
optional_params = litellm.AmazonConverseConfig().map_openai_params(
_converse_config: Final = litellm.AmazonConverseConfig()
# Set before the call, read via getattr inside map_openai_params —
# keeps the override signature-compatible with BaseConfig. See the
# comment on that method for why.
_converse_config.configured_base_model = base_model
optional_params = _converse_config.map_openai_params(
model=model,
non_default_params=non_default_params,
optional_params=optional_params,

View file

@ -2864,6 +2864,27 @@ def test_is_adaptive_thinking_model_is_sourced_from_cost_map(
assert AnthropicConfig._is_adaptive_thinking_model(model, "anthropic") is expected
def test_is_adaptive_thinking_model_falls_back_to_base_model_for_opaque_ids(
local_model_cost_map,
):
"""A Bedrock application inference profile ARN carries no version substring,
so every direct cost-map lookup for it resolves nothing. ``base_model`` is
the same opaque-id fallback Azure deployments use for model-type detection
(``litellm_params.base_model``); when set, it must be retried so a chart/config
pin still resolves adaptive-thinking correctly."""
opaque_model = "arn:aws:bedrock:ap-northeast-1:111111111111:application-inference-profile/abc123"
assert AnthropicConfig._is_adaptive_thinking_model(opaque_model, "bedrock") is False
assert (
AnthropicConfig._is_adaptive_thinking_model(opaque_model, "bedrock", base_model="claude-sonnet-5")
is True
)
assert (
AnthropicConfig._is_adaptive_thinking_model(opaque_model, "bedrock", base_model="claude-opus-4-5")
is False
)
def test_get_supported_params_includes_reasoning_for_sonnet_4_6_alias(
local_model_cost_map,
):