refactor: remove unnecessary comments from #27074

Strip out the explanatory and historical comments that don't carry
business-logic justification. Comments that simply narrate what code
does — or that explain prior behavior, what was changed, or which PR
introduced a fix — are removed. Docstrings are reduced to a one-line
summary where the long form repeated information already evident from
the code or test data.

No code-behavior changes. All 643 affected unit tests still pass.

Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
Cursor Agent 2026-05-04 19:34:56 +00:00
parent 56070b86a3
commit 2cb3f0f027
No known key found for this signature in database
28 changed files with 92 additions and 739 deletions

View file

@ -202,17 +202,6 @@ DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET = int(
DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET = int(
os.getenv("DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET", 4096)
)
# ``xhigh`` / ``max`` budget extrapolation for legacy ``thinking.budget_tokens``
# models (Claude 4.5 series + haiku). Continues the 2× progression
# 1024 → 2048 → 4096 from the existing low/medium/high tiers. These tiers
# also exist as adaptive ``output_config.effort`` enum values on Claude 4.6+
# / 4.7; this constant only governs the budget-tokens fallback for models
# that aren't on the adaptive path. Per
# https://platform.claude.com/docs/en/build-with-claude/effort the ``effort``
# enum is gated by model, but the legacy ``budget_tokens`` knob accepts any
# integer up to the model's max_tokens — adopting #27051's mapping here lets
# ``reasoning_effort=xhigh|max`` Just Work as a unified OpenAI-format knob
# regardless of which Anthropic API surface implements it.
DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET = int(
os.getenv("DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET", 8192)
)
@ -416,14 +405,7 @@ BEDROCK_MAX_POLICY_SIZE = int(os.getenv("BEDROCK_MAX_POLICY_SIZE", 75))
BEDROCK_MIN_THINKING_BUDGET_TOKENS = int(
os.getenv("BEDROCK_MIN_THINKING_BUDGET_TOKENS", 1024)
)
# Anthropic's Messages API rejects ``thinking.budget_tokens < 1024`` with a
# 400. ``reasoning_effort='minimal'`` historically mapped to 128 (the global
# default) which always 400'd against direct Anthropic, Azure AI Anthropic,
# Vertex AI Anthropic, and Bedrock Invoke. Floor at the provider minimum so
# ``minimal`` is a usable tier on every Anthropic-backed route; Bedrock
# Converse already clamps server-side, this just unifies the behavior.
# Constant — not env-overridable — because it tracks Anthropic's published
# wire-protocol minimum, not a tunable.
# Anthropic's Messages API rejects thinking.budget_tokens < 1024.
ANTHROPIC_MIN_THINKING_BUDGET_TOKENS = 1024
REPLICATE_POLLING_DELAY_SECONDS = float(
os.getenv("REPLICATE_POLLING_DELAY_SECONDS", 0.5)

View file

@ -224,11 +224,8 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
def _supports_effort_level(model: str, level: str) -> bool:
"""Check ``supports_{level}_reasoning_effort`` in the model map.
Mirrors the pattern used in ``openai/chat/gpt_5_transformation.py`` so
that adding support for a new effort level is a pure model-map change.
Handles bedrock-prefixed and vertex-prefixed model ids by stripping
the prefix and re-checking against ``litellm.model_cost`` directly,
so a Bedrock-routed Claude 4.6/4.7 keeps its model-map flag.
Strips bedrock/vertex prefixes so a provider-routed Claude still
resolves to the Anthropic model-map entry.
"""
key = f"supports_{level}_reasoning_effort"
try:
@ -240,11 +237,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
return True
except Exception:
pass
# Bedrock and Vertex route the model id with a provider-prefix
# (e.g. ``bedrock/invoke/us.anthropic.claude-opus-4-7``). Strip
# known prefixes and look the resulting Anthropic-flavoured key
# up directly in ``litellm.model_cost`` so the lookup keeps
# working regardless of which route the request arrived on.
candidates = [model]
for prefix in (
"bedrock/converse/",
@ -277,27 +269,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
@staticmethod
def _validate_effort_for_model(model: str, effort: Optional[str]) -> Optional[str]:
"""Return ``None`` if ``effort`` is allowed on ``model``, else an error message.
Centralises per-model gating for ``max`` and ``xhigh`` so the chat
completion path (``_apply_output_config``) and the /v1/messages
pass-through (``AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic``)
can't drift when a new model tier is added. Caller raises the
provider-appropriate exception type using the returned message.
``max`` is supported on Claude 4.6 (Opus + Sonnet) and Claude 4.7
adaptive-thinking models per
https://platform.claude.com/docs/en/build-with-claude/effort. The
data-driven ``supports_max_reasoning_effort`` flag in
``model_prices_and_context_window.json`` is the source of truth;
family-level ``_is_claude_4_6_model`` / ``_is_claude_4_7_model``
checks remain as a fallback for OpenRouter / GitHub Copilot /
Vercel / Bedrock variants whose model-map entries don't yet carry
the flag.
``xhigh`` is purely data-driven via ``supports_xhigh_reasoning_effort``
so enabling it for a new model is a model-map-only change.
"""
"""Return ``None`` if ``effort`` is allowed on ``model``, else an error message."""
if effort == "max" and not (
AnthropicConfig._is_claude_4_6_model(model)
or AnthropicConfig._is_claude_4_7_model(model)
@ -312,15 +284,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
@staticmethod
def _model_supports_effort_param(model: str) -> bool:
"""Whether the model accepts ``output_config.effort`` at all.
Per https://platform.claude.com/docs/en/build-with-claude/effort the
``output_config.effort`` parameter is supported on Opus 4.5+, Sonnet 4.6+
and Mythos Preview; older Claude models reject it with a 400. Support is
encoded in ``model_prices_and_context_window.json`` via the
``supports_*_reasoning_effort`` flags, so adding a new effort-capable
model is a pure model-map change.
"""
"""Whether the model accepts ``output_config.effort`` at all."""
for level in ("low", "minimal", "medium", "high", "xhigh", "max"):
if AnthropicConfig._supports_effort_level(model, level):
return True
@ -908,16 +872,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
model: str,
llm_provider: str = "anthropic",
) -> Optional[AnthropicThinkingParam]:
"""Map an OpenAI-format ``reasoning_effort`` string to Anthropic's
``thinking`` payload.
Raises ``BadRequestError`` (clean 400) instead of ``ValueError`` (500)
on unmapped efforts so every caller — Anthropic native, Bedrock
Invoke/Converse, Databricks, Vertex Anthropic, Azure AI Anthropic,
and the experimental ``/v1/messages`` pass-through — surfaces a
consistent error to the user. Pass ``llm_provider`` so the
``BadRequestError`` carries the right provider name in logs.
"""
if reasoning_effort is None or reasoning_effort == "none":
return None
if AnthropicConfig._is_adaptive_thinking_model(model):
@ -940,32 +894,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
budget_tokens=DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
)
elif reasoning_effort == "xhigh":
# Continues the 2× progression of low/medium/high (1024/2048/4096).
# On adaptive models (Claude 4.6/4.7) the ``xhigh`` tier is
# already routed via ``output_config.effort=xhigh`` above; this
# branch only applies to budget-mode models (Claude 4.5 series +
# haiku) where the OpenAI-format ``reasoning_effort`` knob would
# otherwise 400 with ``Unmapped reasoning effort``. Keeps the
# cross-model UX uniform — ``reasoning_effort=xhigh`` Just Works
# regardless of which Anthropic API surface implements it.
return AnthropicThinkingParam(
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
)
elif reasoning_effort == "max":
# Same rationale as ``xhigh`` above — ``max`` is the adaptive
# enum's top tier on Claude 4.6/4.7, but for budget-mode models
# we extend the 2× progression (8192 → 16384) so the OpenAI-
# format alias is usable on every Claude model.
return AnthropicThinkingParam(
type="enabled",
budget_tokens=DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET,
)
elif reasoning_effort == "minimal":
# Anthropic Messages API rejects ``budget_tokens < 1024`` with a
# 400. Floor at the provider minimum so ``minimal`` is a usable
# tier on Anthropic / Azure AI Anthropic / Vertex AI Anthropic /
# Bedrock Invoke. Bedrock Converse already clamps server-side.
return AnthropicThinkingParam(
type="enabled",
budget_tokens=max(
@ -1246,10 +1184,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
elif param == "thinking":
optional_params["thinking"] = value
elif param == "reasoning_effort" and isinstance(value, str):
# ``_map_reasoning_effort`` raises ``BadRequestError`` (400)
# directly on unmapped efforts (``disabled`` / ``invalid`` /
# ``""`` / ``xhigh``/``max`` on budget-mode Claude 4.5) so
# we no longer need to wrap a ``ValueError`` here.
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=value,
model=model,
@ -1260,20 +1194,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
optional_params.pop("output_config", None)
else:
optional_params["thinking"] = mapped_thinking
# For Claude 4.6+ adaptive-thinking models, effort is
# controlled via ``output_config``, not
# ``thinking.budget_tokens``. Driven by
# ``supports_adaptive_thinking`` in the model map so
# adding a new adaptive Claude is a model-map-only change.
if AnthropicConfig._is_adaptive_thinking_model(model):
# ``_map_reasoning_effort`` returns ``type=adaptive``
# for any string on adaptive models without checking
# the value, so reject unmapped efforts here (matching
# the /v1/messages path) instead of relying on the
# downstream ``_apply_output_config`` check. Co-locating
# validation with the mapping prevents garbage from
# leaking into ``optional_params`` if ``map_openai_params``
# is ever called without a subsequent ``transform_request``.
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
value
)
@ -1703,22 +1624,12 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
def _apply_output_config(
self, data: dict, model: str, optional_params: dict
) -> None:
"""Validate and apply output_config to the request data.
Validation errors raise ``BadRequestError`` (clean 400) so callers
passing ``effort="disabled"`` / ``effort=""`` / unsupported tiers
for the model see a client-side error rather than a 500.
"""
"""Validate and apply output_config to the request data."""
if "output_config" not in optional_params:
return
output_config = optional_params.get("output_config")
if not output_config or not isinstance(output_config, dict):
return
# When ``drop_params`` is set, strip ``output_config`` for models that
# cannot accept it (e.g. proxy fronting Claude Code at haiku-3, where
# the client always sends effort but the model rejects it). The user
# opted into silent fixup via the global flag — log a warning so the
# strip is still visible in logs.
if litellm.drop_params is True and not self._model_supports_effort_param(model):
litellm.verbose_logger.warning(
"Dropping unsupported `output_config` for model=%s "
@ -1730,10 +1641,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
data.pop("output_config", None)
return
effort = output_config.get("effort")
# ``effort=""`` (empty string) and unmapped strings should be treated
# as invalid, not silently passed through. We use ``effort is not None``
# here so empty string fails the membership check below. (The legacy
# ``if effort and ...`` short-circuit silently accepted ``""``.)
valid_efforts = ["high", "medium", "low", "xhigh", "max"]
if effort is not None and effort not in valid_efforts:
raise litellm.exceptions.BadRequestError(
@ -1744,10 +1651,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
model=model,
llm_provider=self.custom_llm_provider or "anthropic",
)
# Per-model gating for ``max`` / ``xhigh`` is centralised in
# ``_validate_effort_for_model`` so the chat path and the
# /v1/messages pass-through stay in lock-step when a new model
# tier lands.
gate_error = self._validate_effort_for_model(model, effort)
if gate_error is not None:
raise litellm.exceptions.BadRequestError(

View file

@ -273,15 +273,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
@staticmethod
def _is_adaptive_thinking_model(model: str) -> bool:
"""Claude 4.6+ models use adaptive thinking with ``output_config.effort``.
Driven by the ``supports_adaptive_thinking`` flag in
``model_prices_and_context_window.json`` so that adding a new
adaptive-thinking model is a pure model-map change. Falls back to
the family-pattern check for OpenRouter / Vercel / Bedrock /
provider-prefixed variants whose model-map entries don't (yet)
carry the flag.
"""
"""Claude 4.6+ models use adaptive thinking with ``output_config.effort``."""
from litellm.utils import _supports_factory
try:

View file

@ -47,9 +47,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
"inference_geo",
"speed",
"output_config",
# OpenAI-style tier knob — translated to native ``thinking`` +
# ``output_config`` in ``transform_anthropic_messages_request``
# and popped before the request is forwarded.
"reasoning_effort",
# TODO: Add Anthropic `metadata` support
# "metadata",
@ -176,22 +173,10 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
) -> None:
"""Map OpenAI-style ``reasoning_effort`` to native Anthropic params.
The /v1/messages spec doesn't include ``reasoning_effort`` — without
this translation it gets silently dropped, leaving every adaptive
tier collapsed to the same behavior on Bedrock Invoke /v1/messages
(and on Anthropic / Azure AI / Vertex AI when callers pass it on
the messages route). Mirrors ``AnthropicConfig.map_openai_params``
on the chat completion path so the two routes can't drift.
- Pops ``reasoning_effort`` from ``optional_params`` so it never
reaches the wire.
- Caller-supplied ``thinking`` / ``output_config`` always win — we
don't override an explicit native value.
- Effort=``none`` clears thinking + output_config so callers can
opt out per request.
- Invalid efforts raise ``BadRequestError`` (clean 400) instead of
surfacing as 500s downstream.
Caller-supplied ``thinking`` / ``output_config`` win over the alias.
``effort='none'`` clears both. Invalid efforts raise a 400.
"""
from litellm.exceptions import BadRequestError as _BadRequestError
from litellm.llms.anthropic.chat.transformation import (
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT,
AnthropicConfig,
@ -201,12 +186,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
if not isinstance(reasoning_effort, str):
return
# ``_map_reasoning_effort`` raises ``BadRequestError`` (400) directly
# on unmapped efforts. The /v1/messages pass-through surfaces errors
# as ``AnthropicError``; convert here so callers see a provider-shaped
# 400 rather than the LiteLLM-shaped one.
from litellm.exceptions import BadRequestError as _BadRequestError
try:
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=reasoning_effort, model=model
@ -224,12 +203,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
reasoning_effort
)
# ``_map_reasoning_effort`` returns ``type=adaptive`` for any
# string on adaptive models without checking the value. The
# chat completion path validates the resolved effort downstream
# via ``_apply_output_config``; /v1/messages has no equivalent
# downstream check, so reject unmapped values here so callers
# see a clean 400 instead of a 500 from the provider.
if mapped_effort is None:
raise AnthropicError(
message=(
@ -239,10 +212,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
),
status_code=400,
)
# Per-model gating for ``max`` / ``xhigh`` is centralised in
# ``AnthropicConfig._validate_effort_for_model`` so the chat
# completion path and this /v1/messages pass-through stay in
# lock-step when a new model tier lands.
gate_error = AnthropicConfig._validate_effort_for_model(
model, mapped_effort
)

View file

@ -15,21 +15,11 @@ if TYPE_CHECKING:
def _promote_extra_body_to_optional_params(optional_params: dict) -> None:
"""Promote anthropic-native passthrough keys out of ``extra_body``.
``azure_ai`` is registered in ``litellm.openai_compatible_providers``, so
``add_provider_specific_params_to_optional_params`` (litellm/utils.py)
auto-stuffs any non-OpenAI kwarg (e.g. ``output_config={"effort": "..."}``)
into ``optional_params["extra_body"]``. For the Azure→Anthropic route the
user's intent is to forward those params to Anthropic, so promote them to
the top level of ``optional_params`` so:
* ``AnthropicConfig._apply_output_config`` validates ``effort`` values
(matching the native ``anthropic`` provider's 400 on bad efforts).
* Valid passthroughs (e.g. ``output_config``, ``thinking``) actually
reach the request body instead of being silently dropped by the
``data.pop("extra_body", None)`` strip below.
``setdefault`` is used so an explicit top-level value is never clobbered
by a duplicate inside ``extra_body``.
``azure_ai`` is an OpenAI-compatible provider, so non-OpenAI kwargs like
``output_config`` get auto-routed into ``extra_body`` by
``add_provider_specific_params_to_optional_params``. For the Azure→Anthropic
route those keys must reach the request body and be validated, so promote
them. ``setdefault`` keeps explicit top-level values authoritative.
"""
extra_body = optional_params.get("extra_body")
if not isinstance(extra_body, dict) or not extra_body:
@ -66,9 +56,6 @@ class AzureAnthropicConfig(AnthropicConfig):
1. API key via 'api-key' header
2. Azure AD token via 'Authorization: Bearer <token>' header
"""
# Promote anthropic-native passthrough keys (``output_config``,
# ``thinking``, ``mcp_servers``, ...) out of ``extra_body`` so the
# flag detection below (``is_mcp_server_used``, etc.) sees them.
_promote_extra_body_to_optional_params(optional_params)
# Convert dict to GenericLiteLLMParams if needed
@ -133,18 +120,8 @@ class AzureAnthropicConfig(AnthropicConfig):
Transform request using parent AnthropicConfig, then remove unsupported params.
Azure Anthropic doesn't support extra_body, max_retries, or stream_options parameters.
"""
# Promote anthropic-native passthrough keys (``output_config``,
# ``thinking``, ...) out of ``extra_body`` BEFORE delegating to
# ``AnthropicConfig.transform_request``. Without this:
# * ``output_config`` is silently dropped by the ``extra_body`` pop
# below, and
# * ``_apply_output_config`` never validates ``effort`` values, so
# ``effort="invalid"`` quietly reaches the model with default
# behavior instead of returning a clean 400 (as the native
# ``anthropic`` provider does).
_promote_extra_body_to_optional_params(optional_params)
# Call parent transform_request
data = super().transform_request(
model=model,
messages=messages,

View file

@ -84,12 +84,6 @@ class BaseConfig(ABC):
@classmethod
def get_config(cls):
# Subclasses lean on this to surface their public default settings
# (e.g. ``max_tokens``) as request params. Anything ``_``-prefixed is
# treated as private (lookup tables, ABC machinery, internal flags)
# and must not leak into the wire body — a tuple/dict/frozenset class
# attribute would otherwise serialise into the request as an extra
# top-level key and the provider would 400 it.
return {
k: v
for k, v in cls.__dict__.items()

View file

@ -189,8 +189,6 @@ class AmazonConverseConfig(BaseConfig):
@classmethod
def get_config(cls):
# ``_``-prefixed names are private (lookup tables, ABC machinery,
# internal flags) and must not leak into the wire body.
return {
k: v
for k, v in cls.__dict__.items()
@ -415,59 +413,19 @@ class AmazonConverseConfig(BaseConfig):
"""
Handle the reasoning_effort parameter based on the model type.
Different model families handle reasoning effort differently:
- GPT-OSS models: Keep reasoning_effort as-is (passed to additionalModelRequestFields)
- Nova 2 models: Transform to reasoningConfig structure
- Other models (Anthropic, etc.): Convert to thinking parameter
For Claude 4.6 / 4.7 (adaptive thinking) the tier is carried via
``output_config.effort`` rather than ``thinking.budget_tokens``. We
validate the effort with the same rules ``AnthropicConfig._apply_output_config``
uses (low/medium/high/xhigh/max + per-model gating) and stage the
validated dict on ``optional_params["output_config"]`` so it rides
along to ``additionalModelRequestFields`` on the Anthropic-on-Bedrock
wire path. Without this the silent strip in ``_prepare_request_params``
collapsed every adaptive tier to identical behavior.
Args:
model: The model identifier
reasoning_effort: The reasoning effort value
optional_params: Dictionary of optional parameters to update in-place
Examples:
>>> config = AmazonConverseConfig()
>>> params = {}
>>> config._handle_reasoning_effort_parameter("gpt-oss-model", "high", params)
>>> params
{'reasoning_effort': 'high'}
>>> params = {}
>>> config._handle_reasoning_effort_parameter("amazon.nova-2-lite-v1:0", "high", params)
>>> params
{'reasoningConfig': {'type': 'enabled', 'maxReasoningEffort': 'high'}}
>>> params = {}
>>> config._handle_reasoning_effort_parameter("anthropic.claude-3", "high", params)
>>> params
{'thinking': {'type': 'enabled', 'budget_tokens': 10000}}
- GPT-OSS models: passed through unchanged via additionalModelRequestFields.
- Nova 2 models: transformed to reasoningConfig.
- Anthropic models: mapped to ``thinking`` (and ``output_config.effort`` on
adaptive Claude 4.6 / 4.7).
"""
if "gpt-oss" in model:
# GPT-OSS models: keep reasoning_effort as-is
# It will be passed through to additionalModelRequestFields
optional_params["reasoning_effort"] = reasoning_effort
elif self._is_nova_2_model(model):
# Nova 2 models: transform to reasoningConfig
reasoning_config = self._transform_reasoning_effort_to_reasoning_config(
reasoning_effort
)
optional_params.update(reasoning_config)
else:
# Anthropic and other models: convert to thinking parameter.
# ``_map_reasoning_effort`` raises ``BadRequestError`` (400)
# directly on unmapped efforts (``disabled`` / ``invalid`` /
# ``""`` / ``xhigh``/``max`` on budget-mode Claude 4.5); pass
# ``llm_provider="bedrock_converse"`` so the error carries the
# right provider name.
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=reasoning_effort,
model=model,
@ -478,22 +436,7 @@ class AmazonConverseConfig(BaseConfig):
optional_params.pop("output_config", None)
else:
optional_params["thinking"] = mapped_thinking
# Adaptive-thinking models (Claude 4.6 / 4.7+) take the
# tier via ``output_config.effort``. Mirror the mapping
# used by ``AnthropicConfig.map_openai_params`` and apply
# the same validation rules so unmapped/garbage efforts
# surface as a 400 instead of being silently flattened on
# the wire. Driven by ``supports_adaptive_thinking`` in
# ``model_prices_and_context_window.json`` so a future
# adaptive Claude release lands as a model-map change.
if AnthropicConfig._is_adaptive_thinking_model(model):
# Use ``.get()`` without a fallback so unmapped efforts
# (e.g. ``"disabled"``) surface as a clean 400 here
# rather than leaking the raw garbage string through to
# ``_validate_anthropic_adaptive_effort`` (which does
# catch it, but only because validation happens to run).
# Matches the /v1/messages pattern where validation is
# co-located with the mapping.
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
reasoning_effort
)
@ -514,16 +457,7 @@ class AmazonConverseConfig(BaseConfig):
@staticmethod
def _validate_anthropic_adaptive_effort(model: str, effort: str) -> None:
"""Validate ``output_config.effort`` for adaptive-thinking Claude 4.6/4.7
on Bedrock. Raises ``BadRequestError`` (clean 400) instead of letting
a downstream ``ValueError`` surface as 500.
Per-model gating for ``max``/``xhigh`` is delegated to
``AnthropicConfig._validate_effort_for_model`` so the Bedrock Converse
path and the Anthropic chat / ``/v1/messages`` paths can't drift when
a new gated effort tier is added. ``_supports_effort_level`` on
``AnthropicConfig`` already handles Bedrock-prefixed model ids.
"""
"""Validate ``output_config.effort`` for adaptive-thinking Claude 4.6/4.7."""
valid_efforts = {"high", "medium", "low", "xhigh", "max"}
if effort not in valid_efforts:
raise litellm.exceptions.BadRequestError(
@ -1283,13 +1217,9 @@ class AmazonConverseConfig(BaseConfig):
)
inference_params.pop("json_mode", None) # used for handling json_schema
# Anthropic-only ``output_config`` (snake_case) is the adaptive-
# thinking effort payload (e.g. ``{"effort": "max"}``) for Claude
# 4.6/4.7. On Bedrock Converse it must ride along inside
# ``additionalModelRequestFields`` so the model actually sees the
# tier; stripping it (the prior behavior) silently flattened every
# adaptive tier to identical thinking. Only the Bedrock-native
# ``outputConfig`` (camelCase) goes at the top level.
# Anthropic-only ``output_config`` (snake_case) — re-attached to
# ``additionalModelRequestFields`` for Anthropic models below. The
# Bedrock-native ``outputConfig`` (camelCase) is handled separately.
anthropic_output_config = inference_params.pop("output_config", None)
# Extract requestMetadata before processing other parameters
@ -1342,18 +1272,11 @@ class AmazonConverseConfig(BaseConfig):
additional_request_params
)
# Re-attach the Anthropic ``output_config`` (e.g. adaptive thinking
# ``{"effort": "max"}``) onto additional_request_params for Anthropic
# Bedrock models so the wire request carries the requested tier. Other
# model families (Nova, GPT-OSS, ...) don't accept it; drop it for them.
if anthropic_output_config is not None and isinstance(
anthropic_output_config, dict
):
base_model = BedrockModelInfo.get_base_model(model)
if base_model.startswith("anthropic"):
# When ``drop_params`` is set, strip for models that don't
# accept effort (e.g. proxy routing Claude Code at haiku-3).
# Otherwise forward and let Bedrock surface the model's error.
if (
litellm.drop_params is True
and not AnthropicConfig._model_supports_effort_param(model)
@ -1495,12 +1418,8 @@ class AmazonConverseConfig(BaseConfig):
# Append pre-formatted tools (systemTool etc.) after transformation
bedrock_tools.extend(pre_formatted_tools)
# Auto-attach the effort beta header for non-adaptive Anthropic
# models on Bedrock Converse (i.e. Opus 4.5). Claude 4.6/4.7 accept
# ``output_config.effort`` as a stable, GA feature with no beta
# header; Opus 4.5 still gates it behind ``effort-2025-11-24``. The
# check mirrors ``AnthropicModelInfo.is_effort_used`` (which returns
# False for adaptive models) so we don't double-flag adaptive routes.
# Opus 4.5 gates ``output_config.effort`` behind a beta header;
# Claude 4.6/4.7 accept it without one.
base_model = BedrockModelInfo.get_base_model(model)
if base_model.startswith("anthropic"):
output_config = additional_request_params.get("output_config")

View file

@ -169,11 +169,6 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
anthropic_request.pop("model", None)
anthropic_request.pop("stream", None)
anthropic_request.pop("output_format", None)
# ``output_config`` (e.g. ``{"effort": "max"}``) is the adaptive-thinking
# tier payload for Claude 4.6 / 4.7. Bedrock Invoke accepts it for
# those models — stripping it (the prior behavior) silently flattened
# every adaptive tier on this route. Forward it; if the model rejects
# it the surfaced error is correct, vs. swallowing the user's knob.
if "anthropic_version" not in anthropic_request:
anthropic_request["anthropic_version"] = self.anthropic_version

View file

@ -582,10 +582,6 @@ class AmazonAnthropicClaudeMessagesConfig(
if filtered_betas:
anthropic_messages_request["anthropic_beta"] = filtered_betas
# 6a. When ``drop_params`` is set, strip ``output_config`` for models
# that don't accept it (e.g. proxy fronting Claude Code at haiku-3).
# Without this, every Claude Code request to a pre-4.5 Anthropic model
# routes a forced 400 from Bedrock that the client can't fix.
if (
litellm.drop_params is True
and "output_config" in anthropic_messages_request

View file

@ -334,11 +334,6 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
) # unsupported for claude models - if json_schema -> convert to tool call
if "reasoning_effort" in non_default_params and "claude" in model:
# ``_map_reasoning_effort`` raises ``BadRequestError`` (400)
# directly on unmapped efforts; pass ``llm_provider="databricks"``
# so the surfaced error carries the correct provider name (the
# default is ``"anthropic"``, which would mislead users routing
# via Databricks Foundation Model APIs).
reasoning_effort_value = non_default_params.get("reasoning_effort")
mapped_thinking = AnthropicConfig._map_reasoning_effort(
reasoning_effort=reasoning_effort_value,
@ -350,21 +345,7 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
optional_params.pop("output_config", None)
else:
optional_params["thinking"] = mapped_thinking
# For Claude 4.6+ adaptive models, ``_map_reasoning_effort``
# returns ``type=adaptive`` for ANY non-None / non-"none"
# string without validating the value, so reject unmapped
# efforts here and set ``output_config.effort`` (matching the
# Anthropic native / Bedrock Converse / Bedrock Invoke /
# /v1/messages paths). Driven by ``supports_adaptive_thinking``
# in the model map so future adaptive Claudes land via a
# model-map update rather than a code release — the
# Anthropic-native and Bedrock routes already use the same
# helper, so all three paths stay in lock-step.
if AnthropicConfig._is_adaptive_thinking_model(model):
# ``reasoning_effort_value`` comes from ``non_default_params``
# so its static type is ``Any | None``. Narrow to ``str`` for
# the mapping lookup; non-strings fall through to the
# ``BadRequestError`` below with a clean validation message.
mapped_effort: Optional[str] = None
if isinstance(reasoning_effort_value, str):
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(

View file

@ -159,11 +159,6 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert
"model", None
) # do not pass model in request body to vertex ai
# Vertex AI Claude accepts ``output_config.format`` (structured outputs),
# ``output_format``, and ``output_config.effort`` (adaptive-thinking
# tier on Claude 4.6 / 4.7, verified end-to-end). The shared sanitize
# helper now no-ops for ``effort`` and remains the single hook for any
# future Vertex-only key drift.
sanitize_vertex_anthropic_output_params(anthropic_messages_request)
return anthropic_messages_request

View file

@ -11,11 +11,8 @@ keeps the parent module's import surface narrow.
"""
# Keys inside ``output_config`` that Vertex AI Claude does not accept.
# Vertex now accepts ``output_config.effort`` for the adaptive-thinking
# Claude 4.6 / 4.7 models on direct ``:rawPredict`` (verified end-to-end
# against ``us-east5`` for ``opus-4-6`` and ``global`` for ``opus-4-7``).
# Keep this set narrow and only add a key here once a 400 "Extra inputs are
# not permitted" is reproducible against the live Vertex endpoint.
# Add an entry only when a 400 "Extra inputs are not permitted" is
# reproducible against the live Vertex endpoint.
VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS: frozenset = frozenset()

View file

@ -106,10 +106,6 @@ class VertexAIAnthropicConfig(AnthropicConfig):
data.pop("model", None) # vertex anthropic doesn't accept 'model' parameter
# Sanitize ``output_config`` / ``output_format`` for Vertex parity.
# Vertex now accepts ``output_config.effort`` for adaptive-thinking Claude
# 4.6 / 4.7 models, so the helper is a no-op for ``effort``; it remains
# the single hook for future Vertex-only sanitization.
sanitize_vertex_anthropic_output_params(data)
tools = optional_params.get("tools")

View file

@ -224,12 +224,6 @@ class XAIChatConfig(OpenAIGPTConfig):
except Exception as e:
verbose_logger.debug(f"Error extracting X.AI web search usage: {e}")
# X.AI excludes reasoning_tokens from completion_tokens, breaking the
# OpenAI invariant total_tokens == prompt_tokens + completion_tokens.
# OpenAI o1/o3 fold reasoning into completion_tokens; align X.AI to
# match so downstream consumers (including litellm's own usage tests)
# see a self-consistent Usage object. Cost calc already accounts for
# reasoning tokens separately in xai.cost_calculator.
self._fold_reasoning_tokens_into_completion(response)
return response
@ -237,15 +231,10 @@ class XAIChatConfig(OpenAIGPTConfig):
def _fold_reasoning_tokens_into_completion(model_response: ModelResponse) -> None:
"""Reconcile xAI Usage to the OpenAI invariant.
xAI returns ``completion_tokens`` covering only visible output and
accounts ``reasoning_tokens`` separately, while still rolling them
into ``total_tokens``. OpenAI's published contract (o1/o3) includes
reasoning in ``completion_tokens``. Tests that assert
``total_tokens == prompt_tokens + completion_tokens`` (e.g.
``_usage_format_tests``) fail on the raw xAI shape.
This helper is idempotent: if ``completion_tokens`` already covers
the gap, no change is made.
xAI accounts ``reasoning_tokens`` separately from
``completion_tokens`` while still summing them into ``total_tokens``.
OpenAI's contract (o1/o3) folds reasoning into ``completion_tokens``,
so fold here to keep ``total = prompt + completion``. Idempotent.
"""
usage = getattr(model_response, "usage", None)
if usage is None:
@ -262,13 +251,10 @@ class XAIChatConfig(OpenAIGPTConfig):
completion_tokens = int(getattr(usage, "completion_tokens", 0) or 0)
total_tokens = int(getattr(usage, "total_tokens", 0) or 0)
# Already consistent — nothing to do.
if total_tokens == prompt_tokens + completion_tokens:
return
# Only fold when xAI's accounting (total = prompt + completion +
# reasoning) explains the gap. This guards against double-counting
# if xAI ever changes their semantics.
# Guard against double-counting if xAI changes accounting.
if total_tokens != prompt_tokens + completion_tokens + reasoning_tokens:
return

View file

@ -25,13 +25,10 @@ def cost_per_token(model: str, usage: Usage) -> Tuple[float, float]:
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
"""
# XAI-specific completion cost calculation
# For XAI models, completion is billed as (visible completion tokens + reasoning tokens).
# The transformation layer normalises Usage to the OpenAI invariant
# (completion_tokens includes reasoning_tokens), so detect that and avoid
# double-counting. Fall back to the raw xAI shape (visible-only completion +
# reasoning kept in completion_tokens_details) for callers that bypass the
# transformation, e.g. proxy logs replayed straight into cost calc.
# XAI-specific completion cost: completion is billed as visible + reasoning
# tokens. Detect when the transformation layer already folded them so we
# don't double-count; fall back to raw xAI shape for callers that bypass
# the transformation (e.g. proxy logs replayed into cost calc).
prompt_tokens = int(getattr(usage, "prompt_tokens", 0) or 0)
completion_tokens = int(getattr(usage, "completion_tokens", 0) or 0)
total_tokens = int(getattr(usage, "total_tokens", 0) or 0)

View file

@ -393,12 +393,6 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False):
AnthropicOutputConfig
] # Configuration for Claude's output behavior
cache_control: Optional[Dict[str, Any]] # Automatic prompt caching
# OpenAI-style ``reasoning_effort`` is accepted on /v1/messages so callers
# can drive adaptive/extended thinking with a single tier-name knob (the
# same vocabulary as the chat completion path). The transformation layer
# maps it to native Anthropic ``thinking`` + ``output_config`` and pops
# this key before the request is forwarded — no provider receives
# ``reasoning_effort`` on the wire.
reasoning_effort: Optional[str]

View file

@ -1041,12 +1041,4 @@ class BedrockInvokeAnthropicMessagesRequest(TypedDict, total=False):
# `metadata` is part of the common Anthropic Messages API shape.
thinking: dict
metadata: dict
# ``output_config`` is the adaptive-thinking effort payload for
# Claude 4.6 / 4.7 (e.g. ``{"effort": "max"}``). Bedrock Invoke
# accepts it for these models when ``thinking={"type": "adaptive"}``.
# Without this field in the allowlist, the runtime filter in
# ``AmazonAnthropicClaudeMessagesConfig.transform_anthropic_messages_request``
# silently drops it and every adaptive tier collapses to identical
# behavior on /v1/messages.
output_config: dict

View file

@ -1,37 +1,10 @@
"""Force ``litellm.model_cost`` to load from the PR-local JSON for these tests.
By default ``litellm.model_cost`` is fetched from the main branch on GitHub,
which lags behind PR-branch flag additions. This fixture loads the local
file so per-model flag tests pass in CI as well as locally.
See https://github.com/BerriAI/litellm/issues/27122.
"""
Local-conftest for ``tests/test_litellm/llms/anthropic/chat``.
Why this exists
---------------
``litellm.model_cost`` is loaded once at ``litellm.__init__`` time. By default
it fetches ``model_prices_and_context_window.json`` from the **main branch on
GitHub** (``litellm.model_cost_map_url``) — *not* from the PR-branch JSON in
the working tree. That works fine in production (operators get new models
without redeploying litellm) but is the wrong default for tests, which need
to validate the code in front of them against the data in front of them.
Several anthropic chat transformation tests (``test_supports_effort_level_*``,
``test_anthropic_model_supports_effort_param_*``) assert per-model flags
like ``supports_max_reasoning_effort`` / ``supports_xhigh_reasoning_effort``
that may exist in the PR-local JSON but not yet in main's JSON. Without this
fixture those tests pass locally (where AGENTS.md tells contributors to set
``LITELLM_LOCAL_MODEL_COST_MAP=True``) but fail in CI (which doesn't set the
env var) — the chicken-and-egg PR adds flag → CI fetches main without flag
→ test fails → flag never lands on main.
The fixture force-loads ``litellm.model_cost`` from the local JSON for every
test in this directory. ``tests/test_litellm/conftest.py`` already snapshots
and restores ``litellm.model_cost`` per-function, so this mutation is safe
and contained.
This is a scoped workaround. The proper fix is to set
``LITELLM_LOCAL_MODEL_COST_MAP=True`` globally in the test workflow once the
~10 inline-set test files have been audited and the few tests that exercise
the remote-fetch / integrity-validation path have been given carve-outs.
Tracked at https://github.com/BerriAI/litellm/issues/27122.
"""
import os
import pytest
@ -41,10 +14,6 @@ from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
@pytest.fixture(autouse=True)
def _use_pr_local_model_cost_map(monkeypatch):
"""Force ``litellm.model_cost`` to the PR-branch JSON for the duration of
each test. ``monkeypatch`` reverts the env var after the test; the parent
conftest restores ``litellm.model_cost`` from its snapshot.
"""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(
litellm,

View file

@ -1632,7 +1632,6 @@ def test_effort_validation():
)
assert result["output_config"]["effort"] == effort
# Invalid value should raise BadRequestError (clean 400, not a 500).
with pytest.raises(
litellm.exceptions.BadRequestError, match="Invalid effort value"
):
@ -1685,10 +1684,7 @@ def test_effort_validation_with_opus_46():
def test_max_effort_rejected_for_opus_45():
"""Test that effort='max' is rejected when using Claude Opus 4.5.
Surfaces as a clean 400 BadRequestError, not a 500 ValueError.
"""
"""Test that effort='max' is rejected when using Claude Opus 4.5."""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Test"}]
@ -1749,12 +1745,7 @@ def test_effort_with_other_features():
def test_anthropic_drop_params_strips_output_config_for_pre_4_5_models():
"""
Proxies fronting Claude Code at pre-4.5 Anthropic models receive
``output_config`` injected by the client; without ``drop_params`` Bedrock /
Anthropic 400s. With ``drop_params=True`` we strip it (logged) so the
request can succeed.
"""
"""``drop_params=True`` strips unsupported ``output_config`` for pre-4.5 models."""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
@ -1796,10 +1787,7 @@ def test_anthropic_drop_params_keeps_output_config_for_supporting_models():
def test_anthropic_drop_params_false_forwards_to_unsupported_model():
"""
Default behavior: forward ``output_config`` and let the provider 400.
This is the contract for users who want strict, debuggable failures.
"""
"""Default ``drop_params=False`` forwards ``output_config`` and lets the provider 400."""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
@ -2059,10 +2047,7 @@ def test_get_config_without_model_uses_fallback():
def test_get_config_does_not_leak_module_constants():
"""``BaseConfig.get_config`` returns class attributes; the
reasoning-effort mapping must not be one of them or it ends up
serialised onto the wire as an extra request key.
"""
"""``get_config`` must not leak the reasoning-effort lookup table onto the wire."""
cfg = AnthropicConfig.get_config(model="claude-opus-4-7")
for forbidden in (
"REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT",
@ -2078,10 +2063,6 @@ def test_get_config_does_not_leak_module_constants():
("claude-opus-4-7", "xhigh", True),
("claude-opus-4-6", "max", True),
("claude-opus-4-6", "xhigh", False),
# ``max`` is documented as supported on Sonnet 4.6 (Claude 4.6 family).
# The model-map JSON now carries ``supports_max_reasoning_effort: true``
# for every Sonnet 4.6 entry; ``_supports_effort_level`` should report
# ``True`` on every route prefix (anthropic, bedrock, vertex, azure).
("claude-sonnet-4-6", "max", True),
("claude-sonnet-4-6", "xhigh", False),
("bedrock/invoke/us.anthropic.claude-opus-4-7", "max", True),
@ -2094,27 +2075,21 @@ def test_get_config_does_not_leak_module_constants():
],
)
def test_supports_effort_level_handles_provider_prefixes(model, level, expected):
"""``_supports_effort_level`` must handle bedrock/ vertex_ai/ azure_ai/
prefixed model ids so per-model gating works on every route, not just
the bare-Anthropic chat completion path.
"""
"""``_supports_effort_level`` resolves bedrock/vertex/azure-prefixed model ids."""
assert AnthropicConfig._supports_effort_level(model, level) is expected
@pytest.mark.parametrize(
"model,effort,expect_error",
[
# ``max`` accepted on 4.6 / 4.7 (family fallback) and rejected on 4.5.
("claude-opus-4-6", "max", False),
("claude-sonnet-4-6", "max", False),
("claude-opus-4-7", "max", False),
("claude-opus-4-5-20251101", "max", True),
("claude-sonnet-4-5", "max", True),
# ``xhigh`` data-driven; only 4.7 carries the flag in the model map.
("claude-opus-4-7", "xhigh", False),
("claude-opus-4-6", "xhigh", True),
("claude-sonnet-4-6", "xhigh", True),
# Lower efforts and ``None`` always pass the gate.
("claude-opus-4-5-20251101", "high", False),
("claude-haiku-4-5", "low", False),
("claude-opus-4-5-20251101", None, False),
@ -2123,13 +2098,6 @@ def test_supports_effort_level_handles_provider_prefixes(model, level, expected)
def test_validate_effort_for_model_centralises_per_model_gating(
model, effort, expect_error
):
"""``_validate_effort_for_model`` is the single source of truth for the
per-model ``max`` / ``xhigh`` gating that ``_apply_output_config`` (chat
completion path) and ``AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic``
(/v1/messages pass-through) both rely on. Both call sites raise their
own provider-appropriate exception, but the gating decision must come
from one place to prevent drift when a new model tier lands.
"""
err = AnthropicConfig._validate_effort_for_model(model, effort)
if expect_error:
assert err is not None
@ -2342,14 +2310,12 @@ def test_reasoning_effort_maps_to_budget_thinking_for_non_opus_4_6():
"""
config = AnthropicConfig()
# Test with Claude Sonnet 4.5 (non-Opus 4.6 model).
# ``minimal`` floors at the Anthropic provider minimum (1024) because
# Anthropic / Azure / Vertex / Bedrock Invoke 400 below that.
# ``minimal`` floors at ANTHROPIC_MIN_THINKING_BUDGET_TOKENS (1024).
test_cases = [
("low", 1024), # DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET
("medium", 2048), # DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET
("high", 4096), # DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
("minimal", 1024), # ANTHROPIC_MIN_THINKING_BUDGET_TOKENS (provider floor)
("low", 1024),
("medium", 2048),
("high", 4096),
("minimal", 1024),
]
for effort, expected_budget in test_cases:
@ -2447,16 +2413,7 @@ def test_reasoning_effort_does_not_set_output_config_for_older_models():
],
)
def test_max_effort_accepted_for_sonnet_46_variants(model):
"""``effort='max'`` is documented as supported on Claude 4.6 (Opus + Sonnet)
and Claude 4.7 (https://platform.claude.com/docs/en/build-with-claude/effort).
Earlier versions of this test asserted a 400 for Sonnet 4.6, mirroring an
Opus-only allow-list in ``_apply_output_config``. That gate has since been
widened to ``_is_claude_4_6_model`` (Opus + Sonnet) and the
``supports_max_reasoning_effort`` JSON flag, matching Anthropic's published
matrix. Verify the param actually flows through every Sonnet 4.6 id
variant our routing layer might see.
"""
"""``effort='max'`` is supported on Claude 4.6 (Opus + Sonnet) and 4.7."""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Test"}]
@ -2552,9 +2509,7 @@ def test_reasoning_effort_none_omits_thinking_and_output_config(model):
["disabled", "invalid", ""],
)
def test_reasoning_effort_garbage_raises_bad_request(effort):
"""Unmapped / garbage / empty-string reasoning_effort surfaces as a clean
400 ``BadRequestError`` instead of letting ``ValueError`` propagate as 500.
"""
"""Unmapped reasoning_effort raises BadRequestError (clean 400, not a 500)."""
config = AnthropicConfig()
with pytest.raises(litellm.exceptions.BadRequestError):
@ -2573,17 +2528,7 @@ def test_reasoning_effort_garbage_raises_bad_request(effort):
def test_reasoning_effort_xhigh_max_maps_to_budget_on_budget_model(
effort, expected_budget
):
"""``xhigh`` / ``max`` extend the legacy ``thinking.budget_tokens``
progression (low=1024 / medium=2048 / high=4096 → xhigh=8192 / max=16384)
on budget-mode Claude models (haiku / 4.5 series).
Adopted from #27051. Keeps the OpenAI-format ``reasoning_effort`` knob
usable across the full Claude lineup — adaptive models (4.6/4.7) route
these tiers via ``output_config.effort``; budget-mode models use the
extended budget. Anthropic's "max only on Mythos / Opus 4.7 / Opus 4.6 /
Sonnet 4.6" gating applies to the *adaptive enum*, not to the legacy
``budget_tokens`` knob, which accepts any integer up to ``max_tokens``.
"""
"""``xhigh`` / ``max`` extend the budget_tokens progression on budget-mode models."""
config = AnthropicConfig()
result = config.map_openai_params(
@ -2595,16 +2540,11 @@ def test_reasoning_effort_xhigh_max_maps_to_budget_on_budget_model(
assert result["thinking"]["type"] == "enabled"
assert result["thinking"]["budget_tokens"] == expected_budget
# Budget-mode models must NOT carry an ``output_config`` payload — that
# path is exclusively for adaptive (4.6+) models.
assert "output_config" not in result
def test_output_config_effort_empty_string_raises_bad_request():
"""``output_config={"effort": ""}`` must be rejected with a 400 — the
legacy ``if effort and ...`` short-circuit silently let it pass
through (verified end-to-end on the QA sweep for PR #27039).
"""
"""``output_config={"effort": ""}`` is rejected with a 400."""
config = AnthropicConfig()
with pytest.raises(litellm.exceptions.BadRequestError, match="Invalid effort"):
@ -2618,10 +2558,7 @@ def test_output_config_effort_empty_string_raises_bad_request():
def test_reasoning_effort_minimal_floors_at_anthropic_provider_minimum():
"""Anthropic Messages API rejects ``budget_tokens < 1024``. ``minimal``
must floor at the provider minimum so it's a usable tier on direct
Anthropic / Azure AI Anthropic / Vertex AI Anthropic / Bedrock Invoke.
"""
"""``minimal`` floors at the Anthropic provider minimum (1024)."""
config = AnthropicConfig()
result = config.map_openai_params(

View file

@ -1,18 +1,4 @@
"""
Tests for OpenAI-style ``reasoning_effort`` translation on the Anthropic
/v1/messages route.
The /v1/messages spec doesn't include ``reasoning_effort`` — without
translation it gets silently dropped at the filter step, leaving every
adaptive tier collapsed to the same behavior on Bedrock Invoke /v1/messages
(and on Anthropic / Azure AI / Vertex AI when callers pass it on the
messages route).
These tests pin the translation and validation behavior at the shared
``AnthropicMessagesConfig`` level so all four /v1/messages routes
(direct Anthropic, Azure AI, Vertex AI, Bedrock Invoke) inherit the
same mapping.
"""
"""Tests for ``reasoning_effort`` translation on the Anthropic /v1/messages route."""
import pytest
@ -36,12 +22,6 @@ from litellm.llms.anthropic.experimental_pass_through.messages.transformation im
def test_reasoning_effort_maps_to_output_config_for_adaptive_model(
reasoning_effort, expected_effort
):
"""
For Claude 4.6 / 4.7, ``reasoning_effort`` is mapped to
``thinking={"type": "adaptive"}`` plus ``output_config.effort=<tier>``,
using the same mapping table as the chat completion path so the two
routes can't drift.
"""
config = AnthropicMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": reasoning_effort}
@ -59,7 +39,6 @@ def test_reasoning_effort_maps_to_output_config_for_adaptive_model(
def test_reasoning_effort_none_clears_thinking_and_output_config():
"""``reasoning_effort='none'`` opts out of extended thinking entirely."""
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
@ -82,11 +61,6 @@ def test_reasoning_effort_none_clears_thinking_and_output_config():
def test_reasoning_effort_on_non_adaptive_model_uses_thinking_budget():
"""
Non-adaptive models (Opus 4.5 / earlier) take ``thinking.budget_tokens``
rather than ``output_config.effort``. The translation falls back to
the budget mapping in that case.
"""
config = AnthropicMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": "high"}
@ -109,11 +83,6 @@ def test_reasoning_effort_on_non_adaptive_model_uses_thinking_budget():
@pytest.mark.parametrize("bad_effort", ["invalid", "disabled", ""])
def test_invalid_reasoning_effort_raises_400(bad_effort):
"""
Garbage ``reasoning_effort`` values surface as a clean 400 instead of
silently passing through to the provider as an unknown
``output_config.effort`` (which would 500).
"""
config = AnthropicMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort}
@ -132,18 +101,12 @@ def test_invalid_reasoning_effort_raises_400(bad_effort):
@pytest.mark.parametrize(
"model,bad_effort",
[
# ``xhigh`` is Opus-4.7-only on the public Anthropic effort matrix,
# so Opus 4.6 / Sonnet 4.6 must still 400 on it.
("claude-opus-4-6", "xhigh"),
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh"),
("claude-sonnet-4-6", "xhigh"),
],
)
def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort):
"""``xhigh`` and ``max`` are gated per-model. The /v1/messages route must
surface a clean 400 client-side instead of forwarding the unsupported
tier and letting the provider 500/400 it.
"""
config = AnthropicMessagesConfig()
optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort}
@ -163,9 +126,6 @@ def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort
@pytest.mark.parametrize(
"model",
[
# ``max`` is documented as supported on Claude 4.6 (Opus + Sonnet)
# and Claude 4.7. Verify the /v1/messages route accepts it for
# Sonnet 4.6 variants instead of 400-ing client-side.
"claude-sonnet-4-6",
"bedrock/invoke/us.anthropic.claude-sonnet-4-6",
],
@ -187,11 +147,6 @@ def test_reasoning_effort_max_accepted_on_sonnet_46_messages(model):
def test_explicit_output_config_wins_over_reasoning_effort():
"""
Explicit native ``output_config.effort`` is never overridden by the
OpenAI alias. Same precedence as
``_translate_legacy_thinking_for_adaptive_model``.
"""
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
@ -212,7 +167,6 @@ def test_explicit_output_config_wins_over_reasoning_effort():
def test_explicit_thinking_wins_over_reasoning_effort():
"""Explicit native ``thinking`` is never overridden by the alias."""
config = AnthropicMessagesConfig()
optional_params = {
"max_tokens": 1024,
@ -233,8 +187,6 @@ def test_explicit_thinking_wins_over_reasoning_effort():
def test_reasoning_effort_in_supported_params():
"""``reasoning_effort`` is advertised as a supported messages param so
callers and validation paths can introspect the schema."""
config = AnthropicMessagesConfig()
assert "reasoning_effort" in config.get_supported_anthropic_messages_params(
"claude-opus-4-7"

View file

@ -292,19 +292,10 @@ class TestAzureAnthropicConfig:
assert "compact-2026-01-12" in headers["anthropic-beta"]
def test_output_config_promoted_from_extra_body(self):
"""Anthropic-native ``output_config`` routed via ``extra_body`` (by the
openai-compatible kwarg stuffer in ``litellm/utils.py``) must be
promoted to top-level ``optional_params`` before delegating to
``AnthropicConfig.transform_request``. Otherwise the value is
silently dropped by the ``extra_body`` pop and validation never
runs.
"""
"""Anthropic ``output_config`` routed via ``extra_body`` reaches the request body."""
config = AzureAnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
# Simulate what ``add_provider_specific_params_to_optional_params``
# produces for ``litellm.completion(model="azure_ai/claude-...",
# output_config={"effort": "low"})``.
optional_params = {
"max_tokens": 100,
"extra_body": {"output_config": {"effort": "low"}},
@ -313,26 +304,18 @@ class TestAzureAnthropicConfig:
headers = {"api-key": "test-key", "anthropic-version": "2023-06-01"}
result = config.transform_request(
model="claude-opus-4-6", # supports output_config.effort
model="claude-opus-4-6",
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
headers=headers,
)
# output_config must reach the request body (not be silently dropped)
assert "output_config" in result
assert result["output_config"] == {"effort": "low"}
# extra_body should be stripped on the way out
assert "extra_body" not in result
def test_invalid_output_config_effort_raises_via_extra_body(self):
"""``effort="invalid"`` arriving via ``extra_body`` must still raise
``BadRequestError`` (matching the native ``anthropic`` provider).
Previously this was silently dropped on Azure, so unsupported
efforts reached the model with default behavior instead of
returning a clean 400.
"""
"""Invalid ``effort`` via ``extra_body`` raises BadRequestError."""
import litellm
config = AzureAnthropicConfig()
@ -356,9 +339,7 @@ class TestAzureAnthropicConfig:
assert "Invalid effort value" in str(exc_info.value)
def test_unsupported_effort_xhigh_raises_via_extra_body(self):
"""``effort="xhigh"`` on a model that does not support it (e.g.
Sonnet 4.6) must raise ``BadRequestError`` even when arriving via
``extra_body`` on Azure."""
"""Unsupported ``effort='xhigh'`` via ``extra_body`` raises BadRequestError."""
import litellm
config = AzureAnthropicConfig()
@ -382,17 +363,14 @@ class TestAzureAnthropicConfig:
assert "xhigh" in str(exc_info.value)
def test_extra_body_promotion_does_not_clobber_top_level(self):
"""If a key exists at both top-level ``optional_params`` and
inside ``extra_body``, the top-level value wins (``setdefault``
semantics). This protects against a caller who explicitly sets a
param at top-level and accidentally also has it in extra_body."""
"""Top-level ``optional_params`` wins over duplicates in ``extra_body``."""
config = AzureAnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
optional_params = {
"max_tokens": 100,
"output_config": {"effort": "low"}, # top-level wins
"extra_body": {"output_config": {"effort": "high"}}, # ignored
"output_config": {"effort": "low"},
"extra_body": {"output_config": {"effort": "high"}},
}
litellm_params = {"api_key": "test-key"}
headers = {"api-key": "test-key", "anthropic-version": "2023-06-01"}

View file

@ -407,18 +407,7 @@ def test_opus_4_5_model_detection():
def test_output_config_forwarded_for_bedrock_chat_invoke_request():
"""
Bedrock Invoke (chat/completions route) must forward
``output_config`` for Anthropic adaptive-thinking models. The earlier
behavior stripped it unconditionally, which silently flattened every
adaptive tier (``low``/``medium``/``high``/``xhigh``/``max``) to identical
behavior on the wire.
The wire QA at https://github.com/BerriAI/litellm/pull/27039 showed
``thinking.type: adaptive`` was forwarded but ``output_config.effort``
was always missing, even though direct curls to Anthropic's Bedrock
Invoke endpoint accept it.
"""
"""Bedrock Invoke chat path forwards ``output_config`` for adaptive Claude models."""
config = AmazonAnthropicClaudeConfig()
messages = [{"role": "user", "content": "test"}]
@ -437,7 +426,6 @@ def test_output_config_forwarded_for_bedrock_chat_invoke_request():
)
assert result.get("output_config") == {"effort": "high"}
# Verify normal params survive
assert result["max_tokens"] == 100

View file

@ -326,10 +326,7 @@ def test_reasoning_effort_none_omits_thinking_for_anthropic_converse(model):
def test_reasoning_effort_sets_output_config_for_adaptive_models_converse(
model, effort, expected_effort
):
"""Adaptive-thinking Claude 4.6 / 4.7 on Bedrock Converse must carry the
requested tier via ``output_config.effort``. The prior strip silently
flattened every adaptive tier on the wire (verified in the PR #27039
QA sweep)."""
"""Adaptive Claude 4.6 / 4.7 on Bedrock Converse routes the tier via ``output_config.effort``."""
config = AmazonConverseConfig()
optional_params = config.map_openai_params(
@ -352,10 +349,7 @@ def test_reasoning_effort_sets_output_config_for_adaptive_models_converse(
],
)
def test_output_config_effort_forwarded_into_additional_request_fields(model):
"""``output_config`` must ride along inside ``additionalModelRequestFields``
so the Anthropic-on-Bedrock wire request actually carries the effort
tier. The prior ``inference_params.pop("output_config")`` dropped it
on the floor for every adaptive tier."""
"""``output_config`` rides along inside ``additionalModelRequestFields``."""
config = AmazonConverseConfig()
messages = [{"role": "user", "content": "hi"}]
@ -380,10 +374,7 @@ def test_output_config_effort_forwarded_into_additional_request_fields(model):
["disabled", "invalid", ""],
)
def test_reasoning_effort_garbage_raises_bad_request_converse(effort):
"""Garbage / empty-string reasoning_effort on Bedrock Converse Anthropic
must surface as a clean 400 ``BadRequestError`` instead of 500. The
earlier ``ValueError`` from ``_map_reasoning_effort`` propagated up as
a generic 500 in the proxy and ate the request."""
"""Unmapped reasoning_effort on Bedrock Converse Anthropic raises BadRequestError."""
config = AmazonConverseConfig()
with pytest.raises(litellm.exceptions.BadRequestError):
@ -405,13 +396,7 @@ def test_reasoning_effort_garbage_raises_bad_request_converse(effort):
],
)
def test_output_config_effort_max_passes_through_on_sonnet_46_variants(model):
"""``effort='max'`` is supported on Claude 4.6 (Opus + Sonnet) per
https://platform.claude.com/docs/en/build-with-claude/effort. The earlier
Opus-only allow-list in ``_validate_anthropic_adaptive_effort`` has been
widened to ``_is_claude_4_6_model`` (Opus + Sonnet) plus the
``supports_max_reasoning_effort`` JSON flag. Verify the param actually
flows through to ``additionalModelRequestFields.output_config.effort``
for every Bedrock Converse Sonnet 4.6 id variant."""
"""``effort='max'`` flows through for every Bedrock Converse Sonnet 4.6 id."""
config = AmazonConverseConfig()
messages = [{"role": "user", "content": "hi"}]
@ -3450,9 +3435,7 @@ def test_transform_request_strips_anthropic_output_config():
def test_converse_drop_params_strips_output_config_for_pre_4_5_anthropic():
"""``drop_params=True`` strips ``output_config`` for pre-4.5 Anthropic
models on Bedrock Converse so a proxy fronting Claude Code at haiku doesn't
force a 400 on every request."""
"""``drop_params=True`` strips unsupported ``output_config`` on Bedrock Converse."""
config = AmazonConverseConfig()
messages = [{"role": "user", "content": "hi"}]
@ -3477,7 +3460,7 @@ def test_converse_drop_params_strips_output_config_for_pre_4_5_anthropic():
def test_converse_drop_params_keeps_output_config_for_supporting_anthropic():
"""``drop_params=True`` must not strip on supporting models."""
"""``drop_params=True`` does not strip on models that support ``output_config``."""
config = AmazonConverseConfig()
messages = [{"role": "user", "content": "hi"}]

View file

@ -593,16 +593,7 @@ def test_remove_scope_from_cache_control():
def test_bedrock_messages_forwards_output_config():
"""
``output_config`` is the adaptive-thinking effort payload for Claude
4.6 / 4.7 (e.g. ``{"effort": "max"}``). Bedrock Invoke accepts it for
those models — the prior behavior of stripping it silently flattened
every adaptive tier on /v1/messages so ``low`` / ``medium`` / ``high`` /
``xhigh`` / ``max`` all collapsed to identical thinking with no tier
differentiation.
Regression coverage for the QA bug listed on PR #27039.
"""
"""Bedrock Invoke /v1/messages forwards ``output_config`` for adaptive Claude models."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
@ -628,12 +619,7 @@ def test_bedrock_messages_forwards_output_config():
def test_bedrock_messages_forwards_output_config_with_output_format():
"""
When both output_config and output_format are present, output_format is
converted to inline schema (Bedrock Invoke doesn't accept output_format
natively), and output_config is forwarded for Claude 4.6/4.7 adaptive
thinking.
"""
"""``output_config`` is forwarded; ``output_format`` is converted to inline schema."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
@ -663,19 +649,7 @@ def test_bedrock_messages_forwards_output_config_with_output_format():
def test_bedrock_messages_forwards_output_config_for_non_adaptive_model():
"""
``output_config`` is forwarded for non-adaptive models too (e.g. haiku).
Bedrock will reject the unsupported key for those models — surfacing the
provider error is the correct behavior, since silently swallowing the
knob would hide caller bugs.
Restores coverage previously asserted by
``test_bedrock_messages_strips_output_config`` (renamed to the
``forwards`` variant on Opus 4.7); the strip path no longer exists,
but the non-adaptive pass-through path needs its own explicit test
so a future regression that silently re-adds the strip can't sneak
through.
"""
"""``output_config`` is forwarded for non-adaptive models so the provider's error surfaces."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
@ -698,14 +672,7 @@ def test_bedrock_messages_forwards_output_config_for_non_adaptive_model():
def test_bedrock_messages_drop_params_strips_output_config_for_pre_4_5():
"""
``drop_params=True`` is the operator opt-in for "silently fix up"
behavior. When a proxy fronts Claude Code at a pre-4.5 Anthropic model
(haiku-3, sonnet-3.5, ...) on the /v1/messages route, the client always
sends ``output_config.effort`` and the model rejects it. Stripping under
``drop_params`` lets those requests succeed; otherwise we forward and
surface the model's 400 as designed.
"""
"""``drop_params=True`` strips ``output_config`` for pre-4.5 Anthropic on /v1/messages."""
import litellm
from litellm.types.router import GenericLiteLLMParams
@ -733,8 +700,7 @@ def test_bedrock_messages_drop_params_strips_output_config_for_pre_4_5():
def test_bedrock_messages_drop_params_keeps_output_config_for_4_7():
"""``drop_params=True`` must not strip on supporting models — opus-4-7
accepts effort, so the client's tier knob has to land on the wire."""
"""``drop_params=True`` does not strip on opus-4-7 (supports effort)."""
import litellm
from litellm.types.router import GenericLiteLLMParams
@ -775,18 +741,7 @@ def test_bedrock_messages_drop_params_keeps_output_config_for_4_7():
def test_bedrock_messages_maps_reasoning_effort_for_adaptive_model(
reasoning_effort, expected_effort
):
"""
OpenAI-style ``reasoning_effort`` is mapped to native Anthropic
``thinking`` + ``output_config.effort`` on the /v1/messages route so
callers can drive adaptive thinking with the same tier vocabulary as
the chat completion path. ``reasoning_effort`` itself is popped — the
/v1/messages spec doesn't define it and Bedrock rejects unknown
top-level fields.
Closes the QA-sweep gap on PR #27074 where Bedrock Invoke /v1/messages
silently dropped ``reasoning_effort`` and every effort tier collapsed
to the same behavior.
"""
"""``reasoning_effort`` maps to ``thinking`` + ``output_config.effort`` on /v1/messages."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
@ -810,16 +765,7 @@ def test_bedrock_messages_maps_reasoning_effort_for_adaptive_model(
def test_bedrock_messages_reasoning_effort_on_non_adaptive_uses_thinking_budget():
"""
For non-adaptive thinking models (e.g. Opus 4.5), ``reasoning_effort``
is mapped to ``thinking.type=enabled`` with a budget_tokens value
instead of ``output_config.effort``. ``output_config`` is not set on
these models because they don't accept it.
Mirrors ``AnthropicConfig._map_reasoning_effort`` behavior for the
non-adaptive branch on Opus 4.5 / earlier Claude 4 models, applied to
the /v1/messages route.
"""
"""Non-adaptive models map ``reasoning_effort`` to ``thinking.budget_tokens``."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
@ -847,11 +793,7 @@ def test_bedrock_messages_reasoning_effort_on_non_adaptive_uses_thinking_budget(
def test_bedrock_messages_reasoning_effort_none_clears_thinking():
"""
``reasoning_effort='none'`` opts out — both ``thinking`` and
``output_config`` are cleared so the request goes out without
extended thinking. Mirrors the chat completion path's behavior.
"""
"""``reasoning_effort='none'`` clears both ``thinking`` and ``output_config``."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
@ -877,11 +819,7 @@ def test_bedrock_messages_reasoning_effort_none_clears_thinking():
def test_bedrock_messages_invalid_reasoning_effort_raises_400():
"""
Garbage ``reasoning_effort`` values (``invalid`` / ``disabled`` / ``""``)
surface as a clean 400 ``AnthropicError`` instead of silently passing
an invalid string through to Bedrock as ``output_config.effort``.
"""
"""Garbage ``reasoning_effort`` raises AnthropicError (400)."""
from litellm.llms.anthropic.common_utils import AnthropicError
from litellm.types.router import GenericLiteLLMParams
@ -903,12 +841,7 @@ def test_bedrock_messages_invalid_reasoning_effort_raises_400():
def test_bedrock_messages_explicit_output_config_wins_over_reasoning_effort():
"""
Caller-supplied native ``output_config.effort`` wins over the OpenAI
``reasoning_effort`` knob. Same precedence as
``_translate_legacy_thinking_for_adaptive_model``: explicit native
Anthropic params are never overridden by the alias.
"""
"""Explicit ``output_config.effort`` wins over the ``reasoning_effort`` alias."""
from litellm.types.router import GenericLiteLLMParams
cfg = AmazonAnthropicClaudeMessagesConfig()
@ -1006,11 +939,6 @@ def test_bedrock_messages_allowlist_filters_anthropic_only_fields():
):
assert bad not in result, f"{bad} should be stripped by the allowlist"
# ``output_config`` rides along — Bedrock Invoke accepts it for Claude
# 4.6/4.7 adaptive thinking and stripping it silently flattens every
# adaptive tier. (Bedrock will reject it for non-adaptive models, which
# is the correct behavior — surface the model error rather than swallow
# the knob.)
assert result.get("output_config") == {"effort": "low"}
# Supported fields pass through.
assert result["max_tokens"] == 4096

View file

@ -499,14 +499,7 @@ def test_vertex_ai_partner_models_anthropic_remove_prompt_caching_scope_beta_hea
def test_vertex_ai_anthropic_output_config_effort_only_forwarded():
"""
Vertex AI Claude 4.6 / 4.7 accept ``output_config.effort`` on direct
``:rawPredict`` (verified end-to-end against ``us-east5`` for
``claude-opus-4-6`` and ``global`` for ``claude-opus-4-7``). The earlier
strip silently flattened every adaptive tier on Vertex, so ``low`` /
``medium`` / ``high`` / ``xhigh`` / ``max`` all produced identical
thinking with no tier differentiation.
"""
"""Vertex AI Claude 4.6/4.7 accept ``output_config.effort`` on rawPredict."""
config = VertexAIAnthropicConfig()
messages = [{"role": "user", "content": "What is 2+2?"}]
@ -568,15 +561,7 @@ def test_vertex_ai_anthropic_output_config_format_passes_through():
def test_vertex_ai_anthropic_output_config_format_plus_effort_preserved():
"""
Vertex AI Claude 4.6 / 4.7 accept ``output_config.effort`` on direct
``:rawPredict`` (verified end-to-end against ``us-east5`` for
``claude-opus-4-6`` and ``global`` for ``claude-opus-4-7``). Since the
strip was unjustified, ``effort`` must now ride along with ``format``.
We use a Claude 4.6 model id here because ``_apply_output_config`` only
accepts ``effort`` on adaptive-thinking 4.6/4.7 model ids.
"""
"""Both ``format`` and ``effort`` ride along on Vertex Claude 4.6/4.7."""
config = VertexAIAnthropicConfig()
messages = [{"role": "user", "content": "Return a person object."}]
@ -625,14 +610,7 @@ def test_vertex_ai_anthropic_output_config_non_dict_dropped():
def test_vertex_ai_anthropic_output_format_and_output_config_effort_preserved():
"""
Vertex AI Claude 4.6 / 4.7 accept ``output_config.effort`` on direct
``:rawPredict`` (verified end-to-end). When both ``output_format`` and
``output_config: {effort}`` are present, both must be forwarded — the
earlier ``effort`` strip caused silent loss of the requested adaptive
thinking tier on Vertex routes (``low``/``medium``/``high``/``xhigh``/``max``
all collapsed to identical adaptive thinking with no tier differentiation).
"""
"""Both ``output_format`` and ``output_config.effort`` are forwarded on Vertex 4.6/4.7."""
config = VertexAIAnthropicConfig()
messages = [{"role": "user", "content": "Extract structured data"}]

View file

@ -14,10 +14,7 @@ from litellm.types.utils import (
class TestXAIReasoningTokenFolding:
"""xAI breaks the OpenAI invariant total = prompt + completion by accounting
reasoning_tokens separately. ``_fold_reasoning_tokens_into_completion``
re-aligns Usage so downstream consumers see the OpenAI shape (o1/o3
semantics: completion_tokens includes reasoning)."""
"""``_fold_reasoning_tokens_into_completion`` re-aligns xAI Usage to the OpenAI invariant."""
@staticmethod
def _make_response(
@ -42,8 +39,7 @@ class TestXAIReasoningTokenFolding:
return response
def test_should_fold_when_total_explained_by_reasoning_gap(self):
# Real xAI live shape captured 2026-05-04: prompt=14, completion=10,
# total=336, reasoning=312. 14+10+312 == 336.
# xAI live shape: 14 + 10 + 312 == 336.
response = self._make_response(
prompt_tokens=14,
completion_tokens=10,
@ -58,7 +54,6 @@ class TestXAIReasoningTokenFolding:
assert usage.total_tokens == usage.prompt_tokens + usage.completion_tokens
def test_should_not_fold_when_already_normalised(self):
# OpenAI-normalised shape: completion already includes reasoning.
response = self._make_response(
prompt_tokens=14,
completion_tokens=322,
@ -68,7 +63,6 @@ class TestXAIReasoningTokenFolding:
XAIChatConfig._fold_reasoning_tokens_into_completion(response)
# Idempotent — no double-fold.
assert response.usage.completion_tokens == 322
def test_should_skip_when_no_reasoning_tokens(self):
@ -84,8 +78,7 @@ class TestXAIReasoningTokenFolding:
assert response.usage.completion_tokens == 10
def test_should_skip_when_gap_does_not_match_reasoning(self):
# Defensive guard: if xAI ever changes accounting and the gap stops
# equalling reasoning_tokens, refuse to fold rather than corrupt.
# Refuse to fold if xAI accounting changes (gap != reasoning_tokens).
response = self._make_response(
prompt_tokens=14,
completion_tokens=10,
@ -95,7 +88,6 @@ class TestXAIReasoningTokenFolding:
XAIChatConfig._fold_reasoning_tokens_into_completion(response)
# No fold; original values preserved.
assert response.usage.completion_tokens == 10
assert response.usage.total_tokens == 999

View file

@ -107,7 +107,6 @@ class TestXAICostCalculator:
def test_grok_4_cost_calculation(self):
"""Test cost calculation for grok-4 model."""
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
usage = Usage(
prompt_tokens=10,
completion_tokens=200,
@ -134,7 +133,6 @@ class TestXAICostCalculator:
def test_grok_3_fast_beta_cost_calculation(self):
"""Test cost calculation for grok-3-fast-beta model."""
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
usage = Usage(
prompt_tokens=20,
completion_tokens=300,
@ -176,7 +174,6 @@ class TestXAICostCalculator:
def test_edge_case_large_reasoning_tokens(self):
"""Test cost calculation when reasoning_tokens is larger than completion_tokens."""
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
usage = Usage(
prompt_tokens=12,
completion_tokens=50, # Less than reasoning_tokens
@ -204,7 +201,6 @@ class TestXAICostCalculator:
def test_tiered_pricing_above_128k_tokens(self):
"""Test tiered pricing for tokens above 128k."""
# Test with grok-4-fast-reasoning which has tiered pricing
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
usage = Usage(
prompt_tokens=150000, # Above 128k threshold
completion_tokens=100000, # Above 128k threshold
@ -234,7 +230,6 @@ class TestXAICostCalculator:
def test_tiered_pricing_below_128k_tokens(self):
"""Test that regular pricing is used for tokens below 128k threshold."""
# Test with grok-4-fast-reasoning which has tiered pricing
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
usage = Usage(
prompt_tokens=100000, # Below 128k threshold
completion_tokens=50000,
@ -263,7 +258,6 @@ class TestXAICostCalculator:
def test_tiered_pricing_grok_4_latest(self):
"""Test tiered pricing for grok-4-latest model."""
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
usage = Usage(
prompt_tokens=200000, # Above 128k threshold
completion_tokens=100000,
@ -292,7 +286,6 @@ class TestXAICostCalculator:
def test_tiered_pricing_output_tokens_below_128k(self):
"""Test that output tokens get tiered rate when input tokens > 128k, even if output tokens < 128k."""
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
usage = Usage(
prompt_tokens=150000, # Above 128k threshold
completion_tokens=50000, # Below 128k threshold
@ -339,16 +332,7 @@ class TestXAICostCalculator:
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
def test_already_normalised_usage_does_not_double_count_reasoning(self):
"""Cost calc receives Usage post-transformation (OpenAI invariant).
After XAIChatConfig.transform_response folds reasoning_tokens into
completion_tokens, the Usage block satisfies
``total_tokens == prompt_tokens + completion_tokens``. Cost calc must
detect this and skip the reasoning_tokens add-on, otherwise it
double-bills the reasoning tokens.
"""
# OpenAI-normalised shape: completion_tokens already includes the
# 100 reasoning tokens (so 100 visible + 100 reasoning -> 200).
"""Cost calc must not double-bill when Usage is already OpenAI-normalised."""
usage = Usage(
prompt_tokens=12,
completion_tokens=200,
@ -364,7 +348,6 @@ class TestXAICostCalculator:
prompt_cost, completion_cost = cost_per_token(model="grok-3-mini", usage=usage)
# Bill exactly what completion_tokens reports — no double-add.
expected_prompt_cost = 12 * 3e-7
expected_completion_cost = 200 * 5e-7

View file

@ -2905,9 +2905,9 @@ def test_gemini_embedding_2_ga_in_cost_map():
assert info.get("input_cost_per_audio_per_second") == 0.00016
assert info.get("input_cost_per_video_per_second") == 0.00079
if provider in ("vertex_ai-embedding-models", "vertex_ai"):
assert info.get("uses_embed_content") is True, (
f"{key} must have uses_embed_content=true for correct Vertex AI routing"
)
assert (
info.get("uses_embed_content") is True
), f"{key} must have uses_embed_content=true for correct Vertex AI routing"
def test_gemini_lyria_3_preview_models_in_cost_map():