From 2cb3f0f027e866498e990f53b5e4a6197b3b7667 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Mon, 4 May 2026 19:34:56 +0000 Subject: [PATCH] refactor: remove unnecessary comments from #27074 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Strip out the explanatory and historical comments that don't carry business-logic justification. Comments that simply narrate what code does — or that explain prior behavior, what was changed, or which PR introduced a fix — are removed. Docstrings are reduced to a one-line summary where the long form repeated information already evident from the code or test data. No code-behavior changes. All 643 affected unit tests still pass. Co-authored-by: Mateo Wang --- litellm/constants.py | 20 +--- litellm/llms/anthropic/chat/transformation.py | 107 +----------------- litellm/llms/anthropic/common_utils.py | 10 +- .../messages/transformation.py | 37 +----- .../llms/azure_ai/anthropic/transformation.py | 33 +----- litellm/llms/base_llm/chat/transformation.py | 6 - .../bedrock/chat/converse_transformation.py | 101 ++--------------- .../anthropic_claude3_transformation.py | 5 - .../anthropic_claude3_transformation.py | 4 - .../llms/databricks/chat/transformation.py | 19 ---- .../transformation.py | 5 - .../anthropic/output_params_utils.py | 7 +- .../anthropic/transformation.py | 4 - litellm/llms/xai/chat/transformation.py | 24 +--- litellm/llms/xai/cost_calculator.py | 11 +- litellm/types/llms/anthropic.py | 6 - litellm/types/llms/bedrock.py | 8 -- .../llms/anthropic/chat/conftest.py | 43 +------ .../test_anthropic_chat_transformation.py | 93 +++------------ .../test_reasoning_effort_translation.py | 50 +------- .../test_azure_anthropic_transformation.py | 36 ++---- ...ations_anthropic_claude3_transformation.py | 14 +-- .../chat/test_converse_transformation.py | 29 +---- .../test_anthropic_claude3_transformation.py | 92 ++------------- ...partner_models_anthropic_transformation.py | 28 +---- .../llms/xai/test_xai_chat_transformation.py | 14 +-- .../llms/xai/test_xai_cost_calculator.py | 19 +--- tests/test_litellm/test_utils.py | 6 +- 28 files changed, 92 insertions(+), 739 deletions(-) diff --git a/litellm/constants.py b/litellm/constants.py index 7708ba8b69a..6918e40cad1 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -202,17 +202,6 @@ DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET = int( DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET = int( os.getenv("DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET", 4096) ) -# ``xhigh`` / ``max`` budget extrapolation for legacy ``thinking.budget_tokens`` -# models (Claude 4.5 series + haiku). Continues the 2× progression -# 1024 → 2048 → 4096 from the existing low/medium/high tiers. These tiers -# also exist as adaptive ``output_config.effort`` enum values on Claude 4.6+ -# / 4.7; this constant only governs the budget-tokens fallback for models -# that aren't on the adaptive path. Per -# https://platform.claude.com/docs/en/build-with-claude/effort the ``effort`` -# enum is gated by model, but the legacy ``budget_tokens`` knob accepts any -# integer up to the model's max_tokens — adopting #27051's mapping here lets -# ``reasoning_effort=xhigh|max`` Just Work as a unified OpenAI-format knob -# regardless of which Anthropic API surface implements it. DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET = int( os.getenv("DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET", 8192) ) @@ -416,14 +405,7 @@ BEDROCK_MAX_POLICY_SIZE = int(os.getenv("BEDROCK_MAX_POLICY_SIZE", 75)) BEDROCK_MIN_THINKING_BUDGET_TOKENS = int( os.getenv("BEDROCK_MIN_THINKING_BUDGET_TOKENS", 1024) ) -# Anthropic's Messages API rejects ``thinking.budget_tokens < 1024`` with a -# 400. ``reasoning_effort='minimal'`` historically mapped to 128 (the global -# default) which always 400'd against direct Anthropic, Azure AI Anthropic, -# Vertex AI Anthropic, and Bedrock Invoke. Floor at the provider minimum so -# ``minimal`` is a usable tier on every Anthropic-backed route; Bedrock -# Converse already clamps server-side, this just unifies the behavior. -# Constant — not env-overridable — because it tracks Anthropic's published -# wire-protocol minimum, not a tunable. +# Anthropic's Messages API rejects thinking.budget_tokens < 1024. ANTHROPIC_MIN_THINKING_BUDGET_TOKENS = 1024 REPLICATE_POLLING_DELAY_SECONDS = float( os.getenv("REPLICATE_POLLING_DELAY_SECONDS", 0.5) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 7f16c69481f..6e76a9c85e9 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -224,11 +224,8 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): def _supports_effort_level(model: str, level: str) -> bool: """Check ``supports_{level}_reasoning_effort`` in the model map. - Mirrors the pattern used in ``openai/chat/gpt_5_transformation.py`` so - that adding support for a new effort level is a pure model-map change. - Handles bedrock-prefixed and vertex-prefixed model ids by stripping - the prefix and re-checking against ``litellm.model_cost`` directly, - so a Bedrock-routed Claude 4.6/4.7 keeps its model-map flag. + Strips bedrock/vertex prefixes so a provider-routed Claude still + resolves to the Anthropic model-map entry. """ key = f"supports_{level}_reasoning_effort" try: @@ -240,11 +237,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): return True except Exception: pass - # Bedrock and Vertex route the model id with a provider-prefix - # (e.g. ``bedrock/invoke/us.anthropic.claude-opus-4-7``). Strip - # known prefixes and look the resulting Anthropic-flavoured key - # up directly in ``litellm.model_cost`` so the lookup keeps - # working regardless of which route the request arrived on. candidates = [model] for prefix in ( "bedrock/converse/", @@ -277,27 +269,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): @staticmethod def _validate_effort_for_model(model: str, effort: Optional[str]) -> Optional[str]: - """Return ``None`` if ``effort`` is allowed on ``model``, else an error message. - - Centralises per-model gating for ``max`` and ``xhigh`` so the chat - completion path (``_apply_output_config``) and the /v1/messages - pass-through (``AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic``) - can't drift when a new model tier is added. Caller raises the - provider-appropriate exception type using the returned message. - - ``max`` is supported on Claude 4.6 (Opus + Sonnet) and Claude 4.7 - adaptive-thinking models per - https://platform.claude.com/docs/en/build-with-claude/effort. The - data-driven ``supports_max_reasoning_effort`` flag in - ``model_prices_and_context_window.json`` is the source of truth; - family-level ``_is_claude_4_6_model`` / ``_is_claude_4_7_model`` - checks remain as a fallback for OpenRouter / GitHub Copilot / - Vercel / Bedrock variants whose model-map entries don't yet carry - the flag. - - ``xhigh`` is purely data-driven via ``supports_xhigh_reasoning_effort`` - so enabling it for a new model is a model-map-only change. - """ + """Return ``None`` if ``effort`` is allowed on ``model``, else an error message.""" if effort == "max" and not ( AnthropicConfig._is_claude_4_6_model(model) or AnthropicConfig._is_claude_4_7_model(model) @@ -312,15 +284,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): @staticmethod def _model_supports_effort_param(model: str) -> bool: - """Whether the model accepts ``output_config.effort`` at all. - - Per https://platform.claude.com/docs/en/build-with-claude/effort the - ``output_config.effort`` parameter is supported on Opus 4.5+, Sonnet 4.6+ - and Mythos Preview; older Claude models reject it with a 400. Support is - encoded in ``model_prices_and_context_window.json`` via the - ``supports_*_reasoning_effort`` flags, so adding a new effort-capable - model is a pure model-map change. - """ + """Whether the model accepts ``output_config.effort`` at all.""" for level in ("low", "minimal", "medium", "high", "xhigh", "max"): if AnthropicConfig._supports_effort_level(model, level): return True @@ -908,16 +872,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): model: str, llm_provider: str = "anthropic", ) -> Optional[AnthropicThinkingParam]: - """Map an OpenAI-format ``reasoning_effort`` string to Anthropic's - ``thinking`` payload. - - Raises ``BadRequestError`` (clean 400) instead of ``ValueError`` (500) - on unmapped efforts so every caller — Anthropic native, Bedrock - Invoke/Converse, Databricks, Vertex Anthropic, Azure AI Anthropic, - and the experimental ``/v1/messages`` pass-through — surfaces a - consistent error to the user. Pass ``llm_provider`` so the - ``BadRequestError`` carries the right provider name in logs. - """ if reasoning_effort is None or reasoning_effort == "none": return None if AnthropicConfig._is_adaptive_thinking_model(model): @@ -940,32 +894,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): budget_tokens=DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, ) elif reasoning_effort == "xhigh": - # Continues the 2× progression of low/medium/high (1024/2048/4096). - # On adaptive models (Claude 4.6/4.7) the ``xhigh`` tier is - # already routed via ``output_config.effort=xhigh`` above; this - # branch only applies to budget-mode models (Claude 4.5 series + - # haiku) where the OpenAI-format ``reasoning_effort`` knob would - # otherwise 400 with ``Unmapped reasoning effort``. Keeps the - # cross-model UX uniform — ``reasoning_effort=xhigh`` Just Works - # regardless of which Anthropic API surface implements it. return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, ) elif reasoning_effort == "max": - # Same rationale as ``xhigh`` above — ``max`` is the adaptive - # enum's top tier on Claude 4.6/4.7, but for budget-mode models - # we extend the 2× progression (8192 → 16384) so the OpenAI- - # format alias is usable on every Claude model. return AnthropicThinkingParam( type="enabled", budget_tokens=DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET, ) elif reasoning_effort == "minimal": - # Anthropic Messages API rejects ``budget_tokens < 1024`` with a - # 400. Floor at the provider minimum so ``minimal`` is a usable - # tier on Anthropic / Azure AI Anthropic / Vertex AI Anthropic / - # Bedrock Invoke. Bedrock Converse already clamps server-side. return AnthropicThinkingParam( type="enabled", budget_tokens=max( @@ -1246,10 +1184,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): elif param == "thinking": optional_params["thinking"] = value elif param == "reasoning_effort" and isinstance(value, str): - # ``_map_reasoning_effort`` raises ``BadRequestError`` (400) - # directly on unmapped efforts (``disabled`` / ``invalid`` / - # ``""`` / ``xhigh``/``max`` on budget-mode Claude 4.5) so - # we no longer need to wrap a ``ValueError`` here. mapped_thinking = AnthropicConfig._map_reasoning_effort( reasoning_effort=value, model=model, @@ -1260,20 +1194,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): optional_params.pop("output_config", None) else: optional_params["thinking"] = mapped_thinking - # For Claude 4.6+ adaptive-thinking models, effort is - # controlled via ``output_config``, not - # ``thinking.budget_tokens``. Driven by - # ``supports_adaptive_thinking`` in the model map so - # adding a new adaptive Claude is a model-map-only change. if AnthropicConfig._is_adaptive_thinking_model(model): - # ``_map_reasoning_effort`` returns ``type=adaptive`` - # for any string on adaptive models without checking - # the value, so reject unmapped efforts here (matching - # the /v1/messages path) instead of relying on the - # downstream ``_apply_output_config`` check. Co-locating - # validation with the mapping prevents garbage from - # leaking into ``optional_params`` if ``map_openai_params`` - # is ever called without a subsequent ``transform_request``. mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( value ) @@ -1703,22 +1624,12 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): def _apply_output_config( self, data: dict, model: str, optional_params: dict ) -> None: - """Validate and apply output_config to the request data. - - Validation errors raise ``BadRequestError`` (clean 400) so callers - passing ``effort="disabled"`` / ``effort=""`` / unsupported tiers - for the model see a client-side error rather than a 500. - """ + """Validate and apply output_config to the request data.""" if "output_config" not in optional_params: return output_config = optional_params.get("output_config") if not output_config or not isinstance(output_config, dict): return - # When ``drop_params`` is set, strip ``output_config`` for models that - # cannot accept it (e.g. proxy fronting Claude Code at haiku-3, where - # the client always sends effort but the model rejects it). The user - # opted into silent fixup via the global flag — log a warning so the - # strip is still visible in logs. if litellm.drop_params is True and not self._model_supports_effort_param(model): litellm.verbose_logger.warning( "Dropping unsupported `output_config` for model=%s " @@ -1730,10 +1641,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): data.pop("output_config", None) return effort = output_config.get("effort") - # ``effort=""`` (empty string) and unmapped strings should be treated - # as invalid, not silently passed through. We use ``effort is not None`` - # here so empty string fails the membership check below. (The legacy - # ``if effort and ...`` short-circuit silently accepted ``""``.) valid_efforts = ["high", "medium", "low", "xhigh", "max"] if effort is not None and effort not in valid_efforts: raise litellm.exceptions.BadRequestError( @@ -1744,10 +1651,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): model=model, llm_provider=self.custom_llm_provider or "anthropic", ) - # Per-model gating for ``max`` / ``xhigh`` is centralised in - # ``_validate_effort_for_model`` so the chat path and the - # /v1/messages pass-through stay in lock-step when a new model - # tier lands. gate_error = self._validate_effort_for_model(model, effort) if gate_error is not None: raise litellm.exceptions.BadRequestError( diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index d711d05c687..869a7c5fbc4 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -273,15 +273,7 @@ class AnthropicModelInfo(BaseLLMModelInfo): @staticmethod def _is_adaptive_thinking_model(model: str) -> bool: - """Claude 4.6+ models use adaptive thinking with ``output_config.effort``. - - Driven by the ``supports_adaptive_thinking`` flag in - ``model_prices_and_context_window.json`` so that adding a new - adaptive-thinking model is a pure model-map change. Falls back to - the family-pattern check for OpenRouter / Vercel / Bedrock / - provider-prefixed variants whose model-map entries don't (yet) - carry the flag. - """ + """Claude 4.6+ models use adaptive thinking with ``output_config.effort``.""" from litellm.utils import _supports_factory try: diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index 8f35e79f568..35495d59610 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -47,9 +47,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): "inference_geo", "speed", "output_config", - # OpenAI-style tier knob — translated to native ``thinking`` + - # ``output_config`` in ``transform_anthropic_messages_request`` - # and popped before the request is forwarded. "reasoning_effort", # TODO: Add Anthropic `metadata` support # "metadata", @@ -176,22 +173,10 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): ) -> None: """Map OpenAI-style ``reasoning_effort`` to native Anthropic params. - The /v1/messages spec doesn't include ``reasoning_effort`` — without - this translation it gets silently dropped, leaving every adaptive - tier collapsed to the same behavior on Bedrock Invoke /v1/messages - (and on Anthropic / Azure AI / Vertex AI when callers pass it on - the messages route). Mirrors ``AnthropicConfig.map_openai_params`` - on the chat completion path so the two routes can't drift. - - - Pops ``reasoning_effort`` from ``optional_params`` so it never - reaches the wire. - - Caller-supplied ``thinking`` / ``output_config`` always win — we - don't override an explicit native value. - - Effort=``none`` clears thinking + output_config so callers can - opt out per request. - - Invalid efforts raise ``BadRequestError`` (clean 400) instead of - surfacing as 500s downstream. + Caller-supplied ``thinking`` / ``output_config`` win over the alias. + ``effort='none'`` clears both. Invalid efforts raise a 400. """ + from litellm.exceptions import BadRequestError as _BadRequestError from litellm.llms.anthropic.chat.transformation import ( REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT, AnthropicConfig, @@ -201,12 +186,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): if not isinstance(reasoning_effort, str): return - # ``_map_reasoning_effort`` raises ``BadRequestError`` (400) directly - # on unmapped efforts. The /v1/messages pass-through surfaces errors - # as ``AnthropicError``; convert here so callers see a provider-shaped - # 400 rather than the LiteLLM-shaped one. - from litellm.exceptions import BadRequestError as _BadRequestError - try: mapped_thinking = AnthropicConfig._map_reasoning_effort( reasoning_effort=reasoning_effort, model=model @@ -224,12 +203,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( reasoning_effort ) - # ``_map_reasoning_effort`` returns ``type=adaptive`` for any - # string on adaptive models without checking the value. The - # chat completion path validates the resolved effort downstream - # via ``_apply_output_config``; /v1/messages has no equivalent - # downstream check, so reject unmapped values here so callers - # see a clean 400 instead of a 500 from the provider. if mapped_effort is None: raise AnthropicError( message=( @@ -239,10 +212,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): ), status_code=400, ) - # Per-model gating for ``max`` / ``xhigh`` is centralised in - # ``AnthropicConfig._validate_effort_for_model`` so the chat - # completion path and this /v1/messages pass-through stay in - # lock-step when a new model tier lands. gate_error = AnthropicConfig._validate_effort_for_model( model, mapped_effort ) diff --git a/litellm/llms/azure_ai/anthropic/transformation.py b/litellm/llms/azure_ai/anthropic/transformation.py index f3bd6012ec5..e176a4d860e 100644 --- a/litellm/llms/azure_ai/anthropic/transformation.py +++ b/litellm/llms/azure_ai/anthropic/transformation.py @@ -15,21 +15,11 @@ if TYPE_CHECKING: def _promote_extra_body_to_optional_params(optional_params: dict) -> None: """Promote anthropic-native passthrough keys out of ``extra_body``. - ``azure_ai`` is registered in ``litellm.openai_compatible_providers``, so - ``add_provider_specific_params_to_optional_params`` (litellm/utils.py) - auto-stuffs any non-OpenAI kwarg (e.g. ``output_config={"effort": "..."}``) - into ``optional_params["extra_body"]``. For the Azure→Anthropic route the - user's intent is to forward those params to Anthropic, so promote them to - the top level of ``optional_params`` so: - - * ``AnthropicConfig._apply_output_config`` validates ``effort`` values - (matching the native ``anthropic`` provider's 400 on bad efforts). - * Valid passthroughs (e.g. ``output_config``, ``thinking``) actually - reach the request body instead of being silently dropped by the - ``data.pop("extra_body", None)`` strip below. - - ``setdefault`` is used so an explicit top-level value is never clobbered - by a duplicate inside ``extra_body``. + ``azure_ai`` is an OpenAI-compatible provider, so non-OpenAI kwargs like + ``output_config`` get auto-routed into ``extra_body`` by + ``add_provider_specific_params_to_optional_params``. For the Azure→Anthropic + route those keys must reach the request body and be validated, so promote + them. ``setdefault`` keeps explicit top-level values authoritative. """ extra_body = optional_params.get("extra_body") if not isinstance(extra_body, dict) or not extra_body: @@ -66,9 +56,6 @@ class AzureAnthropicConfig(AnthropicConfig): 1. API key via 'api-key' header 2. Azure AD token via 'Authorization: Bearer ' header """ - # Promote anthropic-native passthrough keys (``output_config``, - # ``thinking``, ``mcp_servers``, ...) out of ``extra_body`` so the - # flag detection below (``is_mcp_server_used``, etc.) sees them. _promote_extra_body_to_optional_params(optional_params) # Convert dict to GenericLiteLLMParams if needed @@ -133,18 +120,8 @@ class AzureAnthropicConfig(AnthropicConfig): Transform request using parent AnthropicConfig, then remove unsupported params. Azure Anthropic doesn't support extra_body, max_retries, or stream_options parameters. """ - # Promote anthropic-native passthrough keys (``output_config``, - # ``thinking``, ...) out of ``extra_body`` BEFORE delegating to - # ``AnthropicConfig.transform_request``. Without this: - # * ``output_config`` is silently dropped by the ``extra_body`` pop - # below, and - # * ``_apply_output_config`` never validates ``effort`` values, so - # ``effort="invalid"`` quietly reaches the model with default - # behavior instead of returning a clean 400 (as the native - # ``anthropic`` provider does). _promote_extra_body_to_optional_params(optional_params) - # Call parent transform_request data = super().transform_request( model=model, messages=messages, diff --git a/litellm/llms/base_llm/chat/transformation.py b/litellm/llms/base_llm/chat/transformation.py index e36ba0c14a1..bec25916c4b 100644 --- a/litellm/llms/base_llm/chat/transformation.py +++ b/litellm/llms/base_llm/chat/transformation.py @@ -84,12 +84,6 @@ class BaseConfig(ABC): @classmethod def get_config(cls): - # Subclasses lean on this to surface their public default settings - # (e.g. ``max_tokens``) as request params. Anything ``_``-prefixed is - # treated as private (lookup tables, ABC machinery, internal flags) - # and must not leak into the wire body — a tuple/dict/frozenset class - # attribute would otherwise serialise into the request as an extra - # top-level key and the provider would 400 it. return { k: v for k, v in cls.__dict__.items() diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 5a0087a270e..dbd55e957fd 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -189,8 +189,6 @@ class AmazonConverseConfig(BaseConfig): @classmethod def get_config(cls): - # ``_``-prefixed names are private (lookup tables, ABC machinery, - # internal flags) and must not leak into the wire body. return { k: v for k, v in cls.__dict__.items() @@ -415,59 +413,19 @@ class AmazonConverseConfig(BaseConfig): """ Handle the reasoning_effort parameter based on the model type. - Different model families handle reasoning effort differently: - - GPT-OSS models: Keep reasoning_effort as-is (passed to additionalModelRequestFields) - - Nova 2 models: Transform to reasoningConfig structure - - Other models (Anthropic, etc.): Convert to thinking parameter - - For Claude 4.6 / 4.7 (adaptive thinking) the tier is carried via - ``output_config.effort`` rather than ``thinking.budget_tokens``. We - validate the effort with the same rules ``AnthropicConfig._apply_output_config`` - uses (low/medium/high/xhigh/max + per-model gating) and stage the - validated dict on ``optional_params["output_config"]`` so it rides - along to ``additionalModelRequestFields`` on the Anthropic-on-Bedrock - wire path. Without this the silent strip in ``_prepare_request_params`` - collapsed every adaptive tier to identical behavior. - - Args: - model: The model identifier - reasoning_effort: The reasoning effort value - optional_params: Dictionary of optional parameters to update in-place - - Examples: - >>> config = AmazonConverseConfig() - >>> params = {} - >>> config._handle_reasoning_effort_parameter("gpt-oss-model", "high", params) - >>> params - {'reasoning_effort': 'high'} - - >>> params = {} - >>> config._handle_reasoning_effort_parameter("amazon.nova-2-lite-v1:0", "high", params) - >>> params - {'reasoningConfig': {'type': 'enabled', 'maxReasoningEffort': 'high'}} - - >>> params = {} - >>> config._handle_reasoning_effort_parameter("anthropic.claude-3", "high", params) - >>> params - {'thinking': {'type': 'enabled', 'budget_tokens': 10000}} + - GPT-OSS models: passed through unchanged via additionalModelRequestFields. + - Nova 2 models: transformed to reasoningConfig. + - Anthropic models: mapped to ``thinking`` (and ``output_config.effort`` on + adaptive Claude 4.6 / 4.7). """ if "gpt-oss" in model: - # GPT-OSS models: keep reasoning_effort as-is - # It will be passed through to additionalModelRequestFields optional_params["reasoning_effort"] = reasoning_effort elif self._is_nova_2_model(model): - # Nova 2 models: transform to reasoningConfig reasoning_config = self._transform_reasoning_effort_to_reasoning_config( reasoning_effort ) optional_params.update(reasoning_config) else: - # Anthropic and other models: convert to thinking parameter. - # ``_map_reasoning_effort`` raises ``BadRequestError`` (400) - # directly on unmapped efforts (``disabled`` / ``invalid`` / - # ``""`` / ``xhigh``/``max`` on budget-mode Claude 4.5); pass - # ``llm_provider="bedrock_converse"`` so the error carries the - # right provider name. mapped_thinking = AnthropicConfig._map_reasoning_effort( reasoning_effort=reasoning_effort, model=model, @@ -478,22 +436,7 @@ class AmazonConverseConfig(BaseConfig): optional_params.pop("output_config", None) else: optional_params["thinking"] = mapped_thinking - # Adaptive-thinking models (Claude 4.6 / 4.7+) take the - # tier via ``output_config.effort``. Mirror the mapping - # used by ``AnthropicConfig.map_openai_params`` and apply - # the same validation rules so unmapped/garbage efforts - # surface as a 400 instead of being silently flattened on - # the wire. Driven by ``supports_adaptive_thinking`` in - # ``model_prices_and_context_window.json`` so a future - # adaptive Claude release lands as a model-map change. if AnthropicConfig._is_adaptive_thinking_model(model): - # Use ``.get()`` without a fallback so unmapped efforts - # (e.g. ``"disabled"``) surface as a clean 400 here - # rather than leaking the raw garbage string through to - # ``_validate_anthropic_adaptive_effort`` (which does - # catch it, but only because validation happens to run). - # Matches the /v1/messages pattern where validation is - # co-located with the mapping. mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( reasoning_effort ) @@ -514,16 +457,7 @@ class AmazonConverseConfig(BaseConfig): @staticmethod def _validate_anthropic_adaptive_effort(model: str, effort: str) -> None: - """Validate ``output_config.effort`` for adaptive-thinking Claude 4.6/4.7 - on Bedrock. Raises ``BadRequestError`` (clean 400) instead of letting - a downstream ``ValueError`` surface as 500. - - Per-model gating for ``max``/``xhigh`` is delegated to - ``AnthropicConfig._validate_effort_for_model`` so the Bedrock Converse - path and the Anthropic chat / ``/v1/messages`` paths can't drift when - a new gated effort tier is added. ``_supports_effort_level`` on - ``AnthropicConfig`` already handles Bedrock-prefixed model ids. - """ + """Validate ``output_config.effort`` for adaptive-thinking Claude 4.6/4.7.""" valid_efforts = {"high", "medium", "low", "xhigh", "max"} if effort not in valid_efforts: raise litellm.exceptions.BadRequestError( @@ -1283,13 +1217,9 @@ class AmazonConverseConfig(BaseConfig): ) inference_params.pop("json_mode", None) # used for handling json_schema - # Anthropic-only ``output_config`` (snake_case) is the adaptive- - # thinking effort payload (e.g. ``{"effort": "max"}``) for Claude - # 4.6/4.7. On Bedrock Converse it must ride along inside - # ``additionalModelRequestFields`` so the model actually sees the - # tier; stripping it (the prior behavior) silently flattened every - # adaptive tier to identical thinking. Only the Bedrock-native - # ``outputConfig`` (camelCase) goes at the top level. + # Anthropic-only ``output_config`` (snake_case) — re-attached to + # ``additionalModelRequestFields`` for Anthropic models below. The + # Bedrock-native ``outputConfig`` (camelCase) is handled separately. anthropic_output_config = inference_params.pop("output_config", None) # Extract requestMetadata before processing other parameters @@ -1342,18 +1272,11 @@ class AmazonConverseConfig(BaseConfig): additional_request_params ) - # Re-attach the Anthropic ``output_config`` (e.g. adaptive thinking - # ``{"effort": "max"}``) onto additional_request_params for Anthropic - # Bedrock models so the wire request carries the requested tier. Other - # model families (Nova, GPT-OSS, ...) don't accept it; drop it for them. if anthropic_output_config is not None and isinstance( anthropic_output_config, dict ): base_model = BedrockModelInfo.get_base_model(model) if base_model.startswith("anthropic"): - # When ``drop_params`` is set, strip for models that don't - # accept effort (e.g. proxy routing Claude Code at haiku-3). - # Otherwise forward and let Bedrock surface the model's error. if ( litellm.drop_params is True and not AnthropicConfig._model_supports_effort_param(model) @@ -1495,12 +1418,8 @@ class AmazonConverseConfig(BaseConfig): # Append pre-formatted tools (systemTool etc.) after transformation bedrock_tools.extend(pre_formatted_tools) - # Auto-attach the effort beta header for non-adaptive Anthropic - # models on Bedrock Converse (i.e. Opus 4.5). Claude 4.6/4.7 accept - # ``output_config.effort`` as a stable, GA feature with no beta - # header; Opus 4.5 still gates it behind ``effort-2025-11-24``. The - # check mirrors ``AnthropicModelInfo.is_effort_used`` (which returns - # False for adaptive models) so we don't double-flag adaptive routes. + # Opus 4.5 gates ``output_config.effort`` behind a beta header; + # Claude 4.6/4.7 accept it without one. base_model = BedrockModelInfo.get_base_model(model) if base_model.startswith("anthropic"): output_config = additional_request_params.get("output_config") diff --git a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py index 7de05c977ca..c883ab68dff 100644 --- a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py @@ -169,11 +169,6 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): anthropic_request.pop("model", None) anthropic_request.pop("stream", None) anthropic_request.pop("output_format", None) - # ``output_config`` (e.g. ``{"effort": "max"}``) is the adaptive-thinking - # tier payload for Claude 4.6 / 4.7. Bedrock Invoke accepts it for - # those models — stripping it (the prior behavior) silently flattened - # every adaptive tier on this route. Forward it; if the model rejects - # it the surfaced error is correct, vs. swallowing the user's knob. if "anthropic_version" not in anthropic_request: anthropic_request["anthropic_version"] = self.anthropic_version diff --git a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py index a717dfb78ba..a44d5b0ba96 100644 --- a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py @@ -582,10 +582,6 @@ class AmazonAnthropicClaudeMessagesConfig( if filtered_betas: anthropic_messages_request["anthropic_beta"] = filtered_betas - # 6a. When ``drop_params`` is set, strip ``output_config`` for models - # that don't accept it (e.g. proxy fronting Claude Code at haiku-3). - # Without this, every Claude Code request to a pre-4.5 Anthropic model - # routes a forced 400 from Bedrock that the client can't fix. if ( litellm.drop_params is True and "output_config" in anthropic_messages_request diff --git a/litellm/llms/databricks/chat/transformation.py b/litellm/llms/databricks/chat/transformation.py index 4cd3ad9e425..fedb3d1165b 100644 --- a/litellm/llms/databricks/chat/transformation.py +++ b/litellm/llms/databricks/chat/transformation.py @@ -334,11 +334,6 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig): ) # unsupported for claude models - if json_schema -> convert to tool call if "reasoning_effort" in non_default_params and "claude" in model: - # ``_map_reasoning_effort`` raises ``BadRequestError`` (400) - # directly on unmapped efforts; pass ``llm_provider="databricks"`` - # so the surfaced error carries the correct provider name (the - # default is ``"anthropic"``, which would mislead users routing - # via Databricks Foundation Model APIs). reasoning_effort_value = non_default_params.get("reasoning_effort") mapped_thinking = AnthropicConfig._map_reasoning_effort( reasoning_effort=reasoning_effort_value, @@ -350,21 +345,7 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig): optional_params.pop("output_config", None) else: optional_params["thinking"] = mapped_thinking - # For Claude 4.6+ adaptive models, ``_map_reasoning_effort`` - # returns ``type=adaptive`` for ANY non-None / non-"none" - # string without validating the value, so reject unmapped - # efforts here and set ``output_config.effort`` (matching the - # Anthropic native / Bedrock Converse / Bedrock Invoke / - # /v1/messages paths). Driven by ``supports_adaptive_thinking`` - # in the model map so future adaptive Claudes land via a - # model-map update rather than a code release — the - # Anthropic-native and Bedrock routes already use the same - # helper, so all three paths stay in lock-step. if AnthropicConfig._is_adaptive_thinking_model(model): - # ``reasoning_effort_value`` comes from ``non_default_params`` - # so its static type is ``Any | None``. Narrow to ``str`` for - # the mapping lookup; non-strings fall through to the - # ``BadRequestError`` below with a clean validation message. mapped_effort: Optional[str] = None if isinstance(reasoning_effort_value, str): mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get( diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py index 714877457fc..4be4c2d5e78 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py @@ -159,11 +159,6 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert "model", None ) # do not pass model in request body to vertex ai - # Vertex AI Claude accepts ``output_config.format`` (structured outputs), - # ``output_format``, and ``output_config.effort`` (adaptive-thinking - # tier on Claude 4.6 / 4.7, verified end-to-end). The shared sanitize - # helper now no-ops for ``effort`` and remains the single hook for any - # future Vertex-only key drift. sanitize_vertex_anthropic_output_params(anthropic_messages_request) return anthropic_messages_request diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py index 891c8e4d058..a33ad677789 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/output_params_utils.py @@ -11,11 +11,8 @@ keeps the parent module's import surface narrow. """ # Keys inside ``output_config`` that Vertex AI Claude does not accept. -# Vertex now accepts ``output_config.effort`` for the adaptive-thinking -# Claude 4.6 / 4.7 models on direct ``:rawPredict`` (verified end-to-end -# against ``us-east5`` for ``opus-4-6`` and ``global`` for ``opus-4-7``). -# Keep this set narrow and only add a key here once a 400 "Extra inputs are -# not permitted" is reproducible against the live Vertex endpoint. +# Add an entry only when a 400 "Extra inputs are not permitted" is +# reproducible against the live Vertex endpoint. VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS: frozenset = frozenset() diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py index 48920402e26..4627d9f6df3 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/transformation.py @@ -106,10 +106,6 @@ class VertexAIAnthropicConfig(AnthropicConfig): data.pop("model", None) # vertex anthropic doesn't accept 'model' parameter - # Sanitize ``output_config`` / ``output_format`` for Vertex parity. - # Vertex now accepts ``output_config.effort`` for adaptive-thinking Claude - # 4.6 / 4.7 models, so the helper is a no-op for ``effort``; it remains - # the single hook for future Vertex-only sanitization. sanitize_vertex_anthropic_output_params(data) tools = optional_params.get("tools") diff --git a/litellm/llms/xai/chat/transformation.py b/litellm/llms/xai/chat/transformation.py index b1c0aaccf1d..6300868a641 100644 --- a/litellm/llms/xai/chat/transformation.py +++ b/litellm/llms/xai/chat/transformation.py @@ -224,12 +224,6 @@ class XAIChatConfig(OpenAIGPTConfig): except Exception as e: verbose_logger.debug(f"Error extracting X.AI web search usage: {e}") - # X.AI excludes reasoning_tokens from completion_tokens, breaking the - # OpenAI invariant total_tokens == prompt_tokens + completion_tokens. - # OpenAI o1/o3 fold reasoning into completion_tokens; align X.AI to - # match so downstream consumers (including litellm's own usage tests) - # see a self-consistent Usage object. Cost calc already accounts for - # reasoning tokens separately in xai.cost_calculator. self._fold_reasoning_tokens_into_completion(response) return response @@ -237,15 +231,10 @@ class XAIChatConfig(OpenAIGPTConfig): def _fold_reasoning_tokens_into_completion(model_response: ModelResponse) -> None: """Reconcile xAI Usage to the OpenAI invariant. - xAI returns ``completion_tokens`` covering only visible output and - accounts ``reasoning_tokens`` separately, while still rolling them - into ``total_tokens``. OpenAI's published contract (o1/o3) includes - reasoning in ``completion_tokens``. Tests that assert - ``total_tokens == prompt_tokens + completion_tokens`` (e.g. - ``_usage_format_tests``) fail on the raw xAI shape. - - This helper is idempotent: if ``completion_tokens`` already covers - the gap, no change is made. + xAI accounts ``reasoning_tokens`` separately from + ``completion_tokens`` while still summing them into ``total_tokens``. + OpenAI's contract (o1/o3) folds reasoning into ``completion_tokens``, + so fold here to keep ``total = prompt + completion``. Idempotent. """ usage = getattr(model_response, "usage", None) if usage is None: @@ -262,13 +251,10 @@ class XAIChatConfig(OpenAIGPTConfig): completion_tokens = int(getattr(usage, "completion_tokens", 0) or 0) total_tokens = int(getattr(usage, "total_tokens", 0) or 0) - # Already consistent — nothing to do. if total_tokens == prompt_tokens + completion_tokens: return - # Only fold when xAI's accounting (total = prompt + completion + - # reasoning) explains the gap. This guards against double-counting - # if xAI ever changes their semantics. + # Guard against double-counting if xAI changes accounting. if total_tokens != prompt_tokens + completion_tokens + reasoning_tokens: return diff --git a/litellm/llms/xai/cost_calculator.py b/litellm/llms/xai/cost_calculator.py index df5a7beffa3..8edfd0c27ad 100644 --- a/litellm/llms/xai/cost_calculator.py +++ b/litellm/llms/xai/cost_calculator.py @@ -25,13 +25,10 @@ def cost_per_token(model: str, usage: Usage) -> Tuple[float, float]: Returns: Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd """ - # XAI-specific completion cost calculation - # For XAI models, completion is billed as (visible completion tokens + reasoning tokens). - # The transformation layer normalises Usage to the OpenAI invariant - # (completion_tokens includes reasoning_tokens), so detect that and avoid - # double-counting. Fall back to the raw xAI shape (visible-only completion + - # reasoning kept in completion_tokens_details) for callers that bypass the - # transformation, e.g. proxy logs replayed straight into cost calc. + # XAI-specific completion cost: completion is billed as visible + reasoning + # tokens. Detect when the transformation layer already folded them so we + # don't double-count; fall back to raw xAI shape for callers that bypass + # the transformation (e.g. proxy logs replayed into cost calc). prompt_tokens = int(getattr(usage, "prompt_tokens", 0) or 0) completion_tokens = int(getattr(usage, "completion_tokens", 0) or 0) total_tokens = int(getattr(usage, "total_tokens", 0) or 0) diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py index 6c57d3dc07a..1c4d31d21ad 100644 --- a/litellm/types/llms/anthropic.py +++ b/litellm/types/llms/anthropic.py @@ -393,12 +393,6 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False): AnthropicOutputConfig ] # Configuration for Claude's output behavior cache_control: Optional[Dict[str, Any]] # Automatic prompt caching - # OpenAI-style ``reasoning_effort`` is accepted on /v1/messages so callers - # can drive adaptive/extended thinking with a single tier-name knob (the - # same vocabulary as the chat completion path). The transformation layer - # maps it to native Anthropic ``thinking`` + ``output_config`` and pops - # this key before the request is forwarded — no provider receives - # ``reasoning_effort`` on the wire. reasoning_effort: Optional[str] diff --git a/litellm/types/llms/bedrock.py b/litellm/types/llms/bedrock.py index 72e509d178f..64827db13f6 100644 --- a/litellm/types/llms/bedrock.py +++ b/litellm/types/llms/bedrock.py @@ -1041,12 +1041,4 @@ class BedrockInvokeAnthropicMessagesRequest(TypedDict, total=False): # `metadata` is part of the common Anthropic Messages API shape. thinking: dict metadata: dict - - # ``output_config`` is the adaptive-thinking effort payload for - # Claude 4.6 / 4.7 (e.g. ``{"effort": "max"}``). Bedrock Invoke - # accepts it for these models when ``thinking={"type": "adaptive"}``. - # Without this field in the allowlist, the runtime filter in - # ``AmazonAnthropicClaudeMessagesConfig.transform_anthropic_messages_request`` - # silently drops it and every adaptive tier collapses to identical - # behavior on /v1/messages. output_config: dict diff --git a/tests/test_litellm/llms/anthropic/chat/conftest.py b/tests/test_litellm/llms/anthropic/chat/conftest.py index 7dc6aeeb902..7e93e3b539b 100644 --- a/tests/test_litellm/llms/anthropic/chat/conftest.py +++ b/tests/test_litellm/llms/anthropic/chat/conftest.py @@ -1,37 +1,10 @@ +"""Force ``litellm.model_cost`` to load from the PR-local JSON for these tests. + +By default ``litellm.model_cost`` is fetched from the main branch on GitHub, +which lags behind PR-branch flag additions. This fixture loads the local +file so per-model flag tests pass in CI as well as locally. +See https://github.com/BerriAI/litellm/issues/27122. """ -Local-conftest for ``tests/test_litellm/llms/anthropic/chat``. - -Why this exists ---------------- -``litellm.model_cost`` is loaded once at ``litellm.__init__`` time. By default -it fetches ``model_prices_and_context_window.json`` from the **main branch on -GitHub** (``litellm.model_cost_map_url``) — *not* from the PR-branch JSON in -the working tree. That works fine in production (operators get new models -without redeploying litellm) but is the wrong default for tests, which need -to validate the code in front of them against the data in front of them. - -Several anthropic chat transformation tests (``test_supports_effort_level_*``, -``test_anthropic_model_supports_effort_param_*``) assert per-model flags -like ``supports_max_reasoning_effort`` / ``supports_xhigh_reasoning_effort`` -that may exist in the PR-local JSON but not yet in main's JSON. Without this -fixture those tests pass locally (where AGENTS.md tells contributors to set -``LITELLM_LOCAL_MODEL_COST_MAP=True``) but fail in CI (which doesn't set the -env var) — the chicken-and-egg PR adds flag → CI fetches main without flag -→ test fails → flag never lands on main. - -The fixture force-loads ``litellm.model_cost`` from the local JSON for every -test in this directory. ``tests/test_litellm/conftest.py`` already snapshots -and restores ``litellm.model_cost`` per-function, so this mutation is safe -and contained. - -This is a scoped workaround. The proper fix is to set -``LITELLM_LOCAL_MODEL_COST_MAP=True`` globally in the test workflow once the -~10 inline-set test files have been audited and the few tests that exercise -the remote-fetch / integrity-validation path have been given carve-outs. -Tracked at https://github.com/BerriAI/litellm/issues/27122. -""" - -import os import pytest @@ -41,10 +14,6 @@ from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map @pytest.fixture(autouse=True) def _use_pr_local_model_cost_map(monkeypatch): - """Force ``litellm.model_cost`` to the PR-branch JSON for the duration of - each test. ``monkeypatch`` reverts the env var after the test; the parent - conftest restores ``litellm.model_cost`` from its snapshot. - """ monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") monkeypatch.setattr( litellm, diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py index 1aa753ff8f0..26ed8d29c11 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -1632,7 +1632,6 @@ def test_effort_validation(): ) assert result["output_config"]["effort"] == effort - # Invalid value should raise BadRequestError (clean 400, not a 500). with pytest.raises( litellm.exceptions.BadRequestError, match="Invalid effort value" ): @@ -1685,10 +1684,7 @@ def test_effort_validation_with_opus_46(): def test_max_effort_rejected_for_opus_45(): - """Test that effort='max' is rejected when using Claude Opus 4.5. - - Surfaces as a clean 400 BadRequestError, not a 500 ValueError. - """ + """Test that effort='max' is rejected when using Claude Opus 4.5.""" config = AnthropicConfig() messages = [{"role": "user", "content": "Test"}] @@ -1749,12 +1745,7 @@ def test_effort_with_other_features(): def test_anthropic_drop_params_strips_output_config_for_pre_4_5_models(): - """ - Proxies fronting Claude Code at pre-4.5 Anthropic models receive - ``output_config`` injected by the client; without ``drop_params`` Bedrock / - Anthropic 400s. With ``drop_params=True`` we strip it (logged) so the - request can succeed. - """ + """``drop_params=True`` strips unsupported ``output_config`` for pre-4.5 models.""" config = AnthropicConfig() messages = [{"role": "user", "content": "Hello"}] @@ -1796,10 +1787,7 @@ def test_anthropic_drop_params_keeps_output_config_for_supporting_models(): def test_anthropic_drop_params_false_forwards_to_unsupported_model(): - """ - Default behavior: forward ``output_config`` and let the provider 400. - This is the contract for users who want strict, debuggable failures. - """ + """Default ``drop_params=False`` forwards ``output_config`` and lets the provider 400.""" config = AnthropicConfig() messages = [{"role": "user", "content": "Hello"}] @@ -2059,10 +2047,7 @@ def test_get_config_without_model_uses_fallback(): def test_get_config_does_not_leak_module_constants(): - """``BaseConfig.get_config`` returns class attributes; the - reasoning-effort mapping must not be one of them or it ends up - serialised onto the wire as an extra request key. - """ + """``get_config`` must not leak the reasoning-effort lookup table onto the wire.""" cfg = AnthropicConfig.get_config(model="claude-opus-4-7") for forbidden in ( "REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT", @@ -2078,10 +2063,6 @@ def test_get_config_does_not_leak_module_constants(): ("claude-opus-4-7", "xhigh", True), ("claude-opus-4-6", "max", True), ("claude-opus-4-6", "xhigh", False), - # ``max`` is documented as supported on Sonnet 4.6 (Claude 4.6 family). - # The model-map JSON now carries ``supports_max_reasoning_effort: true`` - # for every Sonnet 4.6 entry; ``_supports_effort_level`` should report - # ``True`` on every route prefix (anthropic, bedrock, vertex, azure). ("claude-sonnet-4-6", "max", True), ("claude-sonnet-4-6", "xhigh", False), ("bedrock/invoke/us.anthropic.claude-opus-4-7", "max", True), @@ -2094,27 +2075,21 @@ def test_get_config_does_not_leak_module_constants(): ], ) def test_supports_effort_level_handles_provider_prefixes(model, level, expected): - """``_supports_effort_level`` must handle bedrock/ vertex_ai/ azure_ai/ - prefixed model ids so per-model gating works on every route, not just - the bare-Anthropic chat completion path. - """ + """``_supports_effort_level`` resolves bedrock/vertex/azure-prefixed model ids.""" assert AnthropicConfig._supports_effort_level(model, level) is expected @pytest.mark.parametrize( "model,effort,expect_error", [ - # ``max`` accepted on 4.6 / 4.7 (family fallback) and rejected on 4.5. ("claude-opus-4-6", "max", False), ("claude-sonnet-4-6", "max", False), ("claude-opus-4-7", "max", False), ("claude-opus-4-5-20251101", "max", True), ("claude-sonnet-4-5", "max", True), - # ``xhigh`` data-driven; only 4.7 carries the flag in the model map. ("claude-opus-4-7", "xhigh", False), ("claude-opus-4-6", "xhigh", True), ("claude-sonnet-4-6", "xhigh", True), - # Lower efforts and ``None`` always pass the gate. ("claude-opus-4-5-20251101", "high", False), ("claude-haiku-4-5", "low", False), ("claude-opus-4-5-20251101", None, False), @@ -2123,13 +2098,6 @@ def test_supports_effort_level_handles_provider_prefixes(model, level, expected) def test_validate_effort_for_model_centralises_per_model_gating( model, effort, expect_error ): - """``_validate_effort_for_model`` is the single source of truth for the - per-model ``max`` / ``xhigh`` gating that ``_apply_output_config`` (chat - completion path) and ``AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic`` - (/v1/messages pass-through) both rely on. Both call sites raise their - own provider-appropriate exception, but the gating decision must come - from one place to prevent drift when a new model tier lands. - """ err = AnthropicConfig._validate_effort_for_model(model, effort) if expect_error: assert err is not None @@ -2342,14 +2310,12 @@ def test_reasoning_effort_maps_to_budget_thinking_for_non_opus_4_6(): """ config = AnthropicConfig() - # Test with Claude Sonnet 4.5 (non-Opus 4.6 model). - # ``minimal`` floors at the Anthropic provider minimum (1024) because - # Anthropic / Azure / Vertex / Bedrock Invoke 400 below that. + # ``minimal`` floors at ANTHROPIC_MIN_THINKING_BUDGET_TOKENS (1024). test_cases = [ - ("low", 1024), # DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET - ("medium", 2048), # DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET - ("high", 4096), # DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET - ("minimal", 1024), # ANTHROPIC_MIN_THINKING_BUDGET_TOKENS (provider floor) + ("low", 1024), + ("medium", 2048), + ("high", 4096), + ("minimal", 1024), ] for effort, expected_budget in test_cases: @@ -2447,16 +2413,7 @@ def test_reasoning_effort_does_not_set_output_config_for_older_models(): ], ) def test_max_effort_accepted_for_sonnet_46_variants(model): - """``effort='max'`` is documented as supported on Claude 4.6 (Opus + Sonnet) - and Claude 4.7 (https://platform.claude.com/docs/en/build-with-claude/effort). - - Earlier versions of this test asserted a 400 for Sonnet 4.6, mirroring an - Opus-only allow-list in ``_apply_output_config``. That gate has since been - widened to ``_is_claude_4_6_model`` (Opus + Sonnet) and the - ``supports_max_reasoning_effort`` JSON flag, matching Anthropic's published - matrix. Verify the param actually flows through every Sonnet 4.6 id - variant our routing layer might see. - """ + """``effort='max'`` is supported on Claude 4.6 (Opus + Sonnet) and 4.7.""" config = AnthropicConfig() messages = [{"role": "user", "content": "Test"}] @@ -2552,9 +2509,7 @@ def test_reasoning_effort_none_omits_thinking_and_output_config(model): ["disabled", "invalid", ""], ) def test_reasoning_effort_garbage_raises_bad_request(effort): - """Unmapped / garbage / empty-string reasoning_effort surfaces as a clean - 400 ``BadRequestError`` instead of letting ``ValueError`` propagate as 500. - """ + """Unmapped reasoning_effort raises BadRequestError (clean 400, not a 500).""" config = AnthropicConfig() with pytest.raises(litellm.exceptions.BadRequestError): @@ -2573,17 +2528,7 @@ def test_reasoning_effort_garbage_raises_bad_request(effort): def test_reasoning_effort_xhigh_max_maps_to_budget_on_budget_model( effort, expected_budget ): - """``xhigh`` / ``max`` extend the legacy ``thinking.budget_tokens`` - progression (low=1024 / medium=2048 / high=4096 → xhigh=8192 / max=16384) - on budget-mode Claude models (haiku / 4.5 series). - - Adopted from #27051. Keeps the OpenAI-format ``reasoning_effort`` knob - usable across the full Claude lineup — adaptive models (4.6/4.7) route - these tiers via ``output_config.effort``; budget-mode models use the - extended budget. Anthropic's "max only on Mythos / Opus 4.7 / Opus 4.6 / - Sonnet 4.6" gating applies to the *adaptive enum*, not to the legacy - ``budget_tokens`` knob, which accepts any integer up to ``max_tokens``. - """ + """``xhigh`` / ``max`` extend the budget_tokens progression on budget-mode models.""" config = AnthropicConfig() result = config.map_openai_params( @@ -2595,16 +2540,11 @@ def test_reasoning_effort_xhigh_max_maps_to_budget_on_budget_model( assert result["thinking"]["type"] == "enabled" assert result["thinking"]["budget_tokens"] == expected_budget - # Budget-mode models must NOT carry an ``output_config`` payload — that - # path is exclusively for adaptive (4.6+) models. assert "output_config" not in result def test_output_config_effort_empty_string_raises_bad_request(): - """``output_config={"effort": ""}`` must be rejected with a 400 — the - legacy ``if effort and ...`` short-circuit silently let it pass - through (verified end-to-end on the QA sweep for PR #27039). - """ + """``output_config={"effort": ""}`` is rejected with a 400.""" config = AnthropicConfig() with pytest.raises(litellm.exceptions.BadRequestError, match="Invalid effort"): @@ -2618,10 +2558,7 @@ def test_output_config_effort_empty_string_raises_bad_request(): def test_reasoning_effort_minimal_floors_at_anthropic_provider_minimum(): - """Anthropic Messages API rejects ``budget_tokens < 1024``. ``minimal`` - must floor at the provider minimum so it's a usable tier on direct - Anthropic / Azure AI Anthropic / Vertex AI Anthropic / Bedrock Invoke. - """ + """``minimal`` floors at the Anthropic provider minimum (1024).""" config = AnthropicConfig() result = config.map_openai_params( diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_reasoning_effort_translation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_reasoning_effort_translation.py index 20b5f958396..83716b8c8d3 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_reasoning_effort_translation.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_reasoning_effort_translation.py @@ -1,18 +1,4 @@ -""" -Tests for OpenAI-style ``reasoning_effort`` translation on the Anthropic -/v1/messages route. - -The /v1/messages spec doesn't include ``reasoning_effort`` — without -translation it gets silently dropped at the filter step, leaving every -adaptive tier collapsed to the same behavior on Bedrock Invoke /v1/messages -(and on Anthropic / Azure AI / Vertex AI when callers pass it on the -messages route). - -These tests pin the translation and validation behavior at the shared -``AnthropicMessagesConfig`` level so all four /v1/messages routes -(direct Anthropic, Azure AI, Vertex AI, Bedrock Invoke) inherit the -same mapping. -""" +"""Tests for ``reasoning_effort`` translation on the Anthropic /v1/messages route.""" import pytest @@ -36,12 +22,6 @@ from litellm.llms.anthropic.experimental_pass_through.messages.transformation im def test_reasoning_effort_maps_to_output_config_for_adaptive_model( reasoning_effort, expected_effort ): - """ - For Claude 4.6 / 4.7, ``reasoning_effort`` is mapped to - ``thinking={"type": "adaptive"}`` plus ``output_config.effort=``, - using the same mapping table as the chat completion path so the two - routes can't drift. - """ config = AnthropicMessagesConfig() optional_params = {"max_tokens": 1024, "reasoning_effort": reasoning_effort} @@ -59,7 +39,6 @@ def test_reasoning_effort_maps_to_output_config_for_adaptive_model( def test_reasoning_effort_none_clears_thinking_and_output_config(): - """``reasoning_effort='none'`` opts out of extended thinking entirely.""" config = AnthropicMessagesConfig() optional_params = { "max_tokens": 1024, @@ -82,11 +61,6 @@ def test_reasoning_effort_none_clears_thinking_and_output_config(): def test_reasoning_effort_on_non_adaptive_model_uses_thinking_budget(): - """ - Non-adaptive models (Opus 4.5 / earlier) take ``thinking.budget_tokens`` - rather than ``output_config.effort``. The translation falls back to - the budget mapping in that case. - """ config = AnthropicMessagesConfig() optional_params = {"max_tokens": 1024, "reasoning_effort": "high"} @@ -109,11 +83,6 @@ def test_reasoning_effort_on_non_adaptive_model_uses_thinking_budget(): @pytest.mark.parametrize("bad_effort", ["invalid", "disabled", ""]) def test_invalid_reasoning_effort_raises_400(bad_effort): - """ - Garbage ``reasoning_effort`` values surface as a clean 400 instead of - silently passing through to the provider as an unknown - ``output_config.effort`` (which would 500). - """ config = AnthropicMessagesConfig() optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort} @@ -132,18 +101,12 @@ def test_invalid_reasoning_effort_raises_400(bad_effort): @pytest.mark.parametrize( "model,bad_effort", [ - # ``xhigh`` is Opus-4.7-only on the public Anthropic effort matrix, - # so Opus 4.6 / Sonnet 4.6 must still 400 on it. ("claude-opus-4-6", "xhigh"), ("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh"), ("claude-sonnet-4-6", "xhigh"), ], ) def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort): - """``xhigh`` and ``max`` are gated per-model. The /v1/messages route must - surface a clean 400 client-side instead of forwarding the unsupported - tier and letting the provider 500/400 it. - """ config = AnthropicMessagesConfig() optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort} @@ -163,9 +126,6 @@ def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort @pytest.mark.parametrize( "model", [ - # ``max`` is documented as supported on Claude 4.6 (Opus + Sonnet) - # and Claude 4.7. Verify the /v1/messages route accepts it for - # Sonnet 4.6 variants instead of 400-ing client-side. "claude-sonnet-4-6", "bedrock/invoke/us.anthropic.claude-sonnet-4-6", ], @@ -187,11 +147,6 @@ def test_reasoning_effort_max_accepted_on_sonnet_46_messages(model): def test_explicit_output_config_wins_over_reasoning_effort(): - """ - Explicit native ``output_config.effort`` is never overridden by the - OpenAI alias. Same precedence as - ``_translate_legacy_thinking_for_adaptive_model``. - """ config = AnthropicMessagesConfig() optional_params = { "max_tokens": 1024, @@ -212,7 +167,6 @@ def test_explicit_output_config_wins_over_reasoning_effort(): def test_explicit_thinking_wins_over_reasoning_effort(): - """Explicit native ``thinking`` is never overridden by the alias.""" config = AnthropicMessagesConfig() optional_params = { "max_tokens": 1024, @@ -233,8 +187,6 @@ def test_explicit_thinking_wins_over_reasoning_effort(): def test_reasoning_effort_in_supported_params(): - """``reasoning_effort`` is advertised as a supported messages param so - callers and validation paths can introspect the schema.""" config = AnthropicMessagesConfig() assert "reasoning_effort" in config.get_supported_anthropic_messages_params( "claude-opus-4-7" diff --git a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py index 2504e9d5b89..9dac914ca4d 100644 --- a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py +++ b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py @@ -292,19 +292,10 @@ class TestAzureAnthropicConfig: assert "compact-2026-01-12" in headers["anthropic-beta"] def test_output_config_promoted_from_extra_body(self): - """Anthropic-native ``output_config`` routed via ``extra_body`` (by the - openai-compatible kwarg stuffer in ``litellm/utils.py``) must be - promoted to top-level ``optional_params`` before delegating to - ``AnthropicConfig.transform_request``. Otherwise the value is - silently dropped by the ``extra_body`` pop and validation never - runs. - """ + """Anthropic ``output_config`` routed via ``extra_body`` reaches the request body.""" config = AzureAnthropicConfig() messages = [{"role": "user", "content": "Hello"}] - # Simulate what ``add_provider_specific_params_to_optional_params`` - # produces for ``litellm.completion(model="azure_ai/claude-...", - # output_config={"effort": "low"})``. optional_params = { "max_tokens": 100, "extra_body": {"output_config": {"effort": "low"}}, @@ -313,26 +304,18 @@ class TestAzureAnthropicConfig: headers = {"api-key": "test-key", "anthropic-version": "2023-06-01"} result = config.transform_request( - model="claude-opus-4-6", # supports output_config.effort + model="claude-opus-4-6", messages=messages, optional_params=optional_params, litellm_params=litellm_params, headers=headers, ) - # output_config must reach the request body (not be silently dropped) - assert "output_config" in result assert result["output_config"] == {"effort": "low"} - # extra_body should be stripped on the way out assert "extra_body" not in result def test_invalid_output_config_effort_raises_via_extra_body(self): - """``effort="invalid"`` arriving via ``extra_body`` must still raise - ``BadRequestError`` (matching the native ``anthropic`` provider). - Previously this was silently dropped on Azure, so unsupported - efforts reached the model with default behavior instead of - returning a clean 400. - """ + """Invalid ``effort`` via ``extra_body`` raises BadRequestError.""" import litellm config = AzureAnthropicConfig() @@ -356,9 +339,7 @@ class TestAzureAnthropicConfig: assert "Invalid effort value" in str(exc_info.value) def test_unsupported_effort_xhigh_raises_via_extra_body(self): - """``effort="xhigh"`` on a model that does not support it (e.g. - Sonnet 4.6) must raise ``BadRequestError`` even when arriving via - ``extra_body`` on Azure.""" + """Unsupported ``effort='xhigh'`` via ``extra_body`` raises BadRequestError.""" import litellm config = AzureAnthropicConfig() @@ -382,17 +363,14 @@ class TestAzureAnthropicConfig: assert "xhigh" in str(exc_info.value) def test_extra_body_promotion_does_not_clobber_top_level(self): - """If a key exists at both top-level ``optional_params`` and - inside ``extra_body``, the top-level value wins (``setdefault`` - semantics). This protects against a caller who explicitly sets a - param at top-level and accidentally also has it in extra_body.""" + """Top-level ``optional_params`` wins over duplicates in ``extra_body``.""" config = AzureAnthropicConfig() messages = [{"role": "user", "content": "Hello"}] optional_params = { "max_tokens": 100, - "output_config": {"effort": "low"}, # top-level wins - "extra_body": {"output_config": {"effort": "high"}}, # ignored + "output_config": {"effort": "low"}, + "extra_body": {"output_config": {"effort": "high"}}, } litellm_params = {"api_key": "test-key"} headers = {"api-key": "test-key", "anthropic-version": "2023-06-01"} diff --git a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py index 8689a05e8e6..4495e3f4101 100644 --- a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py @@ -407,18 +407,7 @@ def test_opus_4_5_model_detection(): def test_output_config_forwarded_for_bedrock_chat_invoke_request(): - """ - Bedrock Invoke (chat/completions route) must forward - ``output_config`` for Anthropic adaptive-thinking models. The earlier - behavior stripped it unconditionally, which silently flattened every - adaptive tier (``low``/``medium``/``high``/``xhigh``/``max``) to identical - behavior on the wire. - - The wire QA at https://github.com/BerriAI/litellm/pull/27039 showed - ``thinking.type: adaptive`` was forwarded but ``output_config.effort`` - was always missing, even though direct curls to Anthropic's Bedrock - Invoke endpoint accept it. - """ + """Bedrock Invoke chat path forwards ``output_config`` for adaptive Claude models.""" config = AmazonAnthropicClaudeConfig() messages = [{"role": "user", "content": "test"}] @@ -437,7 +426,6 @@ def test_output_config_forwarded_for_bedrock_chat_invoke_request(): ) assert result.get("output_config") == {"effort": "high"} - # Verify normal params survive assert result["max_tokens"] == 100 diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index 271498887f4..5f2ed3dc00f 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -326,10 +326,7 @@ def test_reasoning_effort_none_omits_thinking_for_anthropic_converse(model): def test_reasoning_effort_sets_output_config_for_adaptive_models_converse( model, effort, expected_effort ): - """Adaptive-thinking Claude 4.6 / 4.7 on Bedrock Converse must carry the - requested tier via ``output_config.effort``. The prior strip silently - flattened every adaptive tier on the wire (verified in the PR #27039 - QA sweep).""" + """Adaptive Claude 4.6 / 4.7 on Bedrock Converse routes the tier via ``output_config.effort``.""" config = AmazonConverseConfig() optional_params = config.map_openai_params( @@ -352,10 +349,7 @@ def test_reasoning_effort_sets_output_config_for_adaptive_models_converse( ], ) def test_output_config_effort_forwarded_into_additional_request_fields(model): - """``output_config`` must ride along inside ``additionalModelRequestFields`` - so the Anthropic-on-Bedrock wire request actually carries the effort - tier. The prior ``inference_params.pop("output_config")`` dropped it - on the floor for every adaptive tier.""" + """``output_config`` rides along inside ``additionalModelRequestFields``.""" config = AmazonConverseConfig() messages = [{"role": "user", "content": "hi"}] @@ -380,10 +374,7 @@ def test_output_config_effort_forwarded_into_additional_request_fields(model): ["disabled", "invalid", ""], ) def test_reasoning_effort_garbage_raises_bad_request_converse(effort): - """Garbage / empty-string reasoning_effort on Bedrock Converse Anthropic - must surface as a clean 400 ``BadRequestError`` instead of 500. The - earlier ``ValueError`` from ``_map_reasoning_effort`` propagated up as - a generic 500 in the proxy and ate the request.""" + """Unmapped reasoning_effort on Bedrock Converse Anthropic raises BadRequestError.""" config = AmazonConverseConfig() with pytest.raises(litellm.exceptions.BadRequestError): @@ -405,13 +396,7 @@ def test_reasoning_effort_garbage_raises_bad_request_converse(effort): ], ) def test_output_config_effort_max_passes_through_on_sonnet_46_variants(model): - """``effort='max'`` is supported on Claude 4.6 (Opus + Sonnet) per - https://platform.claude.com/docs/en/build-with-claude/effort. The earlier - Opus-only allow-list in ``_validate_anthropic_adaptive_effort`` has been - widened to ``_is_claude_4_6_model`` (Opus + Sonnet) plus the - ``supports_max_reasoning_effort`` JSON flag. Verify the param actually - flows through to ``additionalModelRequestFields.output_config.effort`` - for every Bedrock Converse Sonnet 4.6 id variant.""" + """``effort='max'`` flows through for every Bedrock Converse Sonnet 4.6 id.""" config = AmazonConverseConfig() messages = [{"role": "user", "content": "hi"}] @@ -3450,9 +3435,7 @@ def test_transform_request_strips_anthropic_output_config(): def test_converse_drop_params_strips_output_config_for_pre_4_5_anthropic(): - """``drop_params=True`` strips ``output_config`` for pre-4.5 Anthropic - models on Bedrock Converse so a proxy fronting Claude Code at haiku doesn't - force a 400 on every request.""" + """``drop_params=True`` strips unsupported ``output_config`` on Bedrock Converse.""" config = AmazonConverseConfig() messages = [{"role": "user", "content": "hi"}] @@ -3477,7 +3460,7 @@ def test_converse_drop_params_strips_output_config_for_pre_4_5_anthropic(): def test_converse_drop_params_keeps_output_config_for_supporting_anthropic(): - """``drop_params=True`` must not strip on supporting models.""" + """``drop_params=True`` does not strip on models that support ``output_config``.""" config = AmazonConverseConfig() messages = [{"role": "user", "content": "hi"}] diff --git a/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py index 644ff81f6f6..c98f7840343 100644 --- a/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py +++ b/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py @@ -593,16 +593,7 @@ def test_remove_scope_from_cache_control(): def test_bedrock_messages_forwards_output_config(): - """ - ``output_config`` is the adaptive-thinking effort payload for Claude - 4.6 / 4.7 (e.g. ``{"effort": "max"}``). Bedrock Invoke accepts it for - those models — the prior behavior of stripping it silently flattened - every adaptive tier on /v1/messages so ``low`` / ``medium`` / ``high`` / - ``xhigh`` / ``max`` all collapsed to identical thinking with no tier - differentiation. - - Regression coverage for the QA bug listed on PR #27039. - """ + """Bedrock Invoke /v1/messages forwards ``output_config`` for adaptive Claude models.""" from litellm.types.router import GenericLiteLLMParams cfg = AmazonAnthropicClaudeMessagesConfig() @@ -628,12 +619,7 @@ def test_bedrock_messages_forwards_output_config(): def test_bedrock_messages_forwards_output_config_with_output_format(): - """ - When both output_config and output_format are present, output_format is - converted to inline schema (Bedrock Invoke doesn't accept output_format - natively), and output_config is forwarded for Claude 4.6/4.7 adaptive - thinking. - """ + """``output_config`` is forwarded; ``output_format`` is converted to inline schema.""" from litellm.types.router import GenericLiteLLMParams cfg = AmazonAnthropicClaudeMessagesConfig() @@ -663,19 +649,7 @@ def test_bedrock_messages_forwards_output_config_with_output_format(): def test_bedrock_messages_forwards_output_config_for_non_adaptive_model(): - """ - ``output_config`` is forwarded for non-adaptive models too (e.g. haiku). - Bedrock will reject the unsupported key for those models — surfacing the - provider error is the correct behavior, since silently swallowing the - knob would hide caller bugs. - - Restores coverage previously asserted by - ``test_bedrock_messages_strips_output_config`` (renamed to the - ``forwards`` variant on Opus 4.7); the strip path no longer exists, - but the non-adaptive pass-through path needs its own explicit test - so a future regression that silently re-adds the strip can't sneak - through. - """ + """``output_config`` is forwarded for non-adaptive models so the provider's error surfaces.""" from litellm.types.router import GenericLiteLLMParams cfg = AmazonAnthropicClaudeMessagesConfig() @@ -698,14 +672,7 @@ def test_bedrock_messages_forwards_output_config_for_non_adaptive_model(): def test_bedrock_messages_drop_params_strips_output_config_for_pre_4_5(): - """ - ``drop_params=True`` is the operator opt-in for "silently fix up" - behavior. When a proxy fronts Claude Code at a pre-4.5 Anthropic model - (haiku-3, sonnet-3.5, ...) on the /v1/messages route, the client always - sends ``output_config.effort`` and the model rejects it. Stripping under - ``drop_params`` lets those requests succeed; otherwise we forward and - surface the model's 400 as designed. - """ + """``drop_params=True`` strips ``output_config`` for pre-4.5 Anthropic on /v1/messages.""" import litellm from litellm.types.router import GenericLiteLLMParams @@ -733,8 +700,7 @@ def test_bedrock_messages_drop_params_strips_output_config_for_pre_4_5(): def test_bedrock_messages_drop_params_keeps_output_config_for_4_7(): - """``drop_params=True`` must not strip on supporting models — opus-4-7 - accepts effort, so the client's tier knob has to land on the wire.""" + """``drop_params=True`` does not strip on opus-4-7 (supports effort).""" import litellm from litellm.types.router import GenericLiteLLMParams @@ -775,18 +741,7 @@ def test_bedrock_messages_drop_params_keeps_output_config_for_4_7(): def test_bedrock_messages_maps_reasoning_effort_for_adaptive_model( reasoning_effort, expected_effort ): - """ - OpenAI-style ``reasoning_effort`` is mapped to native Anthropic - ``thinking`` + ``output_config.effort`` on the /v1/messages route so - callers can drive adaptive thinking with the same tier vocabulary as - the chat completion path. ``reasoning_effort`` itself is popped — the - /v1/messages spec doesn't define it and Bedrock rejects unknown - top-level fields. - - Closes the QA-sweep gap on PR #27074 where Bedrock Invoke /v1/messages - silently dropped ``reasoning_effort`` and every effort tier collapsed - to the same behavior. - """ + """``reasoning_effort`` maps to ``thinking`` + ``output_config.effort`` on /v1/messages.""" from litellm.types.router import GenericLiteLLMParams cfg = AmazonAnthropicClaudeMessagesConfig() @@ -810,16 +765,7 @@ def test_bedrock_messages_maps_reasoning_effort_for_adaptive_model( def test_bedrock_messages_reasoning_effort_on_non_adaptive_uses_thinking_budget(): - """ - For non-adaptive thinking models (e.g. Opus 4.5), ``reasoning_effort`` - is mapped to ``thinking.type=enabled`` with a budget_tokens value - instead of ``output_config.effort``. ``output_config`` is not set on - these models because they don't accept it. - - Mirrors ``AnthropicConfig._map_reasoning_effort`` behavior for the - non-adaptive branch on Opus 4.5 / earlier Claude 4 models, applied to - the /v1/messages route. - """ + """Non-adaptive models map ``reasoning_effort`` to ``thinking.budget_tokens``.""" from litellm.types.router import GenericLiteLLMParams cfg = AmazonAnthropicClaudeMessagesConfig() @@ -847,11 +793,7 @@ def test_bedrock_messages_reasoning_effort_on_non_adaptive_uses_thinking_budget( def test_bedrock_messages_reasoning_effort_none_clears_thinking(): - """ - ``reasoning_effort='none'`` opts out — both ``thinking`` and - ``output_config`` are cleared so the request goes out without - extended thinking. Mirrors the chat completion path's behavior. - """ + """``reasoning_effort='none'`` clears both ``thinking`` and ``output_config``.""" from litellm.types.router import GenericLiteLLMParams cfg = AmazonAnthropicClaudeMessagesConfig() @@ -877,11 +819,7 @@ def test_bedrock_messages_reasoning_effort_none_clears_thinking(): def test_bedrock_messages_invalid_reasoning_effort_raises_400(): - """ - Garbage ``reasoning_effort`` values (``invalid`` / ``disabled`` / ``""``) - surface as a clean 400 ``AnthropicError`` instead of silently passing - an invalid string through to Bedrock as ``output_config.effort``. - """ + """Garbage ``reasoning_effort`` raises AnthropicError (400).""" from litellm.llms.anthropic.common_utils import AnthropicError from litellm.types.router import GenericLiteLLMParams @@ -903,12 +841,7 @@ def test_bedrock_messages_invalid_reasoning_effort_raises_400(): def test_bedrock_messages_explicit_output_config_wins_over_reasoning_effort(): - """ - Caller-supplied native ``output_config.effort`` wins over the OpenAI - ``reasoning_effort`` knob. Same precedence as - ``_translate_legacy_thinking_for_adaptive_model``: explicit native - Anthropic params are never overridden by the alias. - """ + """Explicit ``output_config.effort`` wins over the ``reasoning_effort`` alias.""" from litellm.types.router import GenericLiteLLMParams cfg = AmazonAnthropicClaudeMessagesConfig() @@ -1006,11 +939,6 @@ def test_bedrock_messages_allowlist_filters_anthropic_only_fields(): ): assert bad not in result, f"{bad} should be stripped by the allowlist" - # ``output_config`` rides along — Bedrock Invoke accepts it for Claude - # 4.6/4.7 adaptive thinking and stripping it silently flattens every - # adaptive tier. (Bedrock will reject it for non-adaptive models, which - # is the correct behavior — surface the model error rather than swallow - # the knob.) assert result.get("output_config") == {"effort": "low"} # Supported fields pass through. assert result["max_tokens"] == 4096 diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py index 0a5ba12f187..d89d09a4e63 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py @@ -499,14 +499,7 @@ def test_vertex_ai_partner_models_anthropic_remove_prompt_caching_scope_beta_hea def test_vertex_ai_anthropic_output_config_effort_only_forwarded(): - """ - Vertex AI Claude 4.6 / 4.7 accept ``output_config.effort`` on direct - ``:rawPredict`` (verified end-to-end against ``us-east5`` for - ``claude-opus-4-6`` and ``global`` for ``claude-opus-4-7``). The earlier - strip silently flattened every adaptive tier on Vertex, so ``low`` / - ``medium`` / ``high`` / ``xhigh`` / ``max`` all produced identical - thinking with no tier differentiation. - """ + """Vertex AI Claude 4.6/4.7 accept ``output_config.effort`` on rawPredict.""" config = VertexAIAnthropicConfig() messages = [{"role": "user", "content": "What is 2+2?"}] @@ -568,15 +561,7 @@ def test_vertex_ai_anthropic_output_config_format_passes_through(): def test_vertex_ai_anthropic_output_config_format_plus_effort_preserved(): - """ - Vertex AI Claude 4.6 / 4.7 accept ``output_config.effort`` on direct - ``:rawPredict`` (verified end-to-end against ``us-east5`` for - ``claude-opus-4-6`` and ``global`` for ``claude-opus-4-7``). Since the - strip was unjustified, ``effort`` must now ride along with ``format``. - - We use a Claude 4.6 model id here because ``_apply_output_config`` only - accepts ``effort`` on adaptive-thinking 4.6/4.7 model ids. - """ + """Both ``format`` and ``effort`` ride along on Vertex Claude 4.6/4.7.""" config = VertexAIAnthropicConfig() messages = [{"role": "user", "content": "Return a person object."}] @@ -625,14 +610,7 @@ def test_vertex_ai_anthropic_output_config_non_dict_dropped(): def test_vertex_ai_anthropic_output_format_and_output_config_effort_preserved(): - """ - Vertex AI Claude 4.6 / 4.7 accept ``output_config.effort`` on direct - ``:rawPredict`` (verified end-to-end). When both ``output_format`` and - ``output_config: {effort}`` are present, both must be forwarded — the - earlier ``effort`` strip caused silent loss of the requested adaptive - thinking tier on Vertex routes (``low``/``medium``/``high``/``xhigh``/``max`` - all collapsed to identical adaptive thinking with no tier differentiation). - """ + """Both ``output_format`` and ``output_config.effort`` are forwarded on Vertex 4.6/4.7.""" config = VertexAIAnthropicConfig() messages = [{"role": "user", "content": "Extract structured data"}] diff --git a/tests/test_litellm/llms/xai/test_xai_chat_transformation.py b/tests/test_litellm/llms/xai/test_xai_chat_transformation.py index a8bbb880c5b..3ae8dfc3c0b 100644 --- a/tests/test_litellm/llms/xai/test_xai_chat_transformation.py +++ b/tests/test_litellm/llms/xai/test_xai_chat_transformation.py @@ -14,10 +14,7 @@ from litellm.types.utils import ( class TestXAIReasoningTokenFolding: - """xAI breaks the OpenAI invariant total = prompt + completion by accounting - reasoning_tokens separately. ``_fold_reasoning_tokens_into_completion`` - re-aligns Usage so downstream consumers see the OpenAI shape (o1/o3 - semantics: completion_tokens includes reasoning).""" + """``_fold_reasoning_tokens_into_completion`` re-aligns xAI Usage to the OpenAI invariant.""" @staticmethod def _make_response( @@ -42,8 +39,7 @@ class TestXAIReasoningTokenFolding: return response def test_should_fold_when_total_explained_by_reasoning_gap(self): - # Real xAI live shape captured 2026-05-04: prompt=14, completion=10, - # total=336, reasoning=312. 14+10+312 == 336. + # xAI live shape: 14 + 10 + 312 == 336. response = self._make_response( prompt_tokens=14, completion_tokens=10, @@ -58,7 +54,6 @@ class TestXAIReasoningTokenFolding: assert usage.total_tokens == usage.prompt_tokens + usage.completion_tokens def test_should_not_fold_when_already_normalised(self): - # OpenAI-normalised shape: completion already includes reasoning. response = self._make_response( prompt_tokens=14, completion_tokens=322, @@ -68,7 +63,6 @@ class TestXAIReasoningTokenFolding: XAIChatConfig._fold_reasoning_tokens_into_completion(response) - # Idempotent — no double-fold. assert response.usage.completion_tokens == 322 def test_should_skip_when_no_reasoning_tokens(self): @@ -84,8 +78,7 @@ class TestXAIReasoningTokenFolding: assert response.usage.completion_tokens == 10 def test_should_skip_when_gap_does_not_match_reasoning(self): - # Defensive guard: if xAI ever changes accounting and the gap stops - # equalling reasoning_tokens, refuse to fold rather than corrupt. + # Refuse to fold if xAI accounting changes (gap != reasoning_tokens). response = self._make_response( prompt_tokens=14, completion_tokens=10, @@ -95,7 +88,6 @@ class TestXAIReasoningTokenFolding: XAIChatConfig._fold_reasoning_tokens_into_completion(response) - # No fold; original values preserved. assert response.usage.completion_tokens == 10 assert response.usage.total_tokens == 999 diff --git a/tests/test_litellm/llms/xai/test_xai_cost_calculator.py b/tests/test_litellm/llms/xai/test_xai_cost_calculator.py index a8aae346224..02fe7c8e68f 100644 --- a/tests/test_litellm/llms/xai/test_xai_cost_calculator.py +++ b/tests/test_litellm/llms/xai/test_xai_cost_calculator.py @@ -107,7 +107,6 @@ class TestXAICostCalculator: def test_grok_4_cost_calculation(self): """Test cost calculation for grok-4 model.""" - # xAI raw API shape: total_tokens = prompt + visible completion + reasoning usage = Usage( prompt_tokens=10, completion_tokens=200, @@ -134,7 +133,6 @@ class TestXAICostCalculator: def test_grok_3_fast_beta_cost_calculation(self): """Test cost calculation for grok-3-fast-beta model.""" - # xAI raw API shape: total_tokens = prompt + visible completion + reasoning usage = Usage( prompt_tokens=20, completion_tokens=300, @@ -176,7 +174,6 @@ class TestXAICostCalculator: def test_edge_case_large_reasoning_tokens(self): """Test cost calculation when reasoning_tokens is larger than completion_tokens.""" - # xAI raw API shape: total_tokens = prompt + visible completion + reasoning usage = Usage( prompt_tokens=12, completion_tokens=50, # Less than reasoning_tokens @@ -204,7 +201,6 @@ class TestXAICostCalculator: def test_tiered_pricing_above_128k_tokens(self): """Test tiered pricing for tokens above 128k.""" # Test with grok-4-fast-reasoning which has tiered pricing - # xAI raw API shape: total_tokens = prompt + visible completion + reasoning usage = Usage( prompt_tokens=150000, # Above 128k threshold completion_tokens=100000, # Above 128k threshold @@ -234,7 +230,6 @@ class TestXAICostCalculator: def test_tiered_pricing_below_128k_tokens(self): """Test that regular pricing is used for tokens below 128k threshold.""" # Test with grok-4-fast-reasoning which has tiered pricing - # xAI raw API shape: total_tokens = prompt + visible completion + reasoning usage = Usage( prompt_tokens=100000, # Below 128k threshold completion_tokens=50000, @@ -263,7 +258,6 @@ class TestXAICostCalculator: def test_tiered_pricing_grok_4_latest(self): """Test tiered pricing for grok-4-latest model.""" - # xAI raw API shape: total_tokens = prompt + visible completion + reasoning usage = Usage( prompt_tokens=200000, # Above 128k threshold completion_tokens=100000, @@ -292,7 +286,6 @@ class TestXAICostCalculator: def test_tiered_pricing_output_tokens_below_128k(self): """Test that output tokens get tiered rate when input tokens > 128k, even if output tokens < 128k.""" - # xAI raw API shape: total_tokens = prompt + visible completion + reasoning usage = Usage( prompt_tokens=150000, # Above 128k threshold completion_tokens=50000, # Below 128k threshold @@ -339,16 +332,7 @@ class TestXAICostCalculator: assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10) def test_already_normalised_usage_does_not_double_count_reasoning(self): - """Cost calc receives Usage post-transformation (OpenAI invariant). - - After XAIChatConfig.transform_response folds reasoning_tokens into - completion_tokens, the Usage block satisfies - ``total_tokens == prompt_tokens + completion_tokens``. Cost calc must - detect this and skip the reasoning_tokens add-on, otherwise it - double-bills the reasoning tokens. - """ - # OpenAI-normalised shape: completion_tokens already includes the - # 100 reasoning tokens (so 100 visible + 100 reasoning -> 200). + """Cost calc must not double-bill when Usage is already OpenAI-normalised.""" usage = Usage( prompt_tokens=12, completion_tokens=200, @@ -364,7 +348,6 @@ class TestXAICostCalculator: prompt_cost, completion_cost = cost_per_token(model="grok-3-mini", usage=usage) - # Bill exactly what completion_tokens reports — no double-add. expected_prompt_cost = 12 * 3e-7 expected_completion_cost = 200 * 5e-7 diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 7e840cc7ad0..65305e1a81e 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -2905,9 +2905,9 @@ def test_gemini_embedding_2_ga_in_cost_map(): assert info.get("input_cost_per_audio_per_second") == 0.00016 assert info.get("input_cost_per_video_per_second") == 0.00079 if provider in ("vertex_ai-embedding-models", "vertex_ai"): - assert info.get("uses_embed_content") is True, ( - f"{key} must have uses_embed_content=true for correct Vertex AI routing" - ) + assert ( + info.get("uses_embed_content") is True + ), f"{key} must have uses_embed_content=true for correct Vertex AI routing" def test_gemini_lyria_3_preview_models_in_cost_map():