mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
refactor: remove unnecessary comments from #27074
Strip out the explanatory and historical comments that don't carry business-logic justification. Comments that simply narrate what code does — or that explain prior behavior, what was changed, or which PR introduced a fix — are removed. Docstrings are reduced to a one-line summary where the long form repeated information already evident from the code or test data. No code-behavior changes. All 643 affected unit tests still pass. Co-authored-by: Mateo Wang <mateo-berri@users.noreply.github.com>
This commit is contained in:
parent
56070b86a3
commit
2cb3f0f027
28 changed files with 92 additions and 739 deletions
|
|
@ -202,17 +202,6 @@ DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET = int(
|
|||
DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET = int(
|
||||
os.getenv("DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET", 4096)
|
||||
)
|
||||
# ``xhigh`` / ``max`` budget extrapolation for legacy ``thinking.budget_tokens``
|
||||
# models (Claude 4.5 series + haiku). Continues the 2× progression
|
||||
# 1024 → 2048 → 4096 from the existing low/medium/high tiers. These tiers
|
||||
# also exist as adaptive ``output_config.effort`` enum values on Claude 4.6+
|
||||
# / 4.7; this constant only governs the budget-tokens fallback for models
|
||||
# that aren't on the adaptive path. Per
|
||||
# https://platform.claude.com/docs/en/build-with-claude/effort the ``effort``
|
||||
# enum is gated by model, but the legacy ``budget_tokens`` knob accepts any
|
||||
# integer up to the model's max_tokens — adopting #27051's mapping here lets
|
||||
# ``reasoning_effort=xhigh|max`` Just Work as a unified OpenAI-format knob
|
||||
# regardless of which Anthropic API surface implements it.
|
||||
DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET = int(
|
||||
os.getenv("DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET", 8192)
|
||||
)
|
||||
|
|
@ -416,14 +405,7 @@ BEDROCK_MAX_POLICY_SIZE = int(os.getenv("BEDROCK_MAX_POLICY_SIZE", 75))
|
|||
BEDROCK_MIN_THINKING_BUDGET_TOKENS = int(
|
||||
os.getenv("BEDROCK_MIN_THINKING_BUDGET_TOKENS", 1024)
|
||||
)
|
||||
# Anthropic's Messages API rejects ``thinking.budget_tokens < 1024`` with a
|
||||
# 400. ``reasoning_effort='minimal'`` historically mapped to 128 (the global
|
||||
# default) which always 400'd against direct Anthropic, Azure AI Anthropic,
|
||||
# Vertex AI Anthropic, and Bedrock Invoke. Floor at the provider minimum so
|
||||
# ``minimal`` is a usable tier on every Anthropic-backed route; Bedrock
|
||||
# Converse already clamps server-side, this just unifies the behavior.
|
||||
# Constant — not env-overridable — because it tracks Anthropic's published
|
||||
# wire-protocol minimum, not a tunable.
|
||||
# Anthropic's Messages API rejects thinking.budget_tokens < 1024.
|
||||
ANTHROPIC_MIN_THINKING_BUDGET_TOKENS = 1024
|
||||
REPLICATE_POLLING_DELAY_SECONDS = float(
|
||||
os.getenv("REPLICATE_POLLING_DELAY_SECONDS", 0.5)
|
||||
|
|
|
|||
|
|
@ -224,11 +224,8 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
def _supports_effort_level(model: str, level: str) -> bool:
|
||||
"""Check ``supports_{level}_reasoning_effort`` in the model map.
|
||||
|
||||
Mirrors the pattern used in ``openai/chat/gpt_5_transformation.py`` so
|
||||
that adding support for a new effort level is a pure model-map change.
|
||||
Handles bedrock-prefixed and vertex-prefixed model ids by stripping
|
||||
the prefix and re-checking against ``litellm.model_cost`` directly,
|
||||
so a Bedrock-routed Claude 4.6/4.7 keeps its model-map flag.
|
||||
Strips bedrock/vertex prefixes so a provider-routed Claude still
|
||||
resolves to the Anthropic model-map entry.
|
||||
"""
|
||||
key = f"supports_{level}_reasoning_effort"
|
||||
try:
|
||||
|
|
@ -240,11 +237,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
return True
|
||||
except Exception:
|
||||
pass
|
||||
# Bedrock and Vertex route the model id with a provider-prefix
|
||||
# (e.g. ``bedrock/invoke/us.anthropic.claude-opus-4-7``). Strip
|
||||
# known prefixes and look the resulting Anthropic-flavoured key
|
||||
# up directly in ``litellm.model_cost`` so the lookup keeps
|
||||
# working regardless of which route the request arrived on.
|
||||
candidates = [model]
|
||||
for prefix in (
|
||||
"bedrock/converse/",
|
||||
|
|
@ -277,27 +269,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
|
||||
@staticmethod
|
||||
def _validate_effort_for_model(model: str, effort: Optional[str]) -> Optional[str]:
|
||||
"""Return ``None`` if ``effort`` is allowed on ``model``, else an error message.
|
||||
|
||||
Centralises per-model gating for ``max`` and ``xhigh`` so the chat
|
||||
completion path (``_apply_output_config``) and the /v1/messages
|
||||
pass-through (``AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic``)
|
||||
can't drift when a new model tier is added. Caller raises the
|
||||
provider-appropriate exception type using the returned message.
|
||||
|
||||
``max`` is supported on Claude 4.6 (Opus + Sonnet) and Claude 4.7
|
||||
adaptive-thinking models per
|
||||
https://platform.claude.com/docs/en/build-with-claude/effort. The
|
||||
data-driven ``supports_max_reasoning_effort`` flag in
|
||||
``model_prices_and_context_window.json`` is the source of truth;
|
||||
family-level ``_is_claude_4_6_model`` / ``_is_claude_4_7_model``
|
||||
checks remain as a fallback for OpenRouter / GitHub Copilot /
|
||||
Vercel / Bedrock variants whose model-map entries don't yet carry
|
||||
the flag.
|
||||
|
||||
``xhigh`` is purely data-driven via ``supports_xhigh_reasoning_effort``
|
||||
so enabling it for a new model is a model-map-only change.
|
||||
"""
|
||||
"""Return ``None`` if ``effort`` is allowed on ``model``, else an error message."""
|
||||
if effort == "max" and not (
|
||||
AnthropicConfig._is_claude_4_6_model(model)
|
||||
or AnthropicConfig._is_claude_4_7_model(model)
|
||||
|
|
@ -312,15 +284,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
|
||||
@staticmethod
|
||||
def _model_supports_effort_param(model: str) -> bool:
|
||||
"""Whether the model accepts ``output_config.effort`` at all.
|
||||
|
||||
Per https://platform.claude.com/docs/en/build-with-claude/effort the
|
||||
``output_config.effort`` parameter is supported on Opus 4.5+, Sonnet 4.6+
|
||||
and Mythos Preview; older Claude models reject it with a 400. Support is
|
||||
encoded in ``model_prices_and_context_window.json`` via the
|
||||
``supports_*_reasoning_effort`` flags, so adding a new effort-capable
|
||||
model is a pure model-map change.
|
||||
"""
|
||||
"""Whether the model accepts ``output_config.effort`` at all."""
|
||||
for level in ("low", "minimal", "medium", "high", "xhigh", "max"):
|
||||
if AnthropicConfig._supports_effort_level(model, level):
|
||||
return True
|
||||
|
|
@ -908,16 +872,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
model: str,
|
||||
llm_provider: str = "anthropic",
|
||||
) -> Optional[AnthropicThinkingParam]:
|
||||
"""Map an OpenAI-format ``reasoning_effort`` string to Anthropic's
|
||||
``thinking`` payload.
|
||||
|
||||
Raises ``BadRequestError`` (clean 400) instead of ``ValueError`` (500)
|
||||
on unmapped efforts so every caller — Anthropic native, Bedrock
|
||||
Invoke/Converse, Databricks, Vertex Anthropic, Azure AI Anthropic,
|
||||
and the experimental ``/v1/messages`` pass-through — surfaces a
|
||||
consistent error to the user. Pass ``llm_provider`` so the
|
||||
``BadRequestError`` carries the right provider name in logs.
|
||||
"""
|
||||
if reasoning_effort is None or reasoning_effort == "none":
|
||||
return None
|
||||
if AnthropicConfig._is_adaptive_thinking_model(model):
|
||||
|
|
@ -940,32 +894,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
budget_tokens=DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET,
|
||||
)
|
||||
elif reasoning_effort == "xhigh":
|
||||
# Continues the 2× progression of low/medium/high (1024/2048/4096).
|
||||
# On adaptive models (Claude 4.6/4.7) the ``xhigh`` tier is
|
||||
# already routed via ``output_config.effort=xhigh`` above; this
|
||||
# branch only applies to budget-mode models (Claude 4.5 series +
|
||||
# haiku) where the OpenAI-format ``reasoning_effort`` knob would
|
||||
# otherwise 400 with ``Unmapped reasoning effort``. Keeps the
|
||||
# cross-model UX uniform — ``reasoning_effort=xhigh`` Just Works
|
||||
# regardless of which Anthropic API surface implements it.
|
||||
return AnthropicThinkingParam(
|
||||
type="enabled",
|
||||
budget_tokens=DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
|
||||
)
|
||||
elif reasoning_effort == "max":
|
||||
# Same rationale as ``xhigh`` above — ``max`` is the adaptive
|
||||
# enum's top tier on Claude 4.6/4.7, but for budget-mode models
|
||||
# we extend the 2× progression (8192 → 16384) so the OpenAI-
|
||||
# format alias is usable on every Claude model.
|
||||
return AnthropicThinkingParam(
|
||||
type="enabled",
|
||||
budget_tokens=DEFAULT_REASONING_EFFORT_MAX_THINKING_BUDGET,
|
||||
)
|
||||
elif reasoning_effort == "minimal":
|
||||
# Anthropic Messages API rejects ``budget_tokens < 1024`` with a
|
||||
# 400. Floor at the provider minimum so ``minimal`` is a usable
|
||||
# tier on Anthropic / Azure AI Anthropic / Vertex AI Anthropic /
|
||||
# Bedrock Invoke. Bedrock Converse already clamps server-side.
|
||||
return AnthropicThinkingParam(
|
||||
type="enabled",
|
||||
budget_tokens=max(
|
||||
|
|
@ -1246,10 +1184,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
elif param == "thinking":
|
||||
optional_params["thinking"] = value
|
||||
elif param == "reasoning_effort" and isinstance(value, str):
|
||||
# ``_map_reasoning_effort`` raises ``BadRequestError`` (400)
|
||||
# directly on unmapped efforts (``disabled`` / ``invalid`` /
|
||||
# ``""`` / ``xhigh``/``max`` on budget-mode Claude 4.5) so
|
||||
# we no longer need to wrap a ``ValueError`` here.
|
||||
mapped_thinking = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort=value,
|
||||
model=model,
|
||||
|
|
@ -1260,20 +1194,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
optional_params.pop("output_config", None)
|
||||
else:
|
||||
optional_params["thinking"] = mapped_thinking
|
||||
# For Claude 4.6+ adaptive-thinking models, effort is
|
||||
# controlled via ``output_config``, not
|
||||
# ``thinking.budget_tokens``. Driven by
|
||||
# ``supports_adaptive_thinking`` in the model map so
|
||||
# adding a new adaptive Claude is a model-map-only change.
|
||||
if AnthropicConfig._is_adaptive_thinking_model(model):
|
||||
# ``_map_reasoning_effort`` returns ``type=adaptive``
|
||||
# for any string on adaptive models without checking
|
||||
# the value, so reject unmapped efforts here (matching
|
||||
# the /v1/messages path) instead of relying on the
|
||||
# downstream ``_apply_output_config`` check. Co-locating
|
||||
# validation with the mapping prevents garbage from
|
||||
# leaking into ``optional_params`` if ``map_openai_params``
|
||||
# is ever called without a subsequent ``transform_request``.
|
||||
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
|
||||
value
|
||||
)
|
||||
|
|
@ -1703,22 +1624,12 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
def _apply_output_config(
|
||||
self, data: dict, model: str, optional_params: dict
|
||||
) -> None:
|
||||
"""Validate and apply output_config to the request data.
|
||||
|
||||
Validation errors raise ``BadRequestError`` (clean 400) so callers
|
||||
passing ``effort="disabled"`` / ``effort=""`` / unsupported tiers
|
||||
for the model see a client-side error rather than a 500.
|
||||
"""
|
||||
"""Validate and apply output_config to the request data."""
|
||||
if "output_config" not in optional_params:
|
||||
return
|
||||
output_config = optional_params.get("output_config")
|
||||
if not output_config or not isinstance(output_config, dict):
|
||||
return
|
||||
# When ``drop_params`` is set, strip ``output_config`` for models that
|
||||
# cannot accept it (e.g. proxy fronting Claude Code at haiku-3, where
|
||||
# the client always sends effort but the model rejects it). The user
|
||||
# opted into silent fixup via the global flag — log a warning so the
|
||||
# strip is still visible in logs.
|
||||
if litellm.drop_params is True and not self._model_supports_effort_param(model):
|
||||
litellm.verbose_logger.warning(
|
||||
"Dropping unsupported `output_config` for model=%s "
|
||||
|
|
@ -1730,10 +1641,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
data.pop("output_config", None)
|
||||
return
|
||||
effort = output_config.get("effort")
|
||||
# ``effort=""`` (empty string) and unmapped strings should be treated
|
||||
# as invalid, not silently passed through. We use ``effort is not None``
|
||||
# here so empty string fails the membership check below. (The legacy
|
||||
# ``if effort and ...`` short-circuit silently accepted ``""``.)
|
||||
valid_efforts = ["high", "medium", "low", "xhigh", "max"]
|
||||
if effort is not None and effort not in valid_efforts:
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
|
|
@ -1744,10 +1651,6 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
model=model,
|
||||
llm_provider=self.custom_llm_provider or "anthropic",
|
||||
)
|
||||
# Per-model gating for ``max`` / ``xhigh`` is centralised in
|
||||
# ``_validate_effort_for_model`` so the chat path and the
|
||||
# /v1/messages pass-through stay in lock-step when a new model
|
||||
# tier lands.
|
||||
gate_error = self._validate_effort_for_model(model, effort)
|
||||
if gate_error is not None:
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
|
|
|
|||
|
|
@ -273,15 +273,7 @@ class AnthropicModelInfo(BaseLLMModelInfo):
|
|||
|
||||
@staticmethod
|
||||
def _is_adaptive_thinking_model(model: str) -> bool:
|
||||
"""Claude 4.6+ models use adaptive thinking with ``output_config.effort``.
|
||||
|
||||
Driven by the ``supports_adaptive_thinking`` flag in
|
||||
``model_prices_and_context_window.json`` so that adding a new
|
||||
adaptive-thinking model is a pure model-map change. Falls back to
|
||||
the family-pattern check for OpenRouter / Vercel / Bedrock /
|
||||
provider-prefixed variants whose model-map entries don't (yet)
|
||||
carry the flag.
|
||||
"""
|
||||
"""Claude 4.6+ models use adaptive thinking with ``output_config.effort``."""
|
||||
from litellm.utils import _supports_factory
|
||||
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -47,9 +47,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
"inference_geo",
|
||||
"speed",
|
||||
"output_config",
|
||||
# OpenAI-style tier knob — translated to native ``thinking`` +
|
||||
# ``output_config`` in ``transform_anthropic_messages_request``
|
||||
# and popped before the request is forwarded.
|
||||
"reasoning_effort",
|
||||
# TODO: Add Anthropic `metadata` support
|
||||
# "metadata",
|
||||
|
|
@ -176,22 +173,10 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
) -> None:
|
||||
"""Map OpenAI-style ``reasoning_effort`` to native Anthropic params.
|
||||
|
||||
The /v1/messages spec doesn't include ``reasoning_effort`` — without
|
||||
this translation it gets silently dropped, leaving every adaptive
|
||||
tier collapsed to the same behavior on Bedrock Invoke /v1/messages
|
||||
(and on Anthropic / Azure AI / Vertex AI when callers pass it on
|
||||
the messages route). Mirrors ``AnthropicConfig.map_openai_params``
|
||||
on the chat completion path so the two routes can't drift.
|
||||
|
||||
- Pops ``reasoning_effort`` from ``optional_params`` so it never
|
||||
reaches the wire.
|
||||
- Caller-supplied ``thinking`` / ``output_config`` always win — we
|
||||
don't override an explicit native value.
|
||||
- Effort=``none`` clears thinking + output_config so callers can
|
||||
opt out per request.
|
||||
- Invalid efforts raise ``BadRequestError`` (clean 400) instead of
|
||||
surfacing as 500s downstream.
|
||||
Caller-supplied ``thinking`` / ``output_config`` win over the alias.
|
||||
``effort='none'`` clears both. Invalid efforts raise a 400.
|
||||
"""
|
||||
from litellm.exceptions import BadRequestError as _BadRequestError
|
||||
from litellm.llms.anthropic.chat.transformation import (
|
||||
REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT,
|
||||
AnthropicConfig,
|
||||
|
|
@ -201,12 +186,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
if not isinstance(reasoning_effort, str):
|
||||
return
|
||||
|
||||
# ``_map_reasoning_effort`` raises ``BadRequestError`` (400) directly
|
||||
# on unmapped efforts. The /v1/messages pass-through surfaces errors
|
||||
# as ``AnthropicError``; convert here so callers see a provider-shaped
|
||||
# 400 rather than the LiteLLM-shaped one.
|
||||
from litellm.exceptions import BadRequestError as _BadRequestError
|
||||
|
||||
try:
|
||||
mapped_thinking = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort=reasoning_effort, model=model
|
||||
|
|
@ -224,12 +203,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
|
||||
reasoning_effort
|
||||
)
|
||||
# ``_map_reasoning_effort`` returns ``type=adaptive`` for any
|
||||
# string on adaptive models without checking the value. The
|
||||
# chat completion path validates the resolved effort downstream
|
||||
# via ``_apply_output_config``; /v1/messages has no equivalent
|
||||
# downstream check, so reject unmapped values here so callers
|
||||
# see a clean 400 instead of a 500 from the provider.
|
||||
if mapped_effort is None:
|
||||
raise AnthropicError(
|
||||
message=(
|
||||
|
|
@ -239,10 +212,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
),
|
||||
status_code=400,
|
||||
)
|
||||
# Per-model gating for ``max`` / ``xhigh`` is centralised in
|
||||
# ``AnthropicConfig._validate_effort_for_model`` so the chat
|
||||
# completion path and this /v1/messages pass-through stay in
|
||||
# lock-step when a new model tier lands.
|
||||
gate_error = AnthropicConfig._validate_effort_for_model(
|
||||
model, mapped_effort
|
||||
)
|
||||
|
|
|
|||
|
|
@ -15,21 +15,11 @@ if TYPE_CHECKING:
|
|||
def _promote_extra_body_to_optional_params(optional_params: dict) -> None:
|
||||
"""Promote anthropic-native passthrough keys out of ``extra_body``.
|
||||
|
||||
``azure_ai`` is registered in ``litellm.openai_compatible_providers``, so
|
||||
``add_provider_specific_params_to_optional_params`` (litellm/utils.py)
|
||||
auto-stuffs any non-OpenAI kwarg (e.g. ``output_config={"effort": "..."}``)
|
||||
into ``optional_params["extra_body"]``. For the Azure→Anthropic route the
|
||||
user's intent is to forward those params to Anthropic, so promote them to
|
||||
the top level of ``optional_params`` so:
|
||||
|
||||
* ``AnthropicConfig._apply_output_config`` validates ``effort`` values
|
||||
(matching the native ``anthropic`` provider's 400 on bad efforts).
|
||||
* Valid passthroughs (e.g. ``output_config``, ``thinking``) actually
|
||||
reach the request body instead of being silently dropped by the
|
||||
``data.pop("extra_body", None)`` strip below.
|
||||
|
||||
``setdefault`` is used so an explicit top-level value is never clobbered
|
||||
by a duplicate inside ``extra_body``.
|
||||
``azure_ai`` is an OpenAI-compatible provider, so non-OpenAI kwargs like
|
||||
``output_config`` get auto-routed into ``extra_body`` by
|
||||
``add_provider_specific_params_to_optional_params``. For the Azure→Anthropic
|
||||
route those keys must reach the request body and be validated, so promote
|
||||
them. ``setdefault`` keeps explicit top-level values authoritative.
|
||||
"""
|
||||
extra_body = optional_params.get("extra_body")
|
||||
if not isinstance(extra_body, dict) or not extra_body:
|
||||
|
|
@ -66,9 +56,6 @@ class AzureAnthropicConfig(AnthropicConfig):
|
|||
1. API key via 'api-key' header
|
||||
2. Azure AD token via 'Authorization: Bearer <token>' header
|
||||
"""
|
||||
# Promote anthropic-native passthrough keys (``output_config``,
|
||||
# ``thinking``, ``mcp_servers``, ...) out of ``extra_body`` so the
|
||||
# flag detection below (``is_mcp_server_used``, etc.) sees them.
|
||||
_promote_extra_body_to_optional_params(optional_params)
|
||||
|
||||
# Convert dict to GenericLiteLLMParams if needed
|
||||
|
|
@ -133,18 +120,8 @@ class AzureAnthropicConfig(AnthropicConfig):
|
|||
Transform request using parent AnthropicConfig, then remove unsupported params.
|
||||
Azure Anthropic doesn't support extra_body, max_retries, or stream_options parameters.
|
||||
"""
|
||||
# Promote anthropic-native passthrough keys (``output_config``,
|
||||
# ``thinking``, ...) out of ``extra_body`` BEFORE delegating to
|
||||
# ``AnthropicConfig.transform_request``. Without this:
|
||||
# * ``output_config`` is silently dropped by the ``extra_body`` pop
|
||||
# below, and
|
||||
# * ``_apply_output_config`` never validates ``effort`` values, so
|
||||
# ``effort="invalid"`` quietly reaches the model with default
|
||||
# behavior instead of returning a clean 400 (as the native
|
||||
# ``anthropic`` provider does).
|
||||
_promote_extra_body_to_optional_params(optional_params)
|
||||
|
||||
# Call parent transform_request
|
||||
data = super().transform_request(
|
||||
model=model,
|
||||
messages=messages,
|
||||
|
|
|
|||
|
|
@ -84,12 +84,6 @@ class BaseConfig(ABC):
|
|||
|
||||
@classmethod
|
||||
def get_config(cls):
|
||||
# Subclasses lean on this to surface their public default settings
|
||||
# (e.g. ``max_tokens``) as request params. Anything ``_``-prefixed is
|
||||
# treated as private (lookup tables, ABC machinery, internal flags)
|
||||
# and must not leak into the wire body — a tuple/dict/frozenset class
|
||||
# attribute would otherwise serialise into the request as an extra
|
||||
# top-level key and the provider would 400 it.
|
||||
return {
|
||||
k: v
|
||||
for k, v in cls.__dict__.items()
|
||||
|
|
|
|||
|
|
@ -189,8 +189,6 @@ class AmazonConverseConfig(BaseConfig):
|
|||
|
||||
@classmethod
|
||||
def get_config(cls):
|
||||
# ``_``-prefixed names are private (lookup tables, ABC machinery,
|
||||
# internal flags) and must not leak into the wire body.
|
||||
return {
|
||||
k: v
|
||||
for k, v in cls.__dict__.items()
|
||||
|
|
@ -415,59 +413,19 @@ class AmazonConverseConfig(BaseConfig):
|
|||
"""
|
||||
Handle the reasoning_effort parameter based on the model type.
|
||||
|
||||
Different model families handle reasoning effort differently:
|
||||
- GPT-OSS models: Keep reasoning_effort as-is (passed to additionalModelRequestFields)
|
||||
- Nova 2 models: Transform to reasoningConfig structure
|
||||
- Other models (Anthropic, etc.): Convert to thinking parameter
|
||||
|
||||
For Claude 4.6 / 4.7 (adaptive thinking) the tier is carried via
|
||||
``output_config.effort`` rather than ``thinking.budget_tokens``. We
|
||||
validate the effort with the same rules ``AnthropicConfig._apply_output_config``
|
||||
uses (low/medium/high/xhigh/max + per-model gating) and stage the
|
||||
validated dict on ``optional_params["output_config"]`` so it rides
|
||||
along to ``additionalModelRequestFields`` on the Anthropic-on-Bedrock
|
||||
wire path. Without this the silent strip in ``_prepare_request_params``
|
||||
collapsed every adaptive tier to identical behavior.
|
||||
|
||||
Args:
|
||||
model: The model identifier
|
||||
reasoning_effort: The reasoning effort value
|
||||
optional_params: Dictionary of optional parameters to update in-place
|
||||
|
||||
Examples:
|
||||
>>> config = AmazonConverseConfig()
|
||||
>>> params = {}
|
||||
>>> config._handle_reasoning_effort_parameter("gpt-oss-model", "high", params)
|
||||
>>> params
|
||||
{'reasoning_effort': 'high'}
|
||||
|
||||
>>> params = {}
|
||||
>>> config._handle_reasoning_effort_parameter("amazon.nova-2-lite-v1:0", "high", params)
|
||||
>>> params
|
||||
{'reasoningConfig': {'type': 'enabled', 'maxReasoningEffort': 'high'}}
|
||||
|
||||
>>> params = {}
|
||||
>>> config._handle_reasoning_effort_parameter("anthropic.claude-3", "high", params)
|
||||
>>> params
|
||||
{'thinking': {'type': 'enabled', 'budget_tokens': 10000}}
|
||||
- GPT-OSS models: passed through unchanged via additionalModelRequestFields.
|
||||
- Nova 2 models: transformed to reasoningConfig.
|
||||
- Anthropic models: mapped to ``thinking`` (and ``output_config.effort`` on
|
||||
adaptive Claude 4.6 / 4.7).
|
||||
"""
|
||||
if "gpt-oss" in model:
|
||||
# GPT-OSS models: keep reasoning_effort as-is
|
||||
# It will be passed through to additionalModelRequestFields
|
||||
optional_params["reasoning_effort"] = reasoning_effort
|
||||
elif self._is_nova_2_model(model):
|
||||
# Nova 2 models: transform to reasoningConfig
|
||||
reasoning_config = self._transform_reasoning_effort_to_reasoning_config(
|
||||
reasoning_effort
|
||||
)
|
||||
optional_params.update(reasoning_config)
|
||||
else:
|
||||
# Anthropic and other models: convert to thinking parameter.
|
||||
# ``_map_reasoning_effort`` raises ``BadRequestError`` (400)
|
||||
# directly on unmapped efforts (``disabled`` / ``invalid`` /
|
||||
# ``""`` / ``xhigh``/``max`` on budget-mode Claude 4.5); pass
|
||||
# ``llm_provider="bedrock_converse"`` so the error carries the
|
||||
# right provider name.
|
||||
mapped_thinking = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort=reasoning_effort,
|
||||
model=model,
|
||||
|
|
@ -478,22 +436,7 @@ class AmazonConverseConfig(BaseConfig):
|
|||
optional_params.pop("output_config", None)
|
||||
else:
|
||||
optional_params["thinking"] = mapped_thinking
|
||||
# Adaptive-thinking models (Claude 4.6 / 4.7+) take the
|
||||
# tier via ``output_config.effort``. Mirror the mapping
|
||||
# used by ``AnthropicConfig.map_openai_params`` and apply
|
||||
# the same validation rules so unmapped/garbage efforts
|
||||
# surface as a 400 instead of being silently flattened on
|
||||
# the wire. Driven by ``supports_adaptive_thinking`` in
|
||||
# ``model_prices_and_context_window.json`` so a future
|
||||
# adaptive Claude release lands as a model-map change.
|
||||
if AnthropicConfig._is_adaptive_thinking_model(model):
|
||||
# Use ``.get()`` without a fallback so unmapped efforts
|
||||
# (e.g. ``"disabled"``) surface as a clean 400 here
|
||||
# rather than leaking the raw garbage string through to
|
||||
# ``_validate_anthropic_adaptive_effort`` (which does
|
||||
# catch it, but only because validation happens to run).
|
||||
# Matches the /v1/messages pattern where validation is
|
||||
# co-located with the mapping.
|
||||
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
|
||||
reasoning_effort
|
||||
)
|
||||
|
|
@ -514,16 +457,7 @@ class AmazonConverseConfig(BaseConfig):
|
|||
|
||||
@staticmethod
|
||||
def _validate_anthropic_adaptive_effort(model: str, effort: str) -> None:
|
||||
"""Validate ``output_config.effort`` for adaptive-thinking Claude 4.6/4.7
|
||||
on Bedrock. Raises ``BadRequestError`` (clean 400) instead of letting
|
||||
a downstream ``ValueError`` surface as 500.
|
||||
|
||||
Per-model gating for ``max``/``xhigh`` is delegated to
|
||||
``AnthropicConfig._validate_effort_for_model`` so the Bedrock Converse
|
||||
path and the Anthropic chat / ``/v1/messages`` paths can't drift when
|
||||
a new gated effort tier is added. ``_supports_effort_level`` on
|
||||
``AnthropicConfig`` already handles Bedrock-prefixed model ids.
|
||||
"""
|
||||
"""Validate ``output_config.effort`` for adaptive-thinking Claude 4.6/4.7."""
|
||||
valid_efforts = {"high", "medium", "low", "xhigh", "max"}
|
||||
if effort not in valid_efforts:
|
||||
raise litellm.exceptions.BadRequestError(
|
||||
|
|
@ -1283,13 +1217,9 @@ class AmazonConverseConfig(BaseConfig):
|
|||
)
|
||||
inference_params.pop("json_mode", None) # used for handling json_schema
|
||||
|
||||
# Anthropic-only ``output_config`` (snake_case) is the adaptive-
|
||||
# thinking effort payload (e.g. ``{"effort": "max"}``) for Claude
|
||||
# 4.6/4.7. On Bedrock Converse it must ride along inside
|
||||
# ``additionalModelRequestFields`` so the model actually sees the
|
||||
# tier; stripping it (the prior behavior) silently flattened every
|
||||
# adaptive tier to identical thinking. Only the Bedrock-native
|
||||
# ``outputConfig`` (camelCase) goes at the top level.
|
||||
# Anthropic-only ``output_config`` (snake_case) — re-attached to
|
||||
# ``additionalModelRequestFields`` for Anthropic models below. The
|
||||
# Bedrock-native ``outputConfig`` (camelCase) is handled separately.
|
||||
anthropic_output_config = inference_params.pop("output_config", None)
|
||||
|
||||
# Extract requestMetadata before processing other parameters
|
||||
|
|
@ -1342,18 +1272,11 @@ class AmazonConverseConfig(BaseConfig):
|
|||
additional_request_params
|
||||
)
|
||||
|
||||
# Re-attach the Anthropic ``output_config`` (e.g. adaptive thinking
|
||||
# ``{"effort": "max"}``) onto additional_request_params for Anthropic
|
||||
# Bedrock models so the wire request carries the requested tier. Other
|
||||
# model families (Nova, GPT-OSS, ...) don't accept it; drop it for them.
|
||||
if anthropic_output_config is not None and isinstance(
|
||||
anthropic_output_config, dict
|
||||
):
|
||||
base_model = BedrockModelInfo.get_base_model(model)
|
||||
if base_model.startswith("anthropic"):
|
||||
# When ``drop_params`` is set, strip for models that don't
|
||||
# accept effort (e.g. proxy routing Claude Code at haiku-3).
|
||||
# Otherwise forward and let Bedrock surface the model's error.
|
||||
if (
|
||||
litellm.drop_params is True
|
||||
and not AnthropicConfig._model_supports_effort_param(model)
|
||||
|
|
@ -1495,12 +1418,8 @@ class AmazonConverseConfig(BaseConfig):
|
|||
# Append pre-formatted tools (systemTool etc.) after transformation
|
||||
bedrock_tools.extend(pre_formatted_tools)
|
||||
|
||||
# Auto-attach the effort beta header for non-adaptive Anthropic
|
||||
# models on Bedrock Converse (i.e. Opus 4.5). Claude 4.6/4.7 accept
|
||||
# ``output_config.effort`` as a stable, GA feature with no beta
|
||||
# header; Opus 4.5 still gates it behind ``effort-2025-11-24``. The
|
||||
# check mirrors ``AnthropicModelInfo.is_effort_used`` (which returns
|
||||
# False for adaptive models) so we don't double-flag adaptive routes.
|
||||
# Opus 4.5 gates ``output_config.effort`` behind a beta header;
|
||||
# Claude 4.6/4.7 accept it without one.
|
||||
base_model = BedrockModelInfo.get_base_model(model)
|
||||
if base_model.startswith("anthropic"):
|
||||
output_config = additional_request_params.get("output_config")
|
||||
|
|
|
|||
|
|
@ -169,11 +169,6 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig):
|
|||
anthropic_request.pop("model", None)
|
||||
anthropic_request.pop("stream", None)
|
||||
anthropic_request.pop("output_format", None)
|
||||
# ``output_config`` (e.g. ``{"effort": "max"}``) is the adaptive-thinking
|
||||
# tier payload for Claude 4.6 / 4.7. Bedrock Invoke accepts it for
|
||||
# those models — stripping it (the prior behavior) silently flattened
|
||||
# every adaptive tier on this route. Forward it; if the model rejects
|
||||
# it the surfaced error is correct, vs. swallowing the user's knob.
|
||||
if "anthropic_version" not in anthropic_request:
|
||||
anthropic_request["anthropic_version"] = self.anthropic_version
|
||||
|
||||
|
|
|
|||
|
|
@ -582,10 +582,6 @@ class AmazonAnthropicClaudeMessagesConfig(
|
|||
if filtered_betas:
|
||||
anthropic_messages_request["anthropic_beta"] = filtered_betas
|
||||
|
||||
# 6a. When ``drop_params`` is set, strip ``output_config`` for models
|
||||
# that don't accept it (e.g. proxy fronting Claude Code at haiku-3).
|
||||
# Without this, every Claude Code request to a pre-4.5 Anthropic model
|
||||
# routes a forced 400 from Bedrock that the client can't fix.
|
||||
if (
|
||||
litellm.drop_params is True
|
||||
and "output_config" in anthropic_messages_request
|
||||
|
|
|
|||
|
|
@ -334,11 +334,6 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
|
|||
) # unsupported for claude models - if json_schema -> convert to tool call
|
||||
|
||||
if "reasoning_effort" in non_default_params and "claude" in model:
|
||||
# ``_map_reasoning_effort`` raises ``BadRequestError`` (400)
|
||||
# directly on unmapped efforts; pass ``llm_provider="databricks"``
|
||||
# so the surfaced error carries the correct provider name (the
|
||||
# default is ``"anthropic"``, which would mislead users routing
|
||||
# via Databricks Foundation Model APIs).
|
||||
reasoning_effort_value = non_default_params.get("reasoning_effort")
|
||||
mapped_thinking = AnthropicConfig._map_reasoning_effort(
|
||||
reasoning_effort=reasoning_effort_value,
|
||||
|
|
@ -350,21 +345,7 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig):
|
|||
optional_params.pop("output_config", None)
|
||||
else:
|
||||
optional_params["thinking"] = mapped_thinking
|
||||
# For Claude 4.6+ adaptive models, ``_map_reasoning_effort``
|
||||
# returns ``type=adaptive`` for ANY non-None / non-"none"
|
||||
# string without validating the value, so reject unmapped
|
||||
# efforts here and set ``output_config.effort`` (matching the
|
||||
# Anthropic native / Bedrock Converse / Bedrock Invoke /
|
||||
# /v1/messages paths). Driven by ``supports_adaptive_thinking``
|
||||
# in the model map so future adaptive Claudes land via a
|
||||
# model-map update rather than a code release — the
|
||||
# Anthropic-native and Bedrock routes already use the same
|
||||
# helper, so all three paths stay in lock-step.
|
||||
if AnthropicConfig._is_adaptive_thinking_model(model):
|
||||
# ``reasoning_effort_value`` comes from ``non_default_params``
|
||||
# so its static type is ``Any | None``. Narrow to ``str`` for
|
||||
# the mapping lookup; non-strings fall through to the
|
||||
# ``BadRequestError`` below with a clean validation message.
|
||||
mapped_effort: Optional[str] = None
|
||||
if isinstance(reasoning_effort_value, str):
|
||||
mapped_effort = REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT.get(
|
||||
|
|
|
|||
|
|
@ -159,11 +159,6 @@ class VertexAIPartnerModelsAnthropicMessagesConfig(AnthropicMessagesConfig, Vert
|
|||
"model", None
|
||||
) # do not pass model in request body to vertex ai
|
||||
|
||||
# Vertex AI Claude accepts ``output_config.format`` (structured outputs),
|
||||
# ``output_format``, and ``output_config.effort`` (adaptive-thinking
|
||||
# tier on Claude 4.6 / 4.7, verified end-to-end). The shared sanitize
|
||||
# helper now no-ops for ``effort`` and remains the single hook for any
|
||||
# future Vertex-only key drift.
|
||||
sanitize_vertex_anthropic_output_params(anthropic_messages_request)
|
||||
|
||||
return anthropic_messages_request
|
||||
|
|
|
|||
|
|
@ -11,11 +11,8 @@ keeps the parent module's import surface narrow.
|
|||
"""
|
||||
|
||||
# Keys inside ``output_config`` that Vertex AI Claude does not accept.
|
||||
# Vertex now accepts ``output_config.effort`` for the adaptive-thinking
|
||||
# Claude 4.6 / 4.7 models on direct ``:rawPredict`` (verified end-to-end
|
||||
# against ``us-east5`` for ``opus-4-6`` and ``global`` for ``opus-4-7``).
|
||||
# Keep this set narrow and only add a key here once a 400 "Extra inputs are
|
||||
# not permitted" is reproducible against the live Vertex endpoint.
|
||||
# Add an entry only when a 400 "Extra inputs are not permitted" is
|
||||
# reproducible against the live Vertex endpoint.
|
||||
VERTEX_UNSUPPORTED_OUTPUT_CONFIG_KEYS: frozenset = frozenset()
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -106,10 +106,6 @@ class VertexAIAnthropicConfig(AnthropicConfig):
|
|||
|
||||
data.pop("model", None) # vertex anthropic doesn't accept 'model' parameter
|
||||
|
||||
# Sanitize ``output_config`` / ``output_format`` for Vertex parity.
|
||||
# Vertex now accepts ``output_config.effort`` for adaptive-thinking Claude
|
||||
# 4.6 / 4.7 models, so the helper is a no-op for ``effort``; it remains
|
||||
# the single hook for future Vertex-only sanitization.
|
||||
sanitize_vertex_anthropic_output_params(data)
|
||||
|
||||
tools = optional_params.get("tools")
|
||||
|
|
|
|||
|
|
@ -224,12 +224,6 @@ class XAIChatConfig(OpenAIGPTConfig):
|
|||
except Exception as e:
|
||||
verbose_logger.debug(f"Error extracting X.AI web search usage: {e}")
|
||||
|
||||
# X.AI excludes reasoning_tokens from completion_tokens, breaking the
|
||||
# OpenAI invariant total_tokens == prompt_tokens + completion_tokens.
|
||||
# OpenAI o1/o3 fold reasoning into completion_tokens; align X.AI to
|
||||
# match so downstream consumers (including litellm's own usage tests)
|
||||
# see a self-consistent Usage object. Cost calc already accounts for
|
||||
# reasoning tokens separately in xai.cost_calculator.
|
||||
self._fold_reasoning_tokens_into_completion(response)
|
||||
return response
|
||||
|
||||
|
|
@ -237,15 +231,10 @@ class XAIChatConfig(OpenAIGPTConfig):
|
|||
def _fold_reasoning_tokens_into_completion(model_response: ModelResponse) -> None:
|
||||
"""Reconcile xAI Usage to the OpenAI invariant.
|
||||
|
||||
xAI returns ``completion_tokens`` covering only visible output and
|
||||
accounts ``reasoning_tokens`` separately, while still rolling them
|
||||
into ``total_tokens``. OpenAI's published contract (o1/o3) includes
|
||||
reasoning in ``completion_tokens``. Tests that assert
|
||||
``total_tokens == prompt_tokens + completion_tokens`` (e.g.
|
||||
``_usage_format_tests``) fail on the raw xAI shape.
|
||||
|
||||
This helper is idempotent: if ``completion_tokens`` already covers
|
||||
the gap, no change is made.
|
||||
xAI accounts ``reasoning_tokens`` separately from
|
||||
``completion_tokens`` while still summing them into ``total_tokens``.
|
||||
OpenAI's contract (o1/o3) folds reasoning into ``completion_tokens``,
|
||||
so fold here to keep ``total = prompt + completion``. Idempotent.
|
||||
"""
|
||||
usage = getattr(model_response, "usage", None)
|
||||
if usage is None:
|
||||
|
|
@ -262,13 +251,10 @@ class XAIChatConfig(OpenAIGPTConfig):
|
|||
completion_tokens = int(getattr(usage, "completion_tokens", 0) or 0)
|
||||
total_tokens = int(getattr(usage, "total_tokens", 0) or 0)
|
||||
|
||||
# Already consistent — nothing to do.
|
||||
if total_tokens == prompt_tokens + completion_tokens:
|
||||
return
|
||||
|
||||
# Only fold when xAI's accounting (total = prompt + completion +
|
||||
# reasoning) explains the gap. This guards against double-counting
|
||||
# if xAI ever changes their semantics.
|
||||
# Guard against double-counting if xAI changes accounting.
|
||||
if total_tokens != prompt_tokens + completion_tokens + reasoning_tokens:
|
||||
return
|
||||
|
||||
|
|
|
|||
|
|
@ -25,13 +25,10 @@ def cost_per_token(model: str, usage: Usage) -> Tuple[float, float]:
|
|||
Returns:
|
||||
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
|
||||
"""
|
||||
# XAI-specific completion cost calculation
|
||||
# For XAI models, completion is billed as (visible completion tokens + reasoning tokens).
|
||||
# The transformation layer normalises Usage to the OpenAI invariant
|
||||
# (completion_tokens includes reasoning_tokens), so detect that and avoid
|
||||
# double-counting. Fall back to the raw xAI shape (visible-only completion +
|
||||
# reasoning kept in completion_tokens_details) for callers that bypass the
|
||||
# transformation, e.g. proxy logs replayed straight into cost calc.
|
||||
# XAI-specific completion cost: completion is billed as visible + reasoning
|
||||
# tokens. Detect when the transformation layer already folded them so we
|
||||
# don't double-count; fall back to raw xAI shape for callers that bypass
|
||||
# the transformation (e.g. proxy logs replayed into cost calc).
|
||||
prompt_tokens = int(getattr(usage, "prompt_tokens", 0) or 0)
|
||||
completion_tokens = int(getattr(usage, "completion_tokens", 0) or 0)
|
||||
total_tokens = int(getattr(usage, "total_tokens", 0) or 0)
|
||||
|
|
|
|||
|
|
@ -393,12 +393,6 @@ class AnthropicMessagesRequestOptionalParams(TypedDict, total=False):
|
|||
AnthropicOutputConfig
|
||||
] # Configuration for Claude's output behavior
|
||||
cache_control: Optional[Dict[str, Any]] # Automatic prompt caching
|
||||
# OpenAI-style ``reasoning_effort`` is accepted on /v1/messages so callers
|
||||
# can drive adaptive/extended thinking with a single tier-name knob (the
|
||||
# same vocabulary as the chat completion path). The transformation layer
|
||||
# maps it to native Anthropic ``thinking`` + ``output_config`` and pops
|
||||
# this key before the request is forwarded — no provider receives
|
||||
# ``reasoning_effort`` on the wire.
|
||||
reasoning_effort: Optional[str]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1041,12 +1041,4 @@ class BedrockInvokeAnthropicMessagesRequest(TypedDict, total=False):
|
|||
# `metadata` is part of the common Anthropic Messages API shape.
|
||||
thinking: dict
|
||||
metadata: dict
|
||||
|
||||
# ``output_config`` is the adaptive-thinking effort payload for
|
||||
# Claude 4.6 / 4.7 (e.g. ``{"effort": "max"}``). Bedrock Invoke
|
||||
# accepts it for these models when ``thinking={"type": "adaptive"}``.
|
||||
# Without this field in the allowlist, the runtime filter in
|
||||
# ``AmazonAnthropicClaudeMessagesConfig.transform_anthropic_messages_request``
|
||||
# silently drops it and every adaptive tier collapses to identical
|
||||
# behavior on /v1/messages.
|
||||
output_config: dict
|
||||
|
|
|
|||
|
|
@ -1,37 +1,10 @@
|
|||
"""Force ``litellm.model_cost`` to load from the PR-local JSON for these tests.
|
||||
|
||||
By default ``litellm.model_cost`` is fetched from the main branch on GitHub,
|
||||
which lags behind PR-branch flag additions. This fixture loads the local
|
||||
file so per-model flag tests pass in CI as well as locally.
|
||||
See https://github.com/BerriAI/litellm/issues/27122.
|
||||
"""
|
||||
Local-conftest for ``tests/test_litellm/llms/anthropic/chat``.
|
||||
|
||||
Why this exists
|
||||
---------------
|
||||
``litellm.model_cost`` is loaded once at ``litellm.__init__`` time. By default
|
||||
it fetches ``model_prices_and_context_window.json`` from the **main branch on
|
||||
GitHub** (``litellm.model_cost_map_url``) — *not* from the PR-branch JSON in
|
||||
the working tree. That works fine in production (operators get new models
|
||||
without redeploying litellm) but is the wrong default for tests, which need
|
||||
to validate the code in front of them against the data in front of them.
|
||||
|
||||
Several anthropic chat transformation tests (``test_supports_effort_level_*``,
|
||||
``test_anthropic_model_supports_effort_param_*``) assert per-model flags
|
||||
like ``supports_max_reasoning_effort`` / ``supports_xhigh_reasoning_effort``
|
||||
that may exist in the PR-local JSON but not yet in main's JSON. Without this
|
||||
fixture those tests pass locally (where AGENTS.md tells contributors to set
|
||||
``LITELLM_LOCAL_MODEL_COST_MAP=True``) but fail in CI (which doesn't set the
|
||||
env var) — the chicken-and-egg PR adds flag → CI fetches main without flag
|
||||
→ test fails → flag never lands on main.
|
||||
|
||||
The fixture force-loads ``litellm.model_cost`` from the local JSON for every
|
||||
test in this directory. ``tests/test_litellm/conftest.py`` already snapshots
|
||||
and restores ``litellm.model_cost`` per-function, so this mutation is safe
|
||||
and contained.
|
||||
|
||||
This is a scoped workaround. The proper fix is to set
|
||||
``LITELLM_LOCAL_MODEL_COST_MAP=True`` globally in the test workflow once the
|
||||
~10 inline-set test files have been audited and the few tests that exercise
|
||||
the remote-fetch / integrity-validation path have been given carve-outs.
|
||||
Tracked at https://github.com/BerriAI/litellm/issues/27122.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -41,10 +14,6 @@ from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
|
|||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _use_pr_local_model_cost_map(monkeypatch):
|
||||
"""Force ``litellm.model_cost`` to the PR-branch JSON for the duration of
|
||||
each test. ``monkeypatch`` reverts the env var after the test; the parent
|
||||
conftest restores ``litellm.model_cost`` from its snapshot.
|
||||
"""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(
|
||||
litellm,
|
||||
|
|
|
|||
|
|
@ -1632,7 +1632,6 @@ def test_effort_validation():
|
|||
)
|
||||
assert result["output_config"]["effort"] == effort
|
||||
|
||||
# Invalid value should raise BadRequestError (clean 400, not a 500).
|
||||
with pytest.raises(
|
||||
litellm.exceptions.BadRequestError, match="Invalid effort value"
|
||||
):
|
||||
|
|
@ -1685,10 +1684,7 @@ def test_effort_validation_with_opus_46():
|
|||
|
||||
|
||||
def test_max_effort_rejected_for_opus_45():
|
||||
"""Test that effort='max' is rejected when using Claude Opus 4.5.
|
||||
|
||||
Surfaces as a clean 400 BadRequestError, not a 500 ValueError.
|
||||
"""
|
||||
"""Test that effort='max' is rejected when using Claude Opus 4.5."""
|
||||
config = AnthropicConfig()
|
||||
|
||||
messages = [{"role": "user", "content": "Test"}]
|
||||
|
|
@ -1749,12 +1745,7 @@ def test_effort_with_other_features():
|
|||
|
||||
|
||||
def test_anthropic_drop_params_strips_output_config_for_pre_4_5_models():
|
||||
"""
|
||||
Proxies fronting Claude Code at pre-4.5 Anthropic models receive
|
||||
``output_config`` injected by the client; without ``drop_params`` Bedrock /
|
||||
Anthropic 400s. With ``drop_params=True`` we strip it (logged) so the
|
||||
request can succeed.
|
||||
"""
|
||||
"""``drop_params=True`` strips unsupported ``output_config`` for pre-4.5 models."""
|
||||
config = AnthropicConfig()
|
||||
messages = [{"role": "user", "content": "Hello"}]
|
||||
|
||||
|
|
@ -1796,10 +1787,7 @@ def test_anthropic_drop_params_keeps_output_config_for_supporting_models():
|
|||
|
||||
|
||||
def test_anthropic_drop_params_false_forwards_to_unsupported_model():
|
||||
"""
|
||||
Default behavior: forward ``output_config`` and let the provider 400.
|
||||
This is the contract for users who want strict, debuggable failures.
|
||||
"""
|
||||
"""Default ``drop_params=False`` forwards ``output_config`` and lets the provider 400."""
|
||||
config = AnthropicConfig()
|
||||
messages = [{"role": "user", "content": "Hello"}]
|
||||
|
||||
|
|
@ -2059,10 +2047,7 @@ def test_get_config_without_model_uses_fallback():
|
|||
|
||||
|
||||
def test_get_config_does_not_leak_module_constants():
|
||||
"""``BaseConfig.get_config`` returns class attributes; the
|
||||
reasoning-effort mapping must not be one of them or it ends up
|
||||
serialised onto the wire as an extra request key.
|
||||
"""
|
||||
"""``get_config`` must not leak the reasoning-effort lookup table onto the wire."""
|
||||
cfg = AnthropicConfig.get_config(model="claude-opus-4-7")
|
||||
for forbidden in (
|
||||
"REASONING_EFFORT_TO_OUTPUT_CONFIG_EFFORT",
|
||||
|
|
@ -2078,10 +2063,6 @@ def test_get_config_does_not_leak_module_constants():
|
|||
("claude-opus-4-7", "xhigh", True),
|
||||
("claude-opus-4-6", "max", True),
|
||||
("claude-opus-4-6", "xhigh", False),
|
||||
# ``max`` is documented as supported on Sonnet 4.6 (Claude 4.6 family).
|
||||
# The model-map JSON now carries ``supports_max_reasoning_effort: true``
|
||||
# for every Sonnet 4.6 entry; ``_supports_effort_level`` should report
|
||||
# ``True`` on every route prefix (anthropic, bedrock, vertex, azure).
|
||||
("claude-sonnet-4-6", "max", True),
|
||||
("claude-sonnet-4-6", "xhigh", False),
|
||||
("bedrock/invoke/us.anthropic.claude-opus-4-7", "max", True),
|
||||
|
|
@ -2094,27 +2075,21 @@ def test_get_config_does_not_leak_module_constants():
|
|||
],
|
||||
)
|
||||
def test_supports_effort_level_handles_provider_prefixes(model, level, expected):
|
||||
"""``_supports_effort_level`` must handle bedrock/ vertex_ai/ azure_ai/
|
||||
prefixed model ids so per-model gating works on every route, not just
|
||||
the bare-Anthropic chat completion path.
|
||||
"""
|
||||
"""``_supports_effort_level`` resolves bedrock/vertex/azure-prefixed model ids."""
|
||||
assert AnthropicConfig._supports_effort_level(model, level) is expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,effort,expect_error",
|
||||
[
|
||||
# ``max`` accepted on 4.6 / 4.7 (family fallback) and rejected on 4.5.
|
||||
("claude-opus-4-6", "max", False),
|
||||
("claude-sonnet-4-6", "max", False),
|
||||
("claude-opus-4-7", "max", False),
|
||||
("claude-opus-4-5-20251101", "max", True),
|
||||
("claude-sonnet-4-5", "max", True),
|
||||
# ``xhigh`` data-driven; only 4.7 carries the flag in the model map.
|
||||
("claude-opus-4-7", "xhigh", False),
|
||||
("claude-opus-4-6", "xhigh", True),
|
||||
("claude-sonnet-4-6", "xhigh", True),
|
||||
# Lower efforts and ``None`` always pass the gate.
|
||||
("claude-opus-4-5-20251101", "high", False),
|
||||
("claude-haiku-4-5", "low", False),
|
||||
("claude-opus-4-5-20251101", None, False),
|
||||
|
|
@ -2123,13 +2098,6 @@ def test_supports_effort_level_handles_provider_prefixes(model, level, expected)
|
|||
def test_validate_effort_for_model_centralises_per_model_gating(
|
||||
model, effort, expect_error
|
||||
):
|
||||
"""``_validate_effort_for_model`` is the single source of truth for the
|
||||
per-model ``max`` / ``xhigh`` gating that ``_apply_output_config`` (chat
|
||||
completion path) and ``AnthropicMessagesConfig._translate_reasoning_effort_to_anthropic``
|
||||
(/v1/messages pass-through) both rely on. Both call sites raise their
|
||||
own provider-appropriate exception, but the gating decision must come
|
||||
from one place to prevent drift when a new model tier lands.
|
||||
"""
|
||||
err = AnthropicConfig._validate_effort_for_model(model, effort)
|
||||
if expect_error:
|
||||
assert err is not None
|
||||
|
|
@ -2342,14 +2310,12 @@ def test_reasoning_effort_maps_to_budget_thinking_for_non_opus_4_6():
|
|||
"""
|
||||
config = AnthropicConfig()
|
||||
|
||||
# Test with Claude Sonnet 4.5 (non-Opus 4.6 model).
|
||||
# ``minimal`` floors at the Anthropic provider minimum (1024) because
|
||||
# Anthropic / Azure / Vertex / Bedrock Invoke 400 below that.
|
||||
# ``minimal`` floors at ANTHROPIC_MIN_THINKING_BUDGET_TOKENS (1024).
|
||||
test_cases = [
|
||||
("low", 1024), # DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET
|
||||
("medium", 2048), # DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET
|
||||
("high", 4096), # DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET
|
||||
("minimal", 1024), # ANTHROPIC_MIN_THINKING_BUDGET_TOKENS (provider floor)
|
||||
("low", 1024),
|
||||
("medium", 2048),
|
||||
("high", 4096),
|
||||
("minimal", 1024),
|
||||
]
|
||||
|
||||
for effort, expected_budget in test_cases:
|
||||
|
|
@ -2447,16 +2413,7 @@ def test_reasoning_effort_does_not_set_output_config_for_older_models():
|
|||
],
|
||||
)
|
||||
def test_max_effort_accepted_for_sonnet_46_variants(model):
|
||||
"""``effort='max'`` is documented as supported on Claude 4.6 (Opus + Sonnet)
|
||||
and Claude 4.7 (https://platform.claude.com/docs/en/build-with-claude/effort).
|
||||
|
||||
Earlier versions of this test asserted a 400 for Sonnet 4.6, mirroring an
|
||||
Opus-only allow-list in ``_apply_output_config``. That gate has since been
|
||||
widened to ``_is_claude_4_6_model`` (Opus + Sonnet) and the
|
||||
``supports_max_reasoning_effort`` JSON flag, matching Anthropic's published
|
||||
matrix. Verify the param actually flows through every Sonnet 4.6 id
|
||||
variant our routing layer might see.
|
||||
"""
|
||||
"""``effort='max'`` is supported on Claude 4.6 (Opus + Sonnet) and 4.7."""
|
||||
config = AnthropicConfig()
|
||||
messages = [{"role": "user", "content": "Test"}]
|
||||
|
||||
|
|
@ -2552,9 +2509,7 @@ def test_reasoning_effort_none_omits_thinking_and_output_config(model):
|
|||
["disabled", "invalid", ""],
|
||||
)
|
||||
def test_reasoning_effort_garbage_raises_bad_request(effort):
|
||||
"""Unmapped / garbage / empty-string reasoning_effort surfaces as a clean
|
||||
400 ``BadRequestError`` instead of letting ``ValueError`` propagate as 500.
|
||||
"""
|
||||
"""Unmapped reasoning_effort raises BadRequestError (clean 400, not a 500)."""
|
||||
config = AnthropicConfig()
|
||||
|
||||
with pytest.raises(litellm.exceptions.BadRequestError):
|
||||
|
|
@ -2573,17 +2528,7 @@ def test_reasoning_effort_garbage_raises_bad_request(effort):
|
|||
def test_reasoning_effort_xhigh_max_maps_to_budget_on_budget_model(
|
||||
effort, expected_budget
|
||||
):
|
||||
"""``xhigh`` / ``max`` extend the legacy ``thinking.budget_tokens``
|
||||
progression (low=1024 / medium=2048 / high=4096 → xhigh=8192 / max=16384)
|
||||
on budget-mode Claude models (haiku / 4.5 series).
|
||||
|
||||
Adopted from #27051. Keeps the OpenAI-format ``reasoning_effort`` knob
|
||||
usable across the full Claude lineup — adaptive models (4.6/4.7) route
|
||||
these tiers via ``output_config.effort``; budget-mode models use the
|
||||
extended budget. Anthropic's "max only on Mythos / Opus 4.7 / Opus 4.6 /
|
||||
Sonnet 4.6" gating applies to the *adaptive enum*, not to the legacy
|
||||
``budget_tokens`` knob, which accepts any integer up to ``max_tokens``.
|
||||
"""
|
||||
"""``xhigh`` / ``max`` extend the budget_tokens progression on budget-mode models."""
|
||||
config = AnthropicConfig()
|
||||
|
||||
result = config.map_openai_params(
|
||||
|
|
@ -2595,16 +2540,11 @@ def test_reasoning_effort_xhigh_max_maps_to_budget_on_budget_model(
|
|||
|
||||
assert result["thinking"]["type"] == "enabled"
|
||||
assert result["thinking"]["budget_tokens"] == expected_budget
|
||||
# Budget-mode models must NOT carry an ``output_config`` payload — that
|
||||
# path is exclusively for adaptive (4.6+) models.
|
||||
assert "output_config" not in result
|
||||
|
||||
|
||||
def test_output_config_effort_empty_string_raises_bad_request():
|
||||
"""``output_config={"effort": ""}`` must be rejected with a 400 — the
|
||||
legacy ``if effort and ...`` short-circuit silently let it pass
|
||||
through (verified end-to-end on the QA sweep for PR #27039).
|
||||
"""
|
||||
"""``output_config={"effort": ""}`` is rejected with a 400."""
|
||||
config = AnthropicConfig()
|
||||
|
||||
with pytest.raises(litellm.exceptions.BadRequestError, match="Invalid effort"):
|
||||
|
|
@ -2618,10 +2558,7 @@ def test_output_config_effort_empty_string_raises_bad_request():
|
|||
|
||||
|
||||
def test_reasoning_effort_minimal_floors_at_anthropic_provider_minimum():
|
||||
"""Anthropic Messages API rejects ``budget_tokens < 1024``. ``minimal``
|
||||
must floor at the provider minimum so it's a usable tier on direct
|
||||
Anthropic / Azure AI Anthropic / Vertex AI Anthropic / Bedrock Invoke.
|
||||
"""
|
||||
"""``minimal`` floors at the Anthropic provider minimum (1024)."""
|
||||
config = AnthropicConfig()
|
||||
|
||||
result = config.map_openai_params(
|
||||
|
|
|
|||
|
|
@ -1,18 +1,4 @@
|
|||
"""
|
||||
Tests for OpenAI-style ``reasoning_effort`` translation on the Anthropic
|
||||
/v1/messages route.
|
||||
|
||||
The /v1/messages spec doesn't include ``reasoning_effort`` — without
|
||||
translation it gets silently dropped at the filter step, leaving every
|
||||
adaptive tier collapsed to the same behavior on Bedrock Invoke /v1/messages
|
||||
(and on Anthropic / Azure AI / Vertex AI when callers pass it on the
|
||||
messages route).
|
||||
|
||||
These tests pin the translation and validation behavior at the shared
|
||||
``AnthropicMessagesConfig`` level so all four /v1/messages routes
|
||||
(direct Anthropic, Azure AI, Vertex AI, Bedrock Invoke) inherit the
|
||||
same mapping.
|
||||
"""
|
||||
"""Tests for ``reasoning_effort`` translation on the Anthropic /v1/messages route."""
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -36,12 +22,6 @@ from litellm.llms.anthropic.experimental_pass_through.messages.transformation im
|
|||
def test_reasoning_effort_maps_to_output_config_for_adaptive_model(
|
||||
reasoning_effort, expected_effort
|
||||
):
|
||||
"""
|
||||
For Claude 4.6 / 4.7, ``reasoning_effort`` is mapped to
|
||||
``thinking={"type": "adaptive"}`` plus ``output_config.effort=<tier>``,
|
||||
using the same mapping table as the chat completion path so the two
|
||||
routes can't drift.
|
||||
"""
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": reasoning_effort}
|
||||
|
||||
|
|
@ -59,7 +39,6 @@ def test_reasoning_effort_maps_to_output_config_for_adaptive_model(
|
|||
|
||||
|
||||
def test_reasoning_effort_none_clears_thinking_and_output_config():
|
||||
"""``reasoning_effort='none'`` opts out of extended thinking entirely."""
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
|
|
@ -82,11 +61,6 @@ def test_reasoning_effort_none_clears_thinking_and_output_config():
|
|||
|
||||
|
||||
def test_reasoning_effort_on_non_adaptive_model_uses_thinking_budget():
|
||||
"""
|
||||
Non-adaptive models (Opus 4.5 / earlier) take ``thinking.budget_tokens``
|
||||
rather than ``output_config.effort``. The translation falls back to
|
||||
the budget mapping in that case.
|
||||
"""
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": "high"}
|
||||
|
||||
|
|
@ -109,11 +83,6 @@ def test_reasoning_effort_on_non_adaptive_model_uses_thinking_budget():
|
|||
|
||||
@pytest.mark.parametrize("bad_effort", ["invalid", "disabled", ""])
|
||||
def test_invalid_reasoning_effort_raises_400(bad_effort):
|
||||
"""
|
||||
Garbage ``reasoning_effort`` values surface as a clean 400 instead of
|
||||
silently passing through to the provider as an unknown
|
||||
``output_config.effort`` (which would 500).
|
||||
"""
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort}
|
||||
|
||||
|
|
@ -132,18 +101,12 @@ def test_invalid_reasoning_effort_raises_400(bad_effort):
|
|||
@pytest.mark.parametrize(
|
||||
"model,bad_effort",
|
||||
[
|
||||
# ``xhigh`` is Opus-4.7-only on the public Anthropic effort matrix,
|
||||
# so Opus 4.6 / Sonnet 4.6 must still 400 on it.
|
||||
("claude-opus-4-6", "xhigh"),
|
||||
("bedrock/invoke/us.anthropic.claude-opus-4-6-v1", "xhigh"),
|
||||
("claude-sonnet-4-6", "xhigh"),
|
||||
],
|
||||
)
|
||||
def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort):
|
||||
"""``xhigh`` and ``max`` are gated per-model. The /v1/messages route must
|
||||
surface a clean 400 client-side instead of forwarding the unsupported
|
||||
tier and letting the provider 500/400 it.
|
||||
"""
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {"max_tokens": 1024, "reasoning_effort": bad_effort}
|
||||
|
||||
|
|
@ -163,9 +126,6 @@ def test_reasoning_effort_unsupported_tier_raises_400_messages(model, bad_effort
|
|||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
# ``max`` is documented as supported on Claude 4.6 (Opus + Sonnet)
|
||||
# and Claude 4.7. Verify the /v1/messages route accepts it for
|
||||
# Sonnet 4.6 variants instead of 400-ing client-side.
|
||||
"claude-sonnet-4-6",
|
||||
"bedrock/invoke/us.anthropic.claude-sonnet-4-6",
|
||||
],
|
||||
|
|
@ -187,11 +147,6 @@ def test_reasoning_effort_max_accepted_on_sonnet_46_messages(model):
|
|||
|
||||
|
||||
def test_explicit_output_config_wins_over_reasoning_effort():
|
||||
"""
|
||||
Explicit native ``output_config.effort`` is never overridden by the
|
||||
OpenAI alias. Same precedence as
|
||||
``_translate_legacy_thinking_for_adaptive_model``.
|
||||
"""
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
|
|
@ -212,7 +167,6 @@ def test_explicit_output_config_wins_over_reasoning_effort():
|
|||
|
||||
|
||||
def test_explicit_thinking_wins_over_reasoning_effort():
|
||||
"""Explicit native ``thinking`` is never overridden by the alias."""
|
||||
config = AnthropicMessagesConfig()
|
||||
optional_params = {
|
||||
"max_tokens": 1024,
|
||||
|
|
@ -233,8 +187,6 @@ def test_explicit_thinking_wins_over_reasoning_effort():
|
|||
|
||||
|
||||
def test_reasoning_effort_in_supported_params():
|
||||
"""``reasoning_effort`` is advertised as a supported messages param so
|
||||
callers and validation paths can introspect the schema."""
|
||||
config = AnthropicMessagesConfig()
|
||||
assert "reasoning_effort" in config.get_supported_anthropic_messages_params(
|
||||
"claude-opus-4-7"
|
||||
|
|
|
|||
|
|
@ -292,19 +292,10 @@ class TestAzureAnthropicConfig:
|
|||
assert "compact-2026-01-12" in headers["anthropic-beta"]
|
||||
|
||||
def test_output_config_promoted_from_extra_body(self):
|
||||
"""Anthropic-native ``output_config`` routed via ``extra_body`` (by the
|
||||
openai-compatible kwarg stuffer in ``litellm/utils.py``) must be
|
||||
promoted to top-level ``optional_params`` before delegating to
|
||||
``AnthropicConfig.transform_request``. Otherwise the value is
|
||||
silently dropped by the ``extra_body`` pop and validation never
|
||||
runs.
|
||||
"""
|
||||
"""Anthropic ``output_config`` routed via ``extra_body`` reaches the request body."""
|
||||
config = AzureAnthropicConfig()
|
||||
|
||||
messages = [{"role": "user", "content": "Hello"}]
|
||||
# Simulate what ``add_provider_specific_params_to_optional_params``
|
||||
# produces for ``litellm.completion(model="azure_ai/claude-...",
|
||||
# output_config={"effort": "low"})``.
|
||||
optional_params = {
|
||||
"max_tokens": 100,
|
||||
"extra_body": {"output_config": {"effort": "low"}},
|
||||
|
|
@ -313,26 +304,18 @@ class TestAzureAnthropicConfig:
|
|||
headers = {"api-key": "test-key", "anthropic-version": "2023-06-01"}
|
||||
|
||||
result = config.transform_request(
|
||||
model="claude-opus-4-6", # supports output_config.effort
|
||||
model="claude-opus-4-6",
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
headers=headers,
|
||||
)
|
||||
|
||||
# output_config must reach the request body (not be silently dropped)
|
||||
assert "output_config" in result
|
||||
assert result["output_config"] == {"effort": "low"}
|
||||
# extra_body should be stripped on the way out
|
||||
assert "extra_body" not in result
|
||||
|
||||
def test_invalid_output_config_effort_raises_via_extra_body(self):
|
||||
"""``effort="invalid"`` arriving via ``extra_body`` must still raise
|
||||
``BadRequestError`` (matching the native ``anthropic`` provider).
|
||||
Previously this was silently dropped on Azure, so unsupported
|
||||
efforts reached the model with default behavior instead of
|
||||
returning a clean 400.
|
||||
"""
|
||||
"""Invalid ``effort`` via ``extra_body`` raises BadRequestError."""
|
||||
import litellm
|
||||
|
||||
config = AzureAnthropicConfig()
|
||||
|
|
@ -356,9 +339,7 @@ class TestAzureAnthropicConfig:
|
|||
assert "Invalid effort value" in str(exc_info.value)
|
||||
|
||||
def test_unsupported_effort_xhigh_raises_via_extra_body(self):
|
||||
"""``effort="xhigh"`` on a model that does not support it (e.g.
|
||||
Sonnet 4.6) must raise ``BadRequestError`` even when arriving via
|
||||
``extra_body`` on Azure."""
|
||||
"""Unsupported ``effort='xhigh'`` via ``extra_body`` raises BadRequestError."""
|
||||
import litellm
|
||||
|
||||
config = AzureAnthropicConfig()
|
||||
|
|
@ -382,17 +363,14 @@ class TestAzureAnthropicConfig:
|
|||
assert "xhigh" in str(exc_info.value)
|
||||
|
||||
def test_extra_body_promotion_does_not_clobber_top_level(self):
|
||||
"""If a key exists at both top-level ``optional_params`` and
|
||||
inside ``extra_body``, the top-level value wins (``setdefault``
|
||||
semantics). This protects against a caller who explicitly sets a
|
||||
param at top-level and accidentally also has it in extra_body."""
|
||||
"""Top-level ``optional_params`` wins over duplicates in ``extra_body``."""
|
||||
config = AzureAnthropicConfig()
|
||||
|
||||
messages = [{"role": "user", "content": "Hello"}]
|
||||
optional_params = {
|
||||
"max_tokens": 100,
|
||||
"output_config": {"effort": "low"}, # top-level wins
|
||||
"extra_body": {"output_config": {"effort": "high"}}, # ignored
|
||||
"output_config": {"effort": "low"},
|
||||
"extra_body": {"output_config": {"effort": "high"}},
|
||||
}
|
||||
litellm_params = {"api_key": "test-key"}
|
||||
headers = {"api-key": "test-key", "anthropic-version": "2023-06-01"}
|
||||
|
|
|
|||
|
|
@ -407,18 +407,7 @@ def test_opus_4_5_model_detection():
|
|||
|
||||
|
||||
def test_output_config_forwarded_for_bedrock_chat_invoke_request():
|
||||
"""
|
||||
Bedrock Invoke (chat/completions route) must forward
|
||||
``output_config`` for Anthropic adaptive-thinking models. The earlier
|
||||
behavior stripped it unconditionally, which silently flattened every
|
||||
adaptive tier (``low``/``medium``/``high``/``xhigh``/``max``) to identical
|
||||
behavior on the wire.
|
||||
|
||||
The wire QA at https://github.com/BerriAI/litellm/pull/27039 showed
|
||||
``thinking.type: adaptive`` was forwarded but ``output_config.effort``
|
||||
was always missing, even though direct curls to Anthropic's Bedrock
|
||||
Invoke endpoint accept it.
|
||||
"""
|
||||
"""Bedrock Invoke chat path forwards ``output_config`` for adaptive Claude models."""
|
||||
config = AmazonAnthropicClaudeConfig()
|
||||
|
||||
messages = [{"role": "user", "content": "test"}]
|
||||
|
|
@ -437,7 +426,6 @@ def test_output_config_forwarded_for_bedrock_chat_invoke_request():
|
|||
)
|
||||
|
||||
assert result.get("output_config") == {"effort": "high"}
|
||||
# Verify normal params survive
|
||||
assert result["max_tokens"] == 100
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -326,10 +326,7 @@ def test_reasoning_effort_none_omits_thinking_for_anthropic_converse(model):
|
|||
def test_reasoning_effort_sets_output_config_for_adaptive_models_converse(
|
||||
model, effort, expected_effort
|
||||
):
|
||||
"""Adaptive-thinking Claude 4.6 / 4.7 on Bedrock Converse must carry the
|
||||
requested tier via ``output_config.effort``. The prior strip silently
|
||||
flattened every adaptive tier on the wire (verified in the PR #27039
|
||||
QA sweep)."""
|
||||
"""Adaptive Claude 4.6 / 4.7 on Bedrock Converse routes the tier via ``output_config.effort``."""
|
||||
config = AmazonConverseConfig()
|
||||
|
||||
optional_params = config.map_openai_params(
|
||||
|
|
@ -352,10 +349,7 @@ def test_reasoning_effort_sets_output_config_for_adaptive_models_converse(
|
|||
],
|
||||
)
|
||||
def test_output_config_effort_forwarded_into_additional_request_fields(model):
|
||||
"""``output_config`` must ride along inside ``additionalModelRequestFields``
|
||||
so the Anthropic-on-Bedrock wire request actually carries the effort
|
||||
tier. The prior ``inference_params.pop("output_config")`` dropped it
|
||||
on the floor for every adaptive tier."""
|
||||
"""``output_config`` rides along inside ``additionalModelRequestFields``."""
|
||||
config = AmazonConverseConfig()
|
||||
messages = [{"role": "user", "content": "hi"}]
|
||||
|
||||
|
|
@ -380,10 +374,7 @@ def test_output_config_effort_forwarded_into_additional_request_fields(model):
|
|||
["disabled", "invalid", ""],
|
||||
)
|
||||
def test_reasoning_effort_garbage_raises_bad_request_converse(effort):
|
||||
"""Garbage / empty-string reasoning_effort on Bedrock Converse Anthropic
|
||||
must surface as a clean 400 ``BadRequestError`` instead of 500. The
|
||||
earlier ``ValueError`` from ``_map_reasoning_effort`` propagated up as
|
||||
a generic 500 in the proxy and ate the request."""
|
||||
"""Unmapped reasoning_effort on Bedrock Converse Anthropic raises BadRequestError."""
|
||||
config = AmazonConverseConfig()
|
||||
|
||||
with pytest.raises(litellm.exceptions.BadRequestError):
|
||||
|
|
@ -405,13 +396,7 @@ def test_reasoning_effort_garbage_raises_bad_request_converse(effort):
|
|||
],
|
||||
)
|
||||
def test_output_config_effort_max_passes_through_on_sonnet_46_variants(model):
|
||||
"""``effort='max'`` is supported on Claude 4.6 (Opus + Sonnet) per
|
||||
https://platform.claude.com/docs/en/build-with-claude/effort. The earlier
|
||||
Opus-only allow-list in ``_validate_anthropic_adaptive_effort`` has been
|
||||
widened to ``_is_claude_4_6_model`` (Opus + Sonnet) plus the
|
||||
``supports_max_reasoning_effort`` JSON flag. Verify the param actually
|
||||
flows through to ``additionalModelRequestFields.output_config.effort``
|
||||
for every Bedrock Converse Sonnet 4.6 id variant."""
|
||||
"""``effort='max'`` flows through for every Bedrock Converse Sonnet 4.6 id."""
|
||||
config = AmazonConverseConfig()
|
||||
messages = [{"role": "user", "content": "hi"}]
|
||||
|
||||
|
|
@ -3450,9 +3435,7 @@ def test_transform_request_strips_anthropic_output_config():
|
|||
|
||||
|
||||
def test_converse_drop_params_strips_output_config_for_pre_4_5_anthropic():
|
||||
"""``drop_params=True`` strips ``output_config`` for pre-4.5 Anthropic
|
||||
models on Bedrock Converse so a proxy fronting Claude Code at haiku doesn't
|
||||
force a 400 on every request."""
|
||||
"""``drop_params=True`` strips unsupported ``output_config`` on Bedrock Converse."""
|
||||
config = AmazonConverseConfig()
|
||||
messages = [{"role": "user", "content": "hi"}]
|
||||
|
||||
|
|
@ -3477,7 +3460,7 @@ def test_converse_drop_params_strips_output_config_for_pre_4_5_anthropic():
|
|||
|
||||
|
||||
def test_converse_drop_params_keeps_output_config_for_supporting_anthropic():
|
||||
"""``drop_params=True`` must not strip on supporting models."""
|
||||
"""``drop_params=True`` does not strip on models that support ``output_config``."""
|
||||
config = AmazonConverseConfig()
|
||||
messages = [{"role": "user", "content": "hi"}]
|
||||
|
||||
|
|
|
|||
|
|
@ -593,16 +593,7 @@ def test_remove_scope_from_cache_control():
|
|||
|
||||
|
||||
def test_bedrock_messages_forwards_output_config():
|
||||
"""
|
||||
``output_config`` is the adaptive-thinking effort payload for Claude
|
||||
4.6 / 4.7 (e.g. ``{"effort": "max"}``). Bedrock Invoke accepts it for
|
||||
those models — the prior behavior of stripping it silently flattened
|
||||
every adaptive tier on /v1/messages so ``low`` / ``medium`` / ``high`` /
|
||||
``xhigh`` / ``max`` all collapsed to identical thinking with no tier
|
||||
differentiation.
|
||||
|
||||
Regression coverage for the QA bug listed on PR #27039.
|
||||
"""
|
||||
"""Bedrock Invoke /v1/messages forwards ``output_config`` for adaptive Claude models."""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
|
|
@ -628,12 +619,7 @@ def test_bedrock_messages_forwards_output_config():
|
|||
|
||||
|
||||
def test_bedrock_messages_forwards_output_config_with_output_format():
|
||||
"""
|
||||
When both output_config and output_format are present, output_format is
|
||||
converted to inline schema (Bedrock Invoke doesn't accept output_format
|
||||
natively), and output_config is forwarded for Claude 4.6/4.7 adaptive
|
||||
thinking.
|
||||
"""
|
||||
"""``output_config`` is forwarded; ``output_format`` is converted to inline schema."""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
|
|
@ -663,19 +649,7 @@ def test_bedrock_messages_forwards_output_config_with_output_format():
|
|||
|
||||
|
||||
def test_bedrock_messages_forwards_output_config_for_non_adaptive_model():
|
||||
"""
|
||||
``output_config`` is forwarded for non-adaptive models too (e.g. haiku).
|
||||
Bedrock will reject the unsupported key for those models — surfacing the
|
||||
provider error is the correct behavior, since silently swallowing the
|
||||
knob would hide caller bugs.
|
||||
|
||||
Restores coverage previously asserted by
|
||||
``test_bedrock_messages_strips_output_config`` (renamed to the
|
||||
``forwards`` variant on Opus 4.7); the strip path no longer exists,
|
||||
but the non-adaptive pass-through path needs its own explicit test
|
||||
so a future regression that silently re-adds the strip can't sneak
|
||||
through.
|
||||
"""
|
||||
"""``output_config`` is forwarded for non-adaptive models so the provider's error surfaces."""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
|
|
@ -698,14 +672,7 @@ def test_bedrock_messages_forwards_output_config_for_non_adaptive_model():
|
|||
|
||||
|
||||
def test_bedrock_messages_drop_params_strips_output_config_for_pre_4_5():
|
||||
"""
|
||||
``drop_params=True`` is the operator opt-in for "silently fix up"
|
||||
behavior. When a proxy fronts Claude Code at a pre-4.5 Anthropic model
|
||||
(haiku-3, sonnet-3.5, ...) on the /v1/messages route, the client always
|
||||
sends ``output_config.effort`` and the model rejects it. Stripping under
|
||||
``drop_params`` lets those requests succeed; otherwise we forward and
|
||||
surface the model's 400 as designed.
|
||||
"""
|
||||
"""``drop_params=True`` strips ``output_config`` for pre-4.5 Anthropic on /v1/messages."""
|
||||
import litellm
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
|
|
@ -733,8 +700,7 @@ def test_bedrock_messages_drop_params_strips_output_config_for_pre_4_5():
|
|||
|
||||
|
||||
def test_bedrock_messages_drop_params_keeps_output_config_for_4_7():
|
||||
"""``drop_params=True`` must not strip on supporting models — opus-4-7
|
||||
accepts effort, so the client's tier knob has to land on the wire."""
|
||||
"""``drop_params=True`` does not strip on opus-4-7 (supports effort)."""
|
||||
import litellm
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
|
|
@ -775,18 +741,7 @@ def test_bedrock_messages_drop_params_keeps_output_config_for_4_7():
|
|||
def test_bedrock_messages_maps_reasoning_effort_for_adaptive_model(
|
||||
reasoning_effort, expected_effort
|
||||
):
|
||||
"""
|
||||
OpenAI-style ``reasoning_effort`` is mapped to native Anthropic
|
||||
``thinking`` + ``output_config.effort`` on the /v1/messages route so
|
||||
callers can drive adaptive thinking with the same tier vocabulary as
|
||||
the chat completion path. ``reasoning_effort`` itself is popped — the
|
||||
/v1/messages spec doesn't define it and Bedrock rejects unknown
|
||||
top-level fields.
|
||||
|
||||
Closes the QA-sweep gap on PR #27074 where Bedrock Invoke /v1/messages
|
||||
silently dropped ``reasoning_effort`` and every effort tier collapsed
|
||||
to the same behavior.
|
||||
"""
|
||||
"""``reasoning_effort`` maps to ``thinking`` + ``output_config.effort`` on /v1/messages."""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
|
|
@ -810,16 +765,7 @@ def test_bedrock_messages_maps_reasoning_effort_for_adaptive_model(
|
|||
|
||||
|
||||
def test_bedrock_messages_reasoning_effort_on_non_adaptive_uses_thinking_budget():
|
||||
"""
|
||||
For non-adaptive thinking models (e.g. Opus 4.5), ``reasoning_effort``
|
||||
is mapped to ``thinking.type=enabled`` with a budget_tokens value
|
||||
instead of ``output_config.effort``. ``output_config`` is not set on
|
||||
these models because they don't accept it.
|
||||
|
||||
Mirrors ``AnthropicConfig._map_reasoning_effort`` behavior for the
|
||||
non-adaptive branch on Opus 4.5 / earlier Claude 4 models, applied to
|
||||
the /v1/messages route.
|
||||
"""
|
||||
"""Non-adaptive models map ``reasoning_effort`` to ``thinking.budget_tokens``."""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
|
|
@ -847,11 +793,7 @@ def test_bedrock_messages_reasoning_effort_on_non_adaptive_uses_thinking_budget(
|
|||
|
||||
|
||||
def test_bedrock_messages_reasoning_effort_none_clears_thinking():
|
||||
"""
|
||||
``reasoning_effort='none'`` opts out — both ``thinking`` and
|
||||
``output_config`` are cleared so the request goes out without
|
||||
extended thinking. Mirrors the chat completion path's behavior.
|
||||
"""
|
||||
"""``reasoning_effort='none'`` clears both ``thinking`` and ``output_config``."""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
|
|
@ -877,11 +819,7 @@ def test_bedrock_messages_reasoning_effort_none_clears_thinking():
|
|||
|
||||
|
||||
def test_bedrock_messages_invalid_reasoning_effort_raises_400():
|
||||
"""
|
||||
Garbage ``reasoning_effort`` values (``invalid`` / ``disabled`` / ``""``)
|
||||
surface as a clean 400 ``AnthropicError`` instead of silently passing
|
||||
an invalid string through to Bedrock as ``output_config.effort``.
|
||||
"""
|
||||
"""Garbage ``reasoning_effort`` raises AnthropicError (400)."""
|
||||
from litellm.llms.anthropic.common_utils import AnthropicError
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
|
|
@ -903,12 +841,7 @@ def test_bedrock_messages_invalid_reasoning_effort_raises_400():
|
|||
|
||||
|
||||
def test_bedrock_messages_explicit_output_config_wins_over_reasoning_effort():
|
||||
"""
|
||||
Caller-supplied native ``output_config.effort`` wins over the OpenAI
|
||||
``reasoning_effort`` knob. Same precedence as
|
||||
``_translate_legacy_thinking_for_adaptive_model``: explicit native
|
||||
Anthropic params are never overridden by the alias.
|
||||
"""
|
||||
"""Explicit ``output_config.effort`` wins over the ``reasoning_effort`` alias."""
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
|
|
@ -1006,11 +939,6 @@ def test_bedrock_messages_allowlist_filters_anthropic_only_fields():
|
|||
):
|
||||
assert bad not in result, f"{bad} should be stripped by the allowlist"
|
||||
|
||||
# ``output_config`` rides along — Bedrock Invoke accepts it for Claude
|
||||
# 4.6/4.7 adaptive thinking and stripping it silently flattens every
|
||||
# adaptive tier. (Bedrock will reject it for non-adaptive models, which
|
||||
# is the correct behavior — surface the model error rather than swallow
|
||||
# the knob.)
|
||||
assert result.get("output_config") == {"effort": "low"}
|
||||
# Supported fields pass through.
|
||||
assert result["max_tokens"] == 4096
|
||||
|
|
|
|||
|
|
@ -499,14 +499,7 @@ def test_vertex_ai_partner_models_anthropic_remove_prompt_caching_scope_beta_hea
|
|||
|
||||
|
||||
def test_vertex_ai_anthropic_output_config_effort_only_forwarded():
|
||||
"""
|
||||
Vertex AI Claude 4.6 / 4.7 accept ``output_config.effort`` on direct
|
||||
``:rawPredict`` (verified end-to-end against ``us-east5`` for
|
||||
``claude-opus-4-6`` and ``global`` for ``claude-opus-4-7``). The earlier
|
||||
strip silently flattened every adaptive tier on Vertex, so ``low`` /
|
||||
``medium`` / ``high`` / ``xhigh`` / ``max`` all produced identical
|
||||
thinking with no tier differentiation.
|
||||
"""
|
||||
"""Vertex AI Claude 4.6/4.7 accept ``output_config.effort`` on rawPredict."""
|
||||
config = VertexAIAnthropicConfig()
|
||||
|
||||
messages = [{"role": "user", "content": "What is 2+2?"}]
|
||||
|
|
@ -568,15 +561,7 @@ def test_vertex_ai_anthropic_output_config_format_passes_through():
|
|||
|
||||
|
||||
def test_vertex_ai_anthropic_output_config_format_plus_effort_preserved():
|
||||
"""
|
||||
Vertex AI Claude 4.6 / 4.7 accept ``output_config.effort`` on direct
|
||||
``:rawPredict`` (verified end-to-end against ``us-east5`` for
|
||||
``claude-opus-4-6`` and ``global`` for ``claude-opus-4-7``). Since the
|
||||
strip was unjustified, ``effort`` must now ride along with ``format``.
|
||||
|
||||
We use a Claude 4.6 model id here because ``_apply_output_config`` only
|
||||
accepts ``effort`` on adaptive-thinking 4.6/4.7 model ids.
|
||||
"""
|
||||
"""Both ``format`` and ``effort`` ride along on Vertex Claude 4.6/4.7."""
|
||||
config = VertexAIAnthropicConfig()
|
||||
messages = [{"role": "user", "content": "Return a person object."}]
|
||||
|
||||
|
|
@ -625,14 +610,7 @@ def test_vertex_ai_anthropic_output_config_non_dict_dropped():
|
|||
|
||||
|
||||
def test_vertex_ai_anthropic_output_format_and_output_config_effort_preserved():
|
||||
"""
|
||||
Vertex AI Claude 4.6 / 4.7 accept ``output_config.effort`` on direct
|
||||
``:rawPredict`` (verified end-to-end). When both ``output_format`` and
|
||||
``output_config: {effort}`` are present, both must be forwarded — the
|
||||
earlier ``effort`` strip caused silent loss of the requested adaptive
|
||||
thinking tier on Vertex routes (``low``/``medium``/``high``/``xhigh``/``max``
|
||||
all collapsed to identical adaptive thinking with no tier differentiation).
|
||||
"""
|
||||
"""Both ``output_format`` and ``output_config.effort`` are forwarded on Vertex 4.6/4.7."""
|
||||
config = VertexAIAnthropicConfig()
|
||||
messages = [{"role": "user", "content": "Extract structured data"}]
|
||||
|
||||
|
|
|
|||
|
|
@ -14,10 +14,7 @@ from litellm.types.utils import (
|
|||
|
||||
|
||||
class TestXAIReasoningTokenFolding:
|
||||
"""xAI breaks the OpenAI invariant total = prompt + completion by accounting
|
||||
reasoning_tokens separately. ``_fold_reasoning_tokens_into_completion``
|
||||
re-aligns Usage so downstream consumers see the OpenAI shape (o1/o3
|
||||
semantics: completion_tokens includes reasoning)."""
|
||||
"""``_fold_reasoning_tokens_into_completion`` re-aligns xAI Usage to the OpenAI invariant."""
|
||||
|
||||
@staticmethod
|
||||
def _make_response(
|
||||
|
|
@ -42,8 +39,7 @@ class TestXAIReasoningTokenFolding:
|
|||
return response
|
||||
|
||||
def test_should_fold_when_total_explained_by_reasoning_gap(self):
|
||||
# Real xAI live shape captured 2026-05-04: prompt=14, completion=10,
|
||||
# total=336, reasoning=312. 14+10+312 == 336.
|
||||
# xAI live shape: 14 + 10 + 312 == 336.
|
||||
response = self._make_response(
|
||||
prompt_tokens=14,
|
||||
completion_tokens=10,
|
||||
|
|
@ -58,7 +54,6 @@ class TestXAIReasoningTokenFolding:
|
|||
assert usage.total_tokens == usage.prompt_tokens + usage.completion_tokens
|
||||
|
||||
def test_should_not_fold_when_already_normalised(self):
|
||||
# OpenAI-normalised shape: completion already includes reasoning.
|
||||
response = self._make_response(
|
||||
prompt_tokens=14,
|
||||
completion_tokens=322,
|
||||
|
|
@ -68,7 +63,6 @@ class TestXAIReasoningTokenFolding:
|
|||
|
||||
XAIChatConfig._fold_reasoning_tokens_into_completion(response)
|
||||
|
||||
# Idempotent — no double-fold.
|
||||
assert response.usage.completion_tokens == 322
|
||||
|
||||
def test_should_skip_when_no_reasoning_tokens(self):
|
||||
|
|
@ -84,8 +78,7 @@ class TestXAIReasoningTokenFolding:
|
|||
assert response.usage.completion_tokens == 10
|
||||
|
||||
def test_should_skip_when_gap_does_not_match_reasoning(self):
|
||||
# Defensive guard: if xAI ever changes accounting and the gap stops
|
||||
# equalling reasoning_tokens, refuse to fold rather than corrupt.
|
||||
# Refuse to fold if xAI accounting changes (gap != reasoning_tokens).
|
||||
response = self._make_response(
|
||||
prompt_tokens=14,
|
||||
completion_tokens=10,
|
||||
|
|
@ -95,7 +88,6 @@ class TestXAIReasoningTokenFolding:
|
|||
|
||||
XAIChatConfig._fold_reasoning_tokens_into_completion(response)
|
||||
|
||||
# No fold; original values preserved.
|
||||
assert response.usage.completion_tokens == 10
|
||||
assert response.usage.total_tokens == 999
|
||||
|
||||
|
|
|
|||
|
|
@ -107,7 +107,6 @@ class TestXAICostCalculator:
|
|||
|
||||
def test_grok_4_cost_calculation(self):
|
||||
"""Test cost calculation for grok-4 model."""
|
||||
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
|
||||
usage = Usage(
|
||||
prompt_tokens=10,
|
||||
completion_tokens=200,
|
||||
|
|
@ -134,7 +133,6 @@ class TestXAICostCalculator:
|
|||
|
||||
def test_grok_3_fast_beta_cost_calculation(self):
|
||||
"""Test cost calculation for grok-3-fast-beta model."""
|
||||
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
|
||||
usage = Usage(
|
||||
prompt_tokens=20,
|
||||
completion_tokens=300,
|
||||
|
|
@ -176,7 +174,6 @@ class TestXAICostCalculator:
|
|||
|
||||
def test_edge_case_large_reasoning_tokens(self):
|
||||
"""Test cost calculation when reasoning_tokens is larger than completion_tokens."""
|
||||
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
|
||||
usage = Usage(
|
||||
prompt_tokens=12,
|
||||
completion_tokens=50, # Less than reasoning_tokens
|
||||
|
|
@ -204,7 +201,6 @@ class TestXAICostCalculator:
|
|||
def test_tiered_pricing_above_128k_tokens(self):
|
||||
"""Test tiered pricing for tokens above 128k."""
|
||||
# Test with grok-4-fast-reasoning which has tiered pricing
|
||||
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
|
||||
usage = Usage(
|
||||
prompt_tokens=150000, # Above 128k threshold
|
||||
completion_tokens=100000, # Above 128k threshold
|
||||
|
|
@ -234,7 +230,6 @@ class TestXAICostCalculator:
|
|||
def test_tiered_pricing_below_128k_tokens(self):
|
||||
"""Test that regular pricing is used for tokens below 128k threshold."""
|
||||
# Test with grok-4-fast-reasoning which has tiered pricing
|
||||
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
|
||||
usage = Usage(
|
||||
prompt_tokens=100000, # Below 128k threshold
|
||||
completion_tokens=50000,
|
||||
|
|
@ -263,7 +258,6 @@ class TestXAICostCalculator:
|
|||
|
||||
def test_tiered_pricing_grok_4_latest(self):
|
||||
"""Test tiered pricing for grok-4-latest model."""
|
||||
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
|
||||
usage = Usage(
|
||||
prompt_tokens=200000, # Above 128k threshold
|
||||
completion_tokens=100000,
|
||||
|
|
@ -292,7 +286,6 @@ class TestXAICostCalculator:
|
|||
|
||||
def test_tiered_pricing_output_tokens_below_128k(self):
|
||||
"""Test that output tokens get tiered rate when input tokens > 128k, even if output tokens < 128k."""
|
||||
# xAI raw API shape: total_tokens = prompt + visible completion + reasoning
|
||||
usage = Usage(
|
||||
prompt_tokens=150000, # Above 128k threshold
|
||||
completion_tokens=50000, # Below 128k threshold
|
||||
|
|
@ -339,16 +332,7 @@ class TestXAICostCalculator:
|
|||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_already_normalised_usage_does_not_double_count_reasoning(self):
|
||||
"""Cost calc receives Usage post-transformation (OpenAI invariant).
|
||||
|
||||
After XAIChatConfig.transform_response folds reasoning_tokens into
|
||||
completion_tokens, the Usage block satisfies
|
||||
``total_tokens == prompt_tokens + completion_tokens``. Cost calc must
|
||||
detect this and skip the reasoning_tokens add-on, otherwise it
|
||||
double-bills the reasoning tokens.
|
||||
"""
|
||||
# OpenAI-normalised shape: completion_tokens already includes the
|
||||
# 100 reasoning tokens (so 100 visible + 100 reasoning -> 200).
|
||||
"""Cost calc must not double-bill when Usage is already OpenAI-normalised."""
|
||||
usage = Usage(
|
||||
prompt_tokens=12,
|
||||
completion_tokens=200,
|
||||
|
|
@ -364,7 +348,6 @@ class TestXAICostCalculator:
|
|||
|
||||
prompt_cost, completion_cost = cost_per_token(model="grok-3-mini", usage=usage)
|
||||
|
||||
# Bill exactly what completion_tokens reports — no double-add.
|
||||
expected_prompt_cost = 12 * 3e-7
|
||||
expected_completion_cost = 200 * 5e-7
|
||||
|
||||
|
|
|
|||
|
|
@ -2905,9 +2905,9 @@ def test_gemini_embedding_2_ga_in_cost_map():
|
|||
assert info.get("input_cost_per_audio_per_second") == 0.00016
|
||||
assert info.get("input_cost_per_video_per_second") == 0.00079
|
||||
if provider in ("vertex_ai-embedding-models", "vertex_ai"):
|
||||
assert info.get("uses_embed_content") is True, (
|
||||
f"{key} must have uses_embed_content=true for correct Vertex AI routing"
|
||||
)
|
||||
assert (
|
||||
info.get("uses_embed_content") is True
|
||||
), f"{key} must have uses_embed_content=true for correct Vertex AI routing"
|
||||
|
||||
|
||||
def test_gemini_lyria_3_preview_models_in_cost_map():
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue