fix(anthropic): drop unsupported speed param with drop_params (#31152)

* fix(anthropic): drop unsupported speed param with drop_params

Anthropic fast mode (speed) is Opus 4.6/4.7/4.8 on the direct API only.
Strip speed when the model map lacks supports_speed and drop_params is set,
for both chat completions and /v1/messages passthrough.

Co-authored-by: Cursor <cursoragent@cursor.com>

* fix(ci): allow supports_speed in model map schema

The new supports_speed flag on Opus entries must pass JSON schema
validation in test_aaamodel_prices_and_context_window_json_is_valid.

Co-authored-by: Cursor <cursoragent@cursor.com>

* fix(review): raise on unsupported speed without drop_params

Passthrough /v1/messages now raises UnsupportedParamsError when speed
is unsupported and drop_params is false. Emit drop warning from
map_openai_params when speed is silently skipped.

Co-authored-by: Cursor <cursoragent@cursor.com>

* fix(anthropic): gate speed param by routed provider, not just model id

Vertex, Azure, and Bedrock reuse the shared Anthropic transform and strip
their provider prefix first, so a bare `claude-opus-4-8` resolved to the
direct-API model-map entry (`supports_speed: true`) and forwarded `speed`
upstream, producing the same 400 that drop_params is meant to prevent.

Gate fast mode on `custom_llm_provider == "anthropic"` so it stays on the
direct Anthropic API across both the chat completions and `/v1/messages`
passthrough paths, and collapse the duplicated drop/raise logic in
map_openai_params into the shared `_maybe_drop_speed_param` helper.

---------

Co-authored-by: Cursor <cursoragent@cursor.com>
Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
This commit is contained in:
Krrish Dholakia 2026-06-23 22:22:49 -07:00 • committed by GitHub
parent c0a146929c
commit d0706c17fe
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
10 changed files with 404 additions and 12 deletions

View file

@ -229,6 +229,11 @@ DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING = (
"Sonnet 4.6+, and Mythos Preview."
)
DROP_UNSUPPORTED_SPEED_WARNING = (
"Dropping unsupported `speed` for model=%s "
"(drop_params=True). Fast mode is only supported on select Opus models."
)
class AnthropicConfig(AnthropicModelInfo, BaseConfig):
"""
@ -374,6 +379,51 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
for level in ("low", "minimal", "medium", "high", "xhigh", "max")
)
@staticmethod
def _model_supports_speed_param(
model: str, custom_llm_provider: Optional[str] = None
) -> bool:
"""Whether the model accepts Anthropic's ``speed`` parameter (fast mode).
Fast mode is direct Anthropic API-only (not Bedrock, Vertex, or Azure).
Those providers strip their prefix before this shared transform runs, so a
bare ``claude-opus-4-8`` would otherwise resolve to the direct-API entry;
the routed provider is checked explicitly to keep them out.
"""
if custom_llm_provider is not None and custom_llm_provider != "anthropic":
return False
return (
AnthropicModelInfo._get_exact_model_capability(model, "supports_speed")
is True
)
@staticmethod
def _maybe_drop_speed_param(
model: str,
optional_params: dict,
drop_params: bool,
custom_llm_provider: Optional[str] = None,
) -> None:
if "speed" not in optional_params:
return
if AnthropicConfig._model_supports_speed_param(model, custom_llm_provider):
return
if not (litellm.drop_params or drop_params):
speed_value = optional_params.get("speed")
raise litellm.utils.UnsupportedParamsError(
message=(
f"{model} does not support speed={speed_value!r}. "
"To drop unsupported params, set "
"`litellm.drop_params = True`."
),
status_code=400,
)
litellm.verbose_logger.warning(
DROP_UNSUPPORTED_SPEED_WARNING,
model,
)
optional_params.pop("speed", None)
@staticmethod
def _raise_invalid_reasoning_effort(
model: str, value: Any, llm_provider: str
@ -1569,8 +1619,13 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
anthropic_context_management
)
elif param == "speed" and isinstance(value, str):
# Pass through Anthropic-specific speed parameter for fast mode
optional_params["speed"] = value
AnthropicConfig._maybe_drop_speed_param(
model=model,
optional_params=optional_params,
drop_params=drop_params,
custom_llm_provider=self.custom_llm_provider,
)
elif param == "cache_control" and isinstance(value, dict):
# Pass through top-level cache_control for automatic prompt caching
optional_params["cache_control"] = value
@ -1875,6 +1930,14 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
"has no thinking_blocks. The model won't use extended thinking for this turn."
)
AnthropicConfig._maybe_drop_speed_param(
model=model,
optional_params=optional_params,
drop_params=litellm.drop_params
or litellm_params.get("drop_params") is True,
custom_llm_provider=self.custom_llm_provider,
)
headers = self.update_headers_with_optional_anthropic_beta(
headers=headers, optional_params=optional_params
)

View file

@ -367,6 +367,16 @@ class AnthropicModelInfo(BaseLLMModelInfo):
pass
return None
@staticmethod
def _get_exact_model_capability(model: str, key: str) -> Optional[bool]:
"""Read boolean capability ``key`` from the exact model-map entry only.
Unlike ``_get_model_capability``, does not walk stripped provider aliases.
Use when a feature is tied to a specific host (e.g. Anthropic API fast mode).
"""
value = litellm.model_cost.get(model, {}).get(key)
return value if isinstance(value, bool) else None
@staticmethod
def _supports_model_capability(model: str, key: str) -> bool:
"""Check a boolean capability ``key`` in the model map.

View file

@ -507,7 +507,10 @@ def anthropic_messages_handler(
local_vars.update(kwargs)
anthropic_messages_optional_request_params = (
AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param(
params=local_vars
params=local_vars,
model=model,
drop_params=litellm_params.get("drop_params") is True,
custom_llm_provider=custom_llm_provider,
)
)
if is_reasoning_auto_summary_enabled():

View file

@ -23,12 +23,19 @@ class AnthropicMessagesRequestUtils:
@staticmethod
def get_requested_anthropic_messages_optional_param(
params: Dict[str, Any],
*,
model: str | None = None,
drop_params: bool = False,
custom_llm_provider: str | None = None,
) -> AnthropicMessagesRequestOptionalParams:
"""
Filter parameters to only include those defined in AnthropicMessagesRequestOptionalParams.
Args:
params: Dictionary of parameters to filter
model: Resolved model id; when set, unsupported params may be dropped
drop_params: Per-request drop_params flag (also respects litellm.drop_params)
custom_llm_provider: Routed provider; fast mode is gated to direct Anthropic
Returns:
AnthropicMessagesRequestOptionalParams instance with only the valid parameters
@ -37,6 +44,15 @@ class AnthropicMessagesRequestUtils:
filtered_params = {
k: v for k, v in params.items() if k in valid_keys and v is not None
}
if model is not None:
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
AnthropicConfig._maybe_drop_speed_param(
model=model,
optional_params=filtered_params,
drop_params=drop_params,
custom_llm_provider=custom_llm_provider,
)
return cast(AnthropicMessagesRequestOptionalParams, filtered_params)

View file

@ -10443,7 +10443,8 @@
"fast": 6.0
},
"supports_output_config": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_speed": true
},
"claude-opus-4-6-20260205": {
"cache_creation_input_token_cost": 6.25e-06,
@ -10476,7 +10477,8 @@
"fast": 6.0
},
"supports_max_reasoning_effort": true,
"supports_output_config": true
"supports_output_config": true,
"supports_speed": true
},
"claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25e-06,
@ -10511,7 +10513,8 @@
"us": 1.1,
"fast": 6.0
},
"supports_output_config": true
"supports_output_config": true,
"supports_speed": true
},
"claude-opus-4-7-20260416": {
"cache_creation_input_token_cost": 6.25e-06,
@ -10546,7 +10549,8 @@
"us": 1.1,
"fast": 6.0
},
"supports_output_config": true
"supports_output_config": true,
"supports_speed": true
},
"claude-fable-5": {
"cache_creation_input_token_cost": 1.25e-05,
@ -10615,7 +10619,8 @@
"us": 1.1,
"fast": 2.0
},
"supports_output_config": true
"supports_output_config": true,
"supports_speed": true
},
"claude-sonnet-4-20250514": {
"deprecation_date": "2026-05-14",

View file

@ -10443,7 +10443,8 @@
"fast": 6.0
},
"supports_output_config": true,
"supports_max_reasoning_effort": true
"supports_max_reasoning_effort": true,
"supports_speed": true
},
"claude-opus-4-6-20260205": {
"cache_creation_input_token_cost": 6.25e-06,
@ -10476,7 +10477,8 @@
"fast": 6.0
},
"supports_max_reasoning_effort": true,
"supports_output_config": true
"supports_output_config": true,
"supports_speed": true
},
"claude-opus-4-7": {
"cache_creation_input_token_cost": 6.25e-06,
@ -10511,7 +10513,8 @@
"us": 1.1,
"fast": 6.0
},
"supports_output_config": true
"supports_output_config": true,
"supports_speed": true
},
"claude-opus-4-7-20260416": {
"cache_creation_input_token_cost": 6.25e-06,
@ -10546,7 +10549,8 @@
"us": 1.1,
"fast": 6.0
},
"supports_output_config": true
"supports_output_config": true,
"supports_speed": true
},
"claude-fable-5": {
"cache_creation_input_token_cost": 1.25e-05,
@ -10615,7 +10619,8 @@
"us": 1.1,
"fast": 2.0
},
"supports_output_config": true
"supports_output_config": true,
"supports_speed": true
},
"claude-sonnet-4-20250514": {
"deprecation_date": "2026-05-14",

View file

@ -1886,6 +1886,89 @@ def test_anthropic_model_supports_effort_param_rejects_non_supporting_models(mod
assert AnthropicConfig._model_supports_effort_param(model) is False
@pytest.mark.parametrize(
"model",
[
"claude-opus-4-6",
"claude-opus-4-7",
"claude-opus-4-8",
"claude-opus-4-6-20260205",
"claude-opus-4-7-20260416",
],
)
def test_anthropic_model_supports_speed_param_recognizes_supporting_models(model):
assert AnthropicConfig._model_supports_speed_param(model) is True
@pytest.mark.parametrize(
"model",
[
"claude-sonnet-4-6",
"claude-fable-5",
"claude-3-haiku-20240307",
"vertex_ai/claude-opus-4-8",
"azure_ai/claude-opus-4-8",
"anthropic.claude-opus-4-8",
],
)
def test_anthropic_model_supports_speed_param_rejects_non_supporting_models(model):
assert AnthropicConfig._model_supports_speed_param(model) is False
@pytest.mark.parametrize("custom_llm_provider", ["vertex_ai", "azure_ai", "bedrock"])
def test_anthropic_model_supports_speed_param_rejects_non_anthropic_providers(
custom_llm_provider,
):
"""Fast mode is direct-Anthropic-only. Vertex/Azure/Bedrock strip their prefix
before the shared transform runs, so the bare Opus id must still be rejected."""
assert (
AnthropicConfig._model_supports_speed_param(
"claude-opus-4-8", custom_llm_provider
)
is False
)
assert (
AnthropicConfig._model_supports_speed_param("claude-opus-4-8", "anthropic")
is True
)
def test_vertex_anthropic_drops_speed_for_opus_with_drop_params(monkeypatch):
"""Regression: vertex_ai Opus must drop ``speed`` even though the prefix-stripped
``claude-opus-4-8`` maps to a fast-mode-capable direct-Anthropic entry."""
from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import (
VertexAIAnthropicConfig,
)
monkeypatch.setattr(litellm, "drop_params", True)
result = VertexAIAnthropicConfig().transform_request(
model="claude-opus-4-8",
messages=[{"role": "user", "content": "Hello"}],
optional_params={"speed": "fast", "max_tokens": 1024},
litellm_params={},
headers={},
)
assert "speed" not in result
def test_vertex_anthropic_raises_on_speed_without_drop_params(monkeypatch):
"""Regression: vertex_ai Opus raises rather than forwarding an unsupported
``speed`` when neither global nor per-request drop_params is set."""
from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import (
VertexAIAnthropicConfig,
)
monkeypatch.setattr(litellm, "drop_params", False)
with pytest.raises(litellm.utils.UnsupportedParamsError, match="drop_params"):
VertexAIAnthropicConfig().map_openai_params(
non_default_params={"speed": "fast"},
optional_params={},
model="claude-opus-4-8",
drop_params=False,
)
def test_translate_system_message_skips_empty_string_content():
"""
Test that translate_system_message skips system messages with empty string content.
@ -3766,6 +3849,61 @@ def test_fast_mode_parameter_mapping():
assert result["speed"] == "fast"
def test_anthropic_drop_params_strips_speed_for_unsupported_models():
"""``drop_params=True`` strips unsupported ``speed`` for non-Opus models."""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
original = litellm.drop_params
litellm.drop_params = True
try:
result = config.transform_request(
model="claude-sonnet-4-6",
messages=messages,
optional_params={"speed": "fast", "max_tokens": 1024},
litellm_params={},
headers={},
)
finally:
litellm.drop_params = original
assert "speed" not in result
def test_anthropic_drop_params_keeps_speed_for_supporting_models():
"""``drop_params=True`` must not strip ``speed`` on Opus fast-mode models."""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
original = litellm.drop_params
litellm.drop_params = True
try:
result = config.transform_request(
model="claude-opus-4-6",
messages=messages,
optional_params={"speed": "fast", "max_tokens": 1024},
litellm_params={},
headers={},
)
finally:
litellm.drop_params = original
assert result.get("speed") == "fast"
def test_speed_raises_clean_error_without_drop_params(monkeypatch):
monkeypatch.setattr(litellm, "drop_params", False)
config = AnthropicConfig()
with pytest.raises(litellm.utils.UnsupportedParamsError, match="drop_params"):
config.map_openai_params(
non_default_params={"speed": "fast"},
optional_params={},
model="claude-sonnet-4-6",
drop_params=False,
)
def test_map_openai_params_max_tokens_normalized_to_int():
"""
Test that map_openai_params normalizes max_tokens to an integer (e.g. 0.7 -> 1).

View file

@ -0,0 +1,117 @@
import litellm
import pytest
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
from litellm.llms.anthropic.experimental_pass_through.messages.utils import (
AnthropicMessagesRequestUtils,
)
def test_messages_drop_params_strips_speed_for_unsupported_models():
original = litellm.drop_params
litellm.drop_params = True
try:
optional_params = AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param(
params={
"max_tokens": 1024,
"speed": "fast",
"messages": [{"role": "user", "content": "Hello"}],
},
model="claude-sonnet-4-6",
drop_params=False,
)
config = AnthropicMessagesConfig()
headers, _ = config.validate_anthropic_messages_environment(
headers={},
model="claude-sonnet-4-6",
messages=[{"role": "user", "content": "Hello"}],
optional_params=dict(optional_params),
litellm_params={},
)
result = config.transform_anthropic_messages_request(
model="claude-sonnet-4-6",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=dict(optional_params),
litellm_params={},
headers=headers,
)
finally:
litellm.drop_params = original
assert "speed" not in optional_params
assert "speed" not in result
assert "fast-mode-2026-02-01" not in headers.get("anthropic-beta", "")
def test_messages_drop_params_keeps_speed_for_supporting_models():
original = litellm.drop_params
litellm.drop_params = True
try:
optional_params = AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param(
params={"max_tokens": 1024, "speed": "fast"},
model="claude-opus-4-6",
drop_params=False,
)
config = AnthropicMessagesConfig()
headers, _ = config.validate_anthropic_messages_environment(
headers={},
model="claude-opus-4-6",
messages=[{"role": "user", "content": "Hello"}],
optional_params=dict(optional_params),
litellm_params={},
)
result = config.transform_anthropic_messages_request(
model="claude-opus-4-6",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params=dict(optional_params),
litellm_params={},
headers=headers,
)
finally:
litellm.drop_params = original
assert optional_params.get("speed") == "fast"
assert result.get("speed") == "fast"
assert "fast-mode-2026-02-01" in headers.get("anthropic-beta", "")
def test_messages_raises_when_speed_unsupported_and_drop_params_false(monkeypatch):
monkeypatch.setattr(litellm, "drop_params", False)
with pytest.raises(litellm.utils.UnsupportedParamsError, match="drop_params"):
AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param(
params={"max_tokens": 1024, "speed": "fast"},
model="claude-sonnet-4-6",
drop_params=False,
)
def test_messages_drops_speed_for_vertex_opus_with_drop_params(monkeypatch):
"""Regression: a vertex_ai Opus passthrough must drop ``speed`` even though the
prefix-stripped model id maps to a fast-mode-capable direct-Anthropic entry."""
monkeypatch.setattr(litellm, "drop_params", True)
optional_params = (
AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param(
params={"max_tokens": 1024, "speed": "fast"},
model="claude-opus-4-8",
drop_params=False,
custom_llm_provider="vertex_ai",
)
)
assert "speed" not in optional_params
def test_messages_raises_for_vertex_opus_without_drop_params(monkeypatch):
"""Regression: vertex_ai Opus passthrough raises rather than forwarding an
unsupported ``speed`` when drop_params is unset."""
monkeypatch.setattr(litellm, "drop_params", False)
with pytest.raises(litellm.utils.UnsupportedParamsError, match="drop_params"):
AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param(
params={"max_tokens": 1024, "speed": "fast"},
model="claude-opus-4-8",
drop_params=False,
custom_llm_provider="vertex_ai",
)

View file

@ -6,6 +6,7 @@ Regression tests for the /v1/messages request-parse fast paths:
while resolving the (static) type hints only once per process.
"""
import litellm
from litellm.llms.anthropic.experimental_pass_through.messages.utils import (
AnthropicMessagesRequestUtils,
_anthropic_messages_optional_param_keys,
@ -54,3 +55,36 @@ def test_empty_params():
)
== {}
)
def test_drop_params_strips_speed_for_unsupported_model():
original = litellm.drop_params
litellm.drop_params = True
try:
result = (
AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param(
params={"speed": "fast", "temperature": 0.5},
model="claude-sonnet-4-6",
)
)
finally:
litellm.drop_params = original
assert result == {"temperature": 0.5}
assert "speed" not in result
def test_drop_params_keeps_speed_for_supporting_model():
original = litellm.drop_params
litellm.drop_params = True
try:
result = (
AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param(
params={"speed": "fast"},
model="claude-opus-4-6",
)
)
finally:
litellm.drop_params = original
assert result == {"speed": "fast"}

View file

@ -865,6 +865,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_service_tier": {"type": "boolean"},
"supports_preset": {"type": "boolean"},
"supports_output_config": {"type": "boolean"},
"supports_speed": {"type": "boolean"},
"bedrock_output_config_effort_ceiling": {
"type": "string",
"enum": ["low", "medium", "high", "max", "xhigh"],