diff --git a/litellm/llms/openai_like/messages/transformation.py b/litellm/llms/openai_like/messages/transformation.py index 2e9a300e2fd..50ffdde1bb0 100644 --- a/litellm/llms/openai_like/messages/transformation.py +++ b/litellm/llms/openai_like/messages/transformation.py @@ -80,7 +80,18 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig): on Anthropic-only ``cache_control`` extensions (``cache_control.ttl: 1h is not supported``), so unless the provider declares ttl support the hints are reduced to their portable ``{"type": ...}`` core. + + The deployment speaks the Messages API natively, so caller-supplied + ``thinking``/``output_config``/``temperature`` are forwarded verbatim: + the base capability checks resolve the provider as "anthropic" and + would otherwise strip modern fields (e.g. adaptive thinking) from a + model the cost map does not know (see #40890). Pop them before the + base transform so no drop warning fires, restore afterwards. """ + stashed: dict = {} # mutable-ok: transient pop/restore stash, never escapes the call + for _key in ("thinking", "output_config", "temperature"): + if _key in anthropic_messages_optional_request_params: + stashed[_key] = anthropic_messages_optional_request_params.pop(_key) request: Final = super().transform_anthropic_messages_request( model=model, messages=messages, @@ -89,8 +100,11 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig): headers=headers, ) if self.supports_cache_control_ttl(): + request.update(stashed) return request - return normalize_cache_control_in_anthropic_payload(request) + normalized = normalize_cache_control_in_anthropic_payload(request) + normalized.update(stashed) + return normalized def get_complete_url( self, diff --git a/tests/unit/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py b/tests/unit/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py index 9167853bf64..755f7f6ae9f 100644 --- a/tests/unit/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py +++ b/tests/unit/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py @@ -516,3 +516,43 @@ def test_request_strips_ttl_only_where_the_messages_api_defines_cache_control(co assert tool_result["content"][0]["cache_control"] == {"type": "ephemeral"} assert payload["messages"][1]["content"][1]["cache_control"] == {"type": "ephemeral"} assert payload["messages"][2] == {"role": "user", "content": "a plain string message"} + + +def test_passthrough_keeps_adaptive_thinking_and_effort(config, caplog): + """Regression #40890: native /v1/messages passthrough must forward + thinking/output_config/temperature verbatim, not strip via cost-map checks.""" + import logging + + optional_params = { + "max_tokens": 4096, + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "high"}, + "temperature": 0.5, + } + with caplog.at_level(logging.WARNING): + payload = config.transform_anthropic_messages_request( + model="my-model", + messages=[{"role": "user", "content": "hi"}], + anthropic_messages_optional_request_params=dict(optional_params), + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + assert payload["thinking"] == {"type": "adaptive"} + assert payload["output_config"] == {"effort": "high"} + assert payload["temperature"] == 0.5 + assert "Dropping adaptive" not in caplog.text + + +def test_passthrough_keeps_legacy_thinking_shape(config): + optional_params = { + "max_tokens": 4096, + "thinking": {"type": "enabled", "budget_tokens": 1024}, + } + payload = config.transform_anthropic_messages_request( + model="my-model", + messages=[{"role": "user", "content": "hi"}], + anthropic_messages_optional_request_params=dict(optional_params), + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + assert payload["thinking"] == {"type": "enabled", "budget_tokens": 1024}