From 96f0ab22a728d82f7a6313a2fc38c95510ab0aef Mon Sep 17 00:00:00 2001 From: ege-arhan Date: Tue, 15 Sep 2026 14:23:18 +0000 Subject: [PATCH 1/2] fix(passthrough): keep adaptive thinking/effort on native /v1/messages path --- .../openai_like/messages/transformation.py | 16 +++++++- ..._like_anthropic_messages_transformation.py | 40 +++++++++++++++++++ 2 files changed, 55 insertions(+), 1 deletion(-) diff --git a/litellm/llms/openai_like/messages/transformation.py b/litellm/llms/openai_like/messages/transformation.py index bae190c88c0..67e2c2e28be 100644 --- a/litellm/llms/openai_like/messages/transformation.py +++ b/litellm/llms/openai_like/messages/transformation.py @@ -80,7 +80,18 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig): on Anthropic-only ``cache_control`` extensions (``cache_control.ttl: 1h is not supported``), so unless the provider declares ttl support the hints are reduced to their portable ``{"type": ...}`` core. + + The deployment speaks the Messages API natively, so caller-supplied + ``thinking``/``output_config``/``temperature`` are forwarded verbatim: + the base capability checks resolve the provider as "anthropic" and + would otherwise strip modern fields (e.g. adaptive thinking) from a + model the cost map does not know (see #40890). Pop them before the + base transform so no drop warning fires, restore afterwards. """ + stashed: dict = {} + for _key in ("thinking", "output_config", "temperature"): + if _key in anthropic_messages_optional_request_params: + stashed[_key] = anthropic_messages_optional_request_params.pop(_key) request: Final = super().transform_anthropic_messages_request( model=model, messages=messages, @@ -89,8 +100,11 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig): headers=headers, ) if self.supports_cache_control_ttl(): + request.update(stashed) return request - return normalize_cache_control_in_anthropic_payload(request) + normalized = normalize_cache_control_in_anthropic_payload(request) + normalized.update(stashed) + return normalized def get_complete_url( self, diff --git a/tests/unit/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py b/tests/unit/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py index 67a56fdcd79..6611f8320c3 100644 --- a/tests/unit/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py +++ b/tests/unit/llms/openai_like/messages/test_openai_like_anthropic_messages_transformation.py @@ -518,3 +518,43 @@ def test_request_strips_ttl_only_where_the_messages_api_defines_cache_control(co assert tool_result["content"][0]["cache_control"] == {"type": "ephemeral"} assert payload["messages"][1]["content"][1]["cache_control"] == {"type": "ephemeral"} assert payload["messages"][2] == {"role": "user", "content": "a plain string message"} + + +def test_passthrough_keeps_adaptive_thinking_and_effort(config, caplog): + """Regression #40890: native /v1/messages passthrough must forward + thinking/output_config/temperature verbatim, not strip via cost-map checks.""" + import logging + + optional_params = { + "max_tokens": 4096, + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "high"}, + "temperature": 0.5, + } + with caplog.at_level(logging.WARNING): + payload = config.transform_anthropic_messages_request( + model="my-model", + messages=[{"role": "user", "content": "hi"}], + anthropic_messages_optional_request_params=dict(optional_params), + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + assert payload["thinking"] == {"type": "adaptive"} + assert payload["output_config"] == {"effort": "high"} + assert payload["temperature"] == 0.5 + assert "Dropping adaptive" not in caplog.text + + +def test_passthrough_keeps_legacy_thinking_shape(config): + optional_params = { + "max_tokens": 4096, + "thinking": {"type": "enabled", "budget_tokens": 1024}, + } + payload = config.transform_anthropic_messages_request( + model="my-model", + messages=[{"role": "user", "content": "hi"}], + anthropic_messages_optional_request_params=dict(optional_params), + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + assert payload["thinking"] == {"type": "enabled", "budget_tokens": 1024} From 6f9c6143d691c910a635be068c713e4b67f40efb Mon Sep 17 00:00:00 2001 From: ege-arhan Date: Mon, 21 Sep 2026 09:35:44 +0000 Subject: [PATCH 2/2] chore: justify LIT002 stash dict with mutable-ok reason (gate delta) --- litellm/llms/openai_like/messages/transformation.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/llms/openai_like/messages/transformation.py b/litellm/llms/openai_like/messages/transformation.py index 67e2c2e28be..78429738ce7 100644 --- a/litellm/llms/openai_like/messages/transformation.py +++ b/litellm/llms/openai_like/messages/transformation.py @@ -88,7 +88,7 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig): model the cost map does not know (see #40890). Pop them before the base transform so no drop warning fires, restore afterwards. """ - stashed: dict = {} + stashed: dict = {} # mutable-ok: transient pop/restore stash, never escapes the call for _key in ("thinking", "output_config", "temperature"): if _key in anthropic_messages_optional_request_params: stashed[_key] = anthropic_messages_optional_request_params.pop(_key)