fix(passthrough): keep adaptive thinking/effort on native /v1/messages path

This commit is contained in:
ege-arhan 2026-09-15 14:23:18 +00:00
parent 1cac8bd9ab
commit 96f0ab22a7
2 changed files with 55 additions and 1 deletions

View file

@ -80,7 +80,18 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig):
on Anthropic-only ``cache_control`` extensions (``cache_control.ttl: 1h
is not supported``), so unless the provider declares ttl support the
hints are reduced to their portable ``{"type": ...}`` core.
The deployment speaks the Messages API natively, so caller-supplied
``thinking``/``output_config``/``temperature`` are forwarded verbatim:
the base capability checks resolve the provider as "anthropic" and
would otherwise strip modern fields (e.g. adaptive thinking) from a
model the cost map does not know (see #40890). Pop them before the
base transform so no drop warning fires, restore afterwards.
"""
stashed: dict = {}
for _key in ("thinking", "output_config", "temperature"):
if _key in anthropic_messages_optional_request_params:
stashed[_key] = anthropic_messages_optional_request_params.pop(_key)
request: Final = super().transform_anthropic_messages_request(
model=model,
messages=messages,
@ -89,8 +100,11 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig):
headers=headers,
)
if self.supports_cache_control_ttl():
request.update(stashed)
return request
return normalize_cache_control_in_anthropic_payload(request)
normalized = normalize_cache_control_in_anthropic_payload(request)
normalized.update(stashed)
return normalized
def get_complete_url(
self,

View file

@ -518,3 +518,43 @@ def test_request_strips_ttl_only_where_the_messages_api_defines_cache_control(co
assert tool_result["content"][0]["cache_control"] == {"type": "ephemeral"}
assert payload["messages"][1]["content"][1]["cache_control"] == {"type": "ephemeral"}
assert payload["messages"][2] == {"role": "user", "content": "a plain string message"}
def test_passthrough_keeps_adaptive_thinking_and_effort(config, caplog):
"""Regression #40890: native /v1/messages passthrough must forward
thinking/output_config/temperature verbatim, not strip via cost-map checks."""
import logging
optional_params = {
"max_tokens": 4096,
"thinking": {"type": "adaptive"},
"output_config": {"effort": "high"},
"temperature": 0.5,
}
with caplog.at_level(logging.WARNING):
payload = config.transform_anthropic_messages_request(
model="my-model",
messages=[{"role": "user", "content": "hi"}],
anthropic_messages_optional_request_params=dict(optional_params),
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert payload["thinking"] == {"type": "adaptive"}
assert payload["output_config"] == {"effort": "high"}
assert payload["temperature"] == 0.5
assert "Dropping adaptive" not in caplog.text
def test_passthrough_keeps_legacy_thinking_shape(config):
optional_params = {
"max_tokens": 4096,
"thinking": {"type": "enabled", "budget_tokens": 1024},
}
payload = config.transform_anthropic_messages_request(
model="my-model",
messages=[{"role": "user", "content": "hi"}],
anthropic_messages_optional_request_params=dict(optional_params),
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert payload["thinking"] == {"type": "enabled", "budget_tokens": 1024}