mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(passthrough): keep adaptive thinking/effort on native /v1/messages path
This commit is contained in:
parent
1cac8bd9ab
commit
96f0ab22a7
2 changed files with 55 additions and 1 deletions
|
|
@ -80,7 +80,18 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig):
|
|||
on Anthropic-only ``cache_control`` extensions (``cache_control.ttl: 1h
|
||||
is not supported``), so unless the provider declares ttl support the
|
||||
hints are reduced to their portable ``{"type": ...}`` core.
|
||||
|
||||
The deployment speaks the Messages API natively, so caller-supplied
|
||||
``thinking``/``output_config``/``temperature`` are forwarded verbatim:
|
||||
the base capability checks resolve the provider as "anthropic" and
|
||||
would otherwise strip modern fields (e.g. adaptive thinking) from a
|
||||
model the cost map does not know (see #40890). Pop them before the
|
||||
base transform so no drop warning fires, restore afterwards.
|
||||
"""
|
||||
stashed: dict = {}
|
||||
for _key in ("thinking", "output_config", "temperature"):
|
||||
if _key in anthropic_messages_optional_request_params:
|
||||
stashed[_key] = anthropic_messages_optional_request_params.pop(_key)
|
||||
request: Final = super().transform_anthropic_messages_request(
|
||||
model=model,
|
||||
messages=messages,
|
||||
|
|
@ -89,8 +100,11 @@ class OpenAILikeAnthropicMessagesConfig(AnthropicMessagesConfig):
|
|||
headers=headers,
|
||||
)
|
||||
if self.supports_cache_control_ttl():
|
||||
request.update(stashed)
|
||||
return request
|
||||
return normalize_cache_control_in_anthropic_payload(request)
|
||||
normalized = normalize_cache_control_in_anthropic_payload(request)
|
||||
normalized.update(stashed)
|
||||
return normalized
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -518,3 +518,43 @@ def test_request_strips_ttl_only_where_the_messages_api_defines_cache_control(co
|
|||
assert tool_result["content"][0]["cache_control"] == {"type": "ephemeral"}
|
||||
assert payload["messages"][1]["content"][1]["cache_control"] == {"type": "ephemeral"}
|
||||
assert payload["messages"][2] == {"role": "user", "content": "a plain string message"}
|
||||
|
||||
|
||||
def test_passthrough_keeps_adaptive_thinking_and_effort(config, caplog):
|
||||
"""Regression #40890: native /v1/messages passthrough must forward
|
||||
thinking/output_config/temperature verbatim, not strip via cost-map checks."""
|
||||
import logging
|
||||
|
||||
optional_params = {
|
||||
"max_tokens": 4096,
|
||||
"thinking": {"type": "adaptive"},
|
||||
"output_config": {"effort": "high"},
|
||||
"temperature": 0.5,
|
||||
}
|
||||
with caplog.at_level(logging.WARNING):
|
||||
payload = config.transform_anthropic_messages_request(
|
||||
model="my-model",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
anthropic_messages_optional_request_params=dict(optional_params),
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
assert payload["thinking"] == {"type": "adaptive"}
|
||||
assert payload["output_config"] == {"effort": "high"}
|
||||
assert payload["temperature"] == 0.5
|
||||
assert "Dropping adaptive" not in caplog.text
|
||||
|
||||
|
||||
def test_passthrough_keeps_legacy_thinking_shape(config):
|
||||
optional_params = {
|
||||
"max_tokens": 4096,
|
||||
"thinking": {"type": "enabled", "budget_tokens": 1024},
|
||||
}
|
||||
payload = config.transform_anthropic_messages_request(
|
||||
model="my-model",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
anthropic_messages_optional_request_params=dict(optional_params),
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
assert payload["thinking"] == {"type": "enabled", "budget_tokens": 1024}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue