mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(anthropic): keep adaptive thinking when modify_params drops missing thinking blocks (#43531)
The modify_params guard (#14194 / #18926) drops the thinking param whenever the last assistant message with tool_calls has no thinking_blocks. For adaptive-thinking models (Claude 4.6+ / Opus 5.x, thinking={"type": "adaptive"}) that guard fires on ordinary traffic and Anthropic does not need it: adaptive thinking may legitimately decide not to think on a trivial first tool call. Dropping it made effort-configured models stream no reasoning (no reasoning_content deltas, empty signed thinking block), and at high effort the silence could exceed intermediary timeouts. Skip the drop for models where AnthropicConfig._is_adaptive_thinking_model() is true; legacy budget-based thinking keeps the existing behavior. Tests: adaptive model keeps thinking/output_config; legacy model still drops.
This commit is contained in:
parent
74cad08997
commit
aebb107803
2 changed files with 97 additions and 0 deletions
|
|
@ -1920,9 +1920,18 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
# If any message has thinking_blocks, we must keep thinking enabled, otherwise
|
||||
# Anthropic errors with: "When thinking is disabled, an assistant message cannot contain thinking"
|
||||
# Related issue: https://github.com/BerriAI/litellm/issues/18926
|
||||
#
|
||||
# Adaptive-thinking models (Claude 4.6+ / Opus 5.x) are exempt: with
|
||||
# thinking={"type": "adaptive"} Anthropic accepts a tool_use-only prior
|
||||
# turn (adaptive thinking may legitimately decide not to think on a
|
||||
# trivial tool call), so the guard applies only to legacy budget-based
|
||||
# thinking. Dropping it there breaks streaming for effort-configured
|
||||
# models (no reasoning_content deltas until first text/tool token).
|
||||
# Related issue: https://github.com/BerriAI/litellm/issues/43531
|
||||
if (
|
||||
optional_params.get("thinking") is not None
|
||||
and messages is not None
|
||||
and not AnthropicConfig._is_adaptive_thinking_model(model, self._resolved_provider)
|
||||
and last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
|
||||
and not any_assistant_message_has_thinking_blocks(messages)
|
||||
):
|
||||
|
|
|
|||
|
|
@ -6798,3 +6798,91 @@ def test_chat_dummy_tool_result_for_an_orphaned_tool_call_replays_a_byte_identic
|
|||
_assert_prefix_stable(requests)
|
||||
assert [m["role"] for m in requests[0]["messages"]] == ["user", "assistant", "user"]
|
||||
assert requests[0]["messages"][2]["content"][0]["type"] == "tool_result"
|
||||
|
||||
|
||||
def test_adaptive_thinking_kept_for_adaptive_model_when_tool_calls_missing_thinking_blocks():
|
||||
"""
|
||||
modify_params must NOT drop thinking for adaptive-thinking models
|
||||
(Claude 4.6+ / Opus 5.x): Anthropic accepts a tool_use-only prior turn
|
||||
with thinking={"type": "adaptive"}, and dropping it breaks reasoning
|
||||
streaming (no reasoning_content deltas until first text/tool token).
|
||||
|
||||
Related issue: https://github.com/BerriAI/litellm/issues/43531
|
||||
"""
|
||||
import litellm
|
||||
|
||||
original_modify_params = litellm.modify_params
|
||||
litellm.modify_params = True
|
||||
try:
|
||||
messages = [
|
||||
{"role": "user", "content": "Run `echo seed` with bash, then reason."},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "toolu_1",
|
||||
"type": "function",
|
||||
"function": {"name": "bash", "arguments": '{"command":"echo seed"}'},
|
||||
}
|
||||
],
|
||||
},
|
||||
{"role": "tool", "tool_call_id": "toolu_1", "content": "seed"},
|
||||
]
|
||||
optional_params = {
|
||||
"thinking": {"type": "adaptive", "display": "summarized"},
|
||||
"output_config": {"effort": "xhigh"},
|
||||
}
|
||||
transformed = AnthropicConfig().transform_request(
|
||||
model="claude-opus-5-5",
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert "thinking" in transformed
|
||||
assert transformed["thinking"] == {"type": "adaptive", "display": "summarized"}
|
||||
assert transformed["output_config"] == {"effort": "xhigh"}
|
||||
finally:
|
||||
litellm.modify_params = original_modify_params
|
||||
|
||||
|
||||
def test_legacy_budget_thinking_still_dropped_when_tool_calls_missing_thinking_blocks():
|
||||
"""
|
||||
The original guard (#14194 / #18926) must keep working for legacy
|
||||
budget-based thinking: a tool_use-only prior turn is rejected by
|
||||
Anthropic unless the request pairs it with thinking blocks.
|
||||
|
||||
Related issue: https://github.com/BerriAI/litellm/issues/43531
|
||||
"""
|
||||
import litellm
|
||||
|
||||
original_modify_params = litellm.modify_params
|
||||
litellm.modify_params = True
|
||||
try:
|
||||
messages = [
|
||||
{"role": "user", "content": "Run `echo seed` with bash, then reason."},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "toolu_1",
|
||||
"type": "function",
|
||||
"function": {"name": "bash", "arguments": '{"command":"echo seed"}'},
|
||||
}
|
||||
],
|
||||
},
|
||||
{"role": "tool", "tool_call_id": "toolu_1", "content": "seed"},
|
||||
]
|
||||
optional_params = {"thinking": {"type": "enabled", "budget_tokens": 1024}}
|
||||
transformed = AnthropicConfig().transform_request(
|
||||
model="claude-3-5-sonnet-20241022",
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert "thinking" not in transformed
|
||||
finally:
|
||||
litellm.modify_params = original_modify_params
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue