fix(anthropic): keep adaptive thinking when modify_params drops missing thinking blocks (#43531)

The modify_params guard (#14194 / #18926) drops the thinking param whenever
the last assistant message with tool_calls has no thinking_blocks. For
adaptive-thinking models (Claude 4.6+ / Opus 5.x, thinking={"type": "adaptive"})
that guard fires on ordinary traffic and Anthropic does not need it: adaptive
thinking may legitimately decide not to think on a trivial first tool call.
Dropping it made effort-configured models stream no reasoning (no
reasoning_content deltas, empty signed thinking block), and at high effort the
silence could exceed intermediary timeouts.

Skip the drop for models where AnthropicConfig._is_adaptive_thinking_model()
is true; legacy budget-based thinking keeps the existing behavior.

Tests: adaptive model keeps thinking/output_config; legacy model still drops.
This commit is contained in:
KaiyiQuan 2026-09-28 14:32:32 +08:00
parent 74cad08997
commit aebb107803
2 changed files with 97 additions and 0 deletions

View file

@ -1920,9 +1920,18 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
# If any message has thinking_blocks, we must keep thinking enabled, otherwise
# Anthropic errors with: "When thinking is disabled, an assistant message cannot contain thinking"
# Related issue: https://github.com/BerriAI/litellm/issues/18926
#
# Adaptive-thinking models (Claude 4.6+ / Opus 5.x) are exempt: with
# thinking={"type": "adaptive"} Anthropic accepts a tool_use-only prior
# turn (adaptive thinking may legitimately decide not to think on a
# trivial tool call), so the guard applies only to legacy budget-based
# thinking. Dropping it there breaks streaming for effort-configured
# models (no reasoning_content deltas until first text/tool token).
# Related issue: https://github.com/BerriAI/litellm/issues/43531
if (
optional_params.get("thinking") is not None
and messages is not None
and not AnthropicConfig._is_adaptive_thinking_model(model, self._resolved_provider)
and last_assistant_with_tool_calls_has_no_thinking_blocks(messages)
and not any_assistant_message_has_thinking_blocks(messages)
):

View file

@ -6798,3 +6798,91 @@ def test_chat_dummy_tool_result_for_an_orphaned_tool_call_replays_a_byte_identic
_assert_prefix_stable(requests)
assert [m["role"] for m in requests[0]["messages"]] == ["user", "assistant", "user"]
assert requests[0]["messages"][2]["content"][0]["type"] == "tool_result"
def test_adaptive_thinking_kept_for_adaptive_model_when_tool_calls_missing_thinking_blocks():
"""
modify_params must NOT drop thinking for adaptive-thinking models
(Claude 4.6+ / Opus 5.x): Anthropic accepts a tool_use-only prior turn
with thinking={"type": "adaptive"}, and dropping it breaks reasoning
streaming (no reasoning_content deltas until first text/tool token).
Related issue: https://github.com/BerriAI/litellm/issues/43531
"""
import litellm
original_modify_params = litellm.modify_params
litellm.modify_params = True
try:
messages = [
{"role": "user", "content": "Run `echo seed` with bash, then reason."},
{
"role": "assistant",
"content": None,
"tool_calls": [
{
"id": "toolu_1",
"type": "function",
"function": {"name": "bash", "arguments": '{"command":"echo seed"}'},
}
],
},
{"role": "tool", "tool_call_id": "toolu_1", "content": "seed"},
]
optional_params = {
"thinking": {"type": "adaptive", "display": "summarized"},
"output_config": {"effort": "xhigh"},
}
transformed = AnthropicConfig().transform_request(
model="claude-opus-5-5",
messages=messages,
optional_params=optional_params,
litellm_params={},
headers={},
)
assert "thinking" in transformed
assert transformed["thinking"] == {"type": "adaptive", "display": "summarized"}
assert transformed["output_config"] == {"effort": "xhigh"}
finally:
litellm.modify_params = original_modify_params
def test_legacy_budget_thinking_still_dropped_when_tool_calls_missing_thinking_blocks():
"""
The original guard (#14194 / #18926) must keep working for legacy
budget-based thinking: a tool_use-only prior turn is rejected by
Anthropic unless the request pairs it with thinking blocks.
Related issue: https://github.com/BerriAI/litellm/issues/43531
"""
import litellm
original_modify_params = litellm.modify_params
litellm.modify_params = True
try:
messages = [
{"role": "user", "content": "Run `echo seed` with bash, then reason."},
{
"role": "assistant",
"content": "",
"tool_calls": [
{
"id": "toolu_1",
"type": "function",
"function": {"name": "bash", "arguments": '{"command":"echo seed"}'},
}
],
},
{"role": "tool", "tool_call_id": "toolu_1", "content": "seed"},
]
optional_params = {"thinking": {"type": "enabled", "budget_tokens": 1024}}
transformed = AnthropicConfig().transform_request(
model="claude-3-5-sonnet-20241022",
messages=messages,
optional_params=optional_params,
litellm_params={},
headers={},
)
assert "thinking" not in transformed
finally:
litellm.modify_params = original_modify_params