diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 3bffee48d6a..4408e4fe254 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1920,9 +1920,24 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): # If any message has thinking_blocks, we must keep thinking enabled, otherwise # Anthropic errors with: "When thinking is disabled, an assistant message cannot contain thinking" # Related issue: https://github.com/BerriAI/litellm/issues/18926 + # + # Adaptive thinking (thinking={"type": "adaptive"}, Claude 4.6+ / Opus 5.x) + # is exempt: Anthropic accepts a tool_use-only prior turn, because adaptive + # thinking may legitimately decide not to think on a trivial tool call. + # Dropping it breaks streaming for effort-configured models (no + # reasoning_content deltas until first text/tool token). + # + # The exemption is keyed off the *parameter shape*, not the model: a + # Claude 4.6+ model explicitly configured with legacy budget-based + # thinking ({"type": "enabled", "budget_tokens": N}) still needs the + # guard, exactly like older models. + # Related issue: https://github.com/BerriAI/litellm/issues/43531 + _thinking_param = optional_params.get("thinking") + _is_adaptive_thinking = isinstance(_thinking_param, dict) and _thinking_param.get("type") == "adaptive" if ( - optional_params.get("thinking") is not None + _thinking_param is not None and messages is not None + and not _is_adaptive_thinking and last_assistant_with_tool_calls_has_no_thinking_blocks(messages) and not any_assistant_message_has_thinking_blocks(messages) ): diff --git a/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py index 332153b4c7d..bf2b6b86964 100644 --- a/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -6798,3 +6798,95 @@ def test_chat_dummy_tool_result_for_an_orphaned_tool_call_replays_a_byte_identic _assert_prefix_stable(requests) assert [m["role"] for m in requests[0]["messages"]] == ["user", "assistant", "user"] assert requests[0]["messages"][2]["content"][0]["type"] == "tool_result" + + +def test_adaptive_thinking_kept_for_adaptive_model_when_tool_calls_missing_thinking_blocks(monkeypatch): + """ + modify_params must NOT drop thinking for adaptive-thinking models + (Claude 4.6+ / Opus 5.x): Anthropic accepts a tool_use-only prior turn + with thinking={"type": "adaptive"}, and dropping it breaks reasoning + streaming (no reasoning_content deltas until first text/tool token). + + Related issue: https://github.com/BerriAI/litellm/issues/43531 + """ + import litellm + + monkeypatch.setattr(litellm, "modify_params", True) + messages = [ + {"role": "user", "content": "Run `echo seed` with bash, then reason."}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "toolu_1", + "type": "function", + "function": {"name": "bash", "arguments": '{"command":"echo seed"}'}, + } + ], + }, + {"role": "tool", "tool_call_id": "toolu_1", "content": "seed"}, + ] + optional_params = { + "thinking": {"type": "adaptive", "display": "summarized"}, + "output_config": {"effort": "xhigh"}, + } + transformed = AnthropicConfig().transform_request( + model="claude-opus-5-5", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + assert "thinking" in transformed + assert transformed["thinking"] == {"type": "adaptive", "display": "summarized"} + assert transformed["output_config"] == {"effort": "xhigh"} + + +def test_legacy_budget_thinking_still_dropped_when_tool_calls_missing_thinking_blocks(monkeypatch): + """ + The original guard (#14194 / #18926) must keep working for legacy + budget-based thinking: a tool_use-only prior turn is rejected by + Anthropic unless the request pairs it with thinking blocks. + + Related issue: https://github.com/BerriAI/litellm/issues/43531 + """ + import litellm + + monkeypatch.setattr(litellm, "modify_params", True) + messages = [ + {"role": "user", "content": "Run `echo seed` with bash, then reason."}, + { + "role": "assistant", + "content": "", + "tool_calls": [ + { + "id": "toolu_1", + "type": "function", + "function": {"name": "bash", "arguments": '{"command":"echo seed"}'}, + } + ], + }, + {"role": "tool", "tool_call_id": "toolu_1", "content": "seed"}, + ] + # Legacy budget-based thinking on an older model: guard still drops. + optional_params = {"thinking": {"type": "enabled", "budget_tokens": 1024}} + transformed = AnthropicConfig().transform_request( + model="claude-3-5-sonnet-20241022", + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + assert "thinking" not in transformed + # Legacy budget-based thinking on an adaptive-capable model (Claude 4.6+): + # the exemption is keyed off the parameter shape, so the guard must still drop. + optional_params2 = {"thinking": {"type": "enabled", "budget_tokens": 2048}} + transformed2 = AnthropicConfig().transform_request( + model="claude-opus-5-5", + messages=messages, + optional_params=optional_params2, + litellm_params={}, + headers={}, + ) + assert "thinking" not in transformed2