From 07a360bb1b2e132e0a95e37ad09b2a56b2d18214 Mon Sep 17 00:00:00 2001 From: Kent <72616338+kingdoooo@users.noreply.github.com> Date: Sun, 17 May 2026 07:50:09 +0000 Subject: [PATCH] fix(anthropic): don't force tool_choice on adaptive-thinking models MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adaptive-thinking models (Claude 4.7+) use thinking.type == "adaptive" rather than "enabled", so is_thinking_enabled() — which only matches "enabled" or reasoning_effort — reports False for them. On the response_format → synthetic-tool fallback path, the "if not is_thinking_enabled" guard then synthesizes a forced tool_choice = {"name": "json_tool_call", "type": "tool"} while "thinking": {"type": "adaptive"} is still in the optional_params, and Anthropic / Bedrock reject the request with: Thinking may not be enabled when tool_choice forces tool use. This is unreachable for the in-tree Claude 4.6/4.7 models today because they all match the response_format output_format whitelist a few lines earlier, but it bites any deployment that pushes those models off the native output_format path (e.g. yaml override of supports_response_schema, or future adaptive models that aren't yet on the whitelist). A parallel symptom in another SDK is documented in vercel/ai#14773. Add a carve-out using the existing _is_adaptive_thinking_model() helper so the forced tool_choice is skipped for adaptive-thinking models. The synthetic JSON tool itself is still added, and json_mode is still set, so the response path is unchanged — the model can still emit JSON via the tool when it chooses to. Two unit tests: - test_adaptive_thinking_model_response_format_fallback_no_forced_tool_choice pins the new behaviour - test_non_adaptive_model_response_format_fallback_still_forces_tool_choice guards against silently widening the carve-out to non-adaptive models --- litellm/llms/anthropic/chat/transformation.py | 14 ++- .../test_anthropic_completion.py | 98 +++++++++++++++++++ 2 files changed, 111 insertions(+), 1 deletion(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 1ce80207552..5fc8a343226 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1486,7 +1486,19 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): ) if _tool is None: continue - if not is_thinking_enabled: + # Adaptive thinking models (Claude 4.7+) use + # ``thinking.type == "adaptive"`` rather than + # ``"enabled"``, so ``is_thinking_enabled`` (which only + # matches ``"enabled"`` / ``reasoning_effort``) reports + # False. Without this extra check we set a forced + # ``tool_choice`` while ``thinking: adaptive`` is still on + # the wire body, and Anthropic / Bedrock reject the + # request with "Thinking may not be enabled when + # tool_choice forces tool use". + is_adaptive_thinking = self._is_adaptive_thinking_model( + model + ) + if not is_thinking_enabled and not is_adaptive_thinking: _tool_choice = { "name": RESPONSE_FORMAT_TOOL_NAME, "type": "tool", diff --git a/tests/llm_translation/test_anthropic_completion.py b/tests/llm_translation/test_anthropic_completion.py index 7a478e494b1..a546345b8ed 100644 --- a/tests/llm_translation/test_anthropic_completion.py +++ b/tests/llm_translation/test_anthropic_completion.py @@ -1924,3 +1924,101 @@ def test_anthropic_streaming_completion_replay(): assert chunk_count > 1, "expected multiple SSE chunks from streaming response" assert collected_text.strip(), collected_text assert finish_reason in {"stop", "length"} + + +def test_adaptive_thinking_model_response_format_fallback_no_forced_tool_choice( + monkeypatch, +): + """Adaptive-thinking models must not get a forced ``tool_choice`` when + response_format falls back to the synthetic-tool path. + + ``thinking.type == "adaptive"`` is not matched by ``is_thinking_enabled`` + (which only recognises ``"enabled"`` or ``reasoning_effort``), so the + ``if not is_thinking_enabled`` guard around the synthetic + ``RESPONSE_FORMAT_TOOL_NAME`` tool_choice would fire and pin the model to + that tool. Anthropic / Bedrock then reject the request with + "Thinking may not be enabled when tool_choice forces tool use". + + We construct a model name outside the response-format whitelist so the + fallback branch is reached, and force ``_is_adaptive_thinking_model`` to + True via monkeypatch so the test does not depend on which models the + model_cost map happens to flag as adaptive. + + See vercel/ai#14773 for the parallel symptom in another SDK. + """ + config = litellm.AnthropicConfig() + monkeypatch.setattr( + litellm.AnthropicConfig, + "_is_adaptive_thinking_model", + staticmethod(lambda model: True), + ) + + mapped = config.map_openai_params( + non_default_params={ + "response_format": { + "type": "json_schema", + "json_schema": { + "schema": { + "type": "object", + "properties": {"answer": {"type": "integer"}}, + "required": ["answer"], + }, + "name": "out", + }, + }, + }, + optional_params={}, + # Deliberately outside the hardcoded response_format whitelist so the + # synthetic-tool fallback branch is exercised. + model="some-future-adaptive-model", + drop_params=False, + ) + + assert "tool_choice" not in mapped, ( + "Adaptive-thinking models must not have a forced tool_choice " + "synthesised on the response_format fallback path; got: " + f"{mapped.get('tool_choice')!r}" + ) + # The tool itself is still added so the model can still emit JSON via + # the synthetic tool when it chooses to — only the *forcing* is removed. + assert mapped.get("json_mode") is True + assert any( + t.get("name") == "json_tool_call" for t in mapped.get("tools", []) + ), f"expected synthetic json_tool_call tool to be present, got tools={mapped.get('tools')}" + + +def test_non_adaptive_model_response_format_fallback_still_forces_tool_choice(): + """Regression guard: non-adaptive, non-thinking models keep the existing + forced ``tool_choice`` behaviour on the response_format fallback path. + + This test exists to make sure the adaptive-thinking carve-out added in + the same change does not silently widen and stop forcing tool_choice + for the original (non-thinking) callers, which depend on it to make + the model actually invoke the synthetic JSON tool. + """ + mapped = litellm.AnthropicConfig().map_openai_params( + non_default_params={ + "response_format": { + "type": "json_schema", + "json_schema": { + "schema": { + "type": "object", + "properties": {"answer": {"type": "integer"}}, + "required": ["answer"], + }, + "name": "out", + }, + }, + }, + optional_params={}, + # claude-3-5-sonnet is not in the response_format output_format + # whitelist and is not flagged as adaptive in the model_cost map, + # so the fallback branch runs with the original behaviour. + model="claude-3-5-sonnet-20240620", + drop_params=False, + ) + + assert mapped.get("tool_choice") == { + "name": "json_tool_call", + "type": "tool", + }, f"expected forced json_tool_call tool_choice, got {mapped.get('tool_choice')!r}"