From b3e27a0bc30985300b8986ca1a3fcd7638cc0ab9 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Fri, 24 Jul 2026 18:07:00 -0700 Subject: [PATCH 1/5] fix(anthropic-adapter): translate stop_sequences and disabled thinking for non-Claude targets Claude Code's auto-mode classifier sends stop_sequences and thinking: {type: disabled} on /v1/messages. The Anthropic adapter passed stop_sequences through unchanged instead of mapping it to OpenAI's stop, which Fireworks' OpenAI-compatible endpoint rejects with HTTP 400. It also dropped disabled thinking instead of mapping it to reasoning_effort: none, so the model spent its output budget on reasoning it was told to skip. Resolves LIT-4798 --- .../adapters/transformation.py | 20 ++++++++++++- ...al_pass_through_adapters_transformation.py | 28 +++++++++++++++++++ 2 files changed, 47 insertions(+), 1 deletion(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index 4b6617fbeac..edcb474b01f 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -331,6 +331,7 @@ class LiteLLMAnthropicMessagesAdapter: "thinking", "output_format", "output_config", + "stop_sequences", ] def _is_web_search_tool(self, tool: Dict[str, Any]) -> bool: @@ -615,7 +616,7 @@ class LiteLLMAnthropicMessagesAdapter: thinking_type = thinking.get("type", "disabled") if thinking_type == "disabled": - return None + return "none" elif thinking_type == "enabled": return reasoning_effort_from_thinking_budget(thinking.get("budget_tokens", 0)) elif thinking_type == "adaptive": @@ -919,6 +920,18 @@ class LiteLLMAnthropicMessagesAdapter: tool_choice=cast(AnthropicMessagesToolChoice, tool_choice) ) + def _translate_stop_sequences_to_openai( + self, + anthropic_message_request: AnthropicMessagesRequest, + new_kwargs: ChatCompletionRequest, + ) -> None: + if "stop_sequences" not in anthropic_message_request: + return + stop_sequences = anthropic_message_request["stop_sequences"] + if not stop_sequences: + return + new_kwargs["stop"] = stop_sequences + def _translate_tools_to_openai( self, anthropic_message_request: AnthropicMessagesRequest, @@ -1098,6 +1111,11 @@ class LiteLLMAnthropicMessagesAdapter: anthropic_message_request=anthropic_message_request, new_kwargs=new_kwargs, ) + ## CONVERT STOP_SEQUENCES + self._translate_stop_sequences_to_openai( + anthropic_message_request=anthropic_message_request, + new_kwargs=new_kwargs, + ) ## CONVERT OUTPUT_FORMAT to RESPONSE_FORMAT self._translate_output_format_to_openai( anthropic_message_request=anthropic_message_request, diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py index dfe7e0c3a51..f8984e6f4e6 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py @@ -1582,6 +1582,34 @@ def test_thinking_still_translated_to_reasoning_effort_for_non_claude_model(): assert new_kwargs["reasoning_effort"] == "low" +def test_thinking_disabled_translated_to_reasoning_effort_none_for_non_claude_model(): + adapter = LiteLLMAnthropicMessagesAdapter() + thinking = {"type": "disabled"} + + new_kwargs = {"model": CACHE_CONTROL_NON_ANTHROPIC_MODEL} + adapter._translate_thinking_to_openai(cast(Any, {"thinking": thinking}), cast(Any, new_kwargs)) + + assert "thinking" not in new_kwargs + assert new_kwargs["reasoning_effort"] == "none" + + +def test_stop_sequences_translated_to_stop_for_non_claude_model(): + from litellm.types.llms.anthropic import AnthropicMessagesRequest + + anthropic_request = AnthropicMessagesRequest( + model=CACHE_CONTROL_NON_ANTHROPIC_MODEL, + max_tokens=1024, + messages=[{"role": "user", "content": "hi"}], + stop_sequences=[""], + ) + + adapter = LiteLLMAnthropicMessagesAdapter() + openai_request, _ = adapter.translate_anthropic_to_openai(anthropic_message_request=anthropic_request) + + assert openai_request["stop"] == [""] + assert "stop_sequences" not in openai_request + + def test_cache_control_preserved_in_image_content_for_claude(): """Cache control should be preserved in image content for Claude models.""" anthropic_messages = [ From 9da21f38a92d2d3165482f5ec07049f8bef4f93b Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Fri, 24 Jul 2026 18:29:58 -0700 Subject: [PATCH 2/5] fix(anthropic-adapter): keep disabled-thinking reasoning_effort a plain string MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Guard against reasoning_auto_summary wrapping "none" into a dict when thinking is disabled — there's no reasoning trace to summarize, and non-Claude providers (e.g. Fireworks) expect reasoning_effort as a plain string. --- .../adapters/transformation.py | 8 +++++++- ...ntal_pass_through_adapters_transformation.py | 17 +++++++++++++++++ 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index edcb474b01f..3d4233e719a 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -989,12 +989,18 @@ class LiteLLMAnthropicMessagesAdapter: if not reasoning_effort: return + thinking_type = thinking.get("type") if isinstance(thinking, dict) else None + # For adaptive thinking, override with output_config.effort if available - if isinstance(thinking, dict) and thinking.get("type") == "adaptive": + if thinking_type == "adaptive": output_config = anthropic_message_request.get("output_config") if isinstance(output_config, dict) and output_config.get("effort"): reasoning_effort = output_config["effort"] + if thinking_type == "disabled": + new_kwargs["reasoning_effort"] = reasoning_effort + return + summary = thinking.get("summary") if isinstance(thinking, dict) else None auto_summary = is_reasoning_auto_summary_enabled() if summary: diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py index f8984e6f4e6..326c8858551 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py @@ -1593,6 +1593,23 @@ def test_thinking_disabled_translated_to_reasoning_effort_none_for_non_claude_mo assert new_kwargs["reasoning_effort"] == "none" +def test_thinking_disabled_stays_plain_string_when_auto_summary_enabled(): + import litellm + + adapter = LiteLLMAnthropicMessagesAdapter() + thinking = {"type": "disabled"} + + original = litellm.reasoning_auto_summary + try: + litellm.reasoning_auto_summary = True + new_kwargs = {"model": CACHE_CONTROL_NON_ANTHROPIC_MODEL} + adapter._translate_thinking_to_openai(cast(Any, {"thinking": thinking}), cast(Any, new_kwargs)) + finally: + litellm.reasoning_auto_summary = original + + assert new_kwargs["reasoning_effort"] == "none" + + def test_stop_sequences_translated_to_stop_for_non_claude_model(): from litellm.types.llms.anthropic import AnthropicMessagesRequest From fed03a41d1e47be0c41e87ee3b24d688ba83f0a8 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Fri, 24 Jul 2026 19:00:20 -0700 Subject: [PATCH 3/5] test(anthropic-adapter): cover empty stop_sequences edge case MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codecov flagged the empty-list early-return in _translate_stop_sequences_to_openai as an uncovered line in the diff — add a regression test asserting stop_sequences=[] does not set new_kwargs["stop"]. --- ...ental_pass_through_adapters_transformation.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py index 326c8858551..c0c6e315b5b 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py @@ -1627,6 +1627,22 @@ def test_stop_sequences_translated_to_stop_for_non_claude_model(): assert "stop_sequences" not in openai_request +def test_empty_stop_sequences_does_not_set_stop(): + from litellm.types.llms.anthropic import AnthropicMessagesRequest + + anthropic_request = AnthropicMessagesRequest( + model=CACHE_CONTROL_NON_ANTHROPIC_MODEL, + max_tokens=1024, + messages=[{"role": "user", "content": "hi"}], + stop_sequences=[], + ) + + adapter = LiteLLMAnthropicMessagesAdapter() + openai_request, _ = adapter.translate_anthropic_to_openai(anthropic_message_request=anthropic_request) + + assert "stop" not in openai_request + + def test_cache_control_preserved_in_image_content_for_claude(): """Cache control should be preserved in image content for Claude models.""" anthropic_messages = [ From 5072590c27aba27212be4d86c2daba03c05ea0cf Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Fri, 24 Jul 2026 19:31:29 -0700 Subject: [PATCH 4/5] fix(anthropic-adapter): dedupe reasoning_effort wrapping to close sibling gap translate_thinking_for_model duplicated the same summary/auto_summary wrapping logic as _translate_thinking_to_openai without the disabled-thinking guard, so it could still wrap "none" into an {effort, summary} dict when reasoning_auto_summary is enabled (caught by Cursor Bugbot). Extract the wrapping rule into one shared _apply_reasoning_summary_wrapping helper used by both call sites so this invariant can't drift apart again. --- .../adapters/transformation.py | 73 ++++++++----------- ...erimental_pass_through_messages_handler.py | 20 +++++ 2 files changed, 52 insertions(+), 41 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index 3d4233e719a..d046ff1eaeb 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -684,25 +684,37 @@ class LiteLLMAnthropicMessagesAdapter: thinking ) if reasoning_effort: - summary = thinking.get("summary") if isinstance(thinking, dict) else None - auto_summary = is_reasoning_auto_summary_enabled() - if summary: - return { - "reasoning_effort": { - "effort": reasoning_effort, - "summary": summary, - } - } - elif auto_summary: - return { - "reasoning_effort": { - "effort": reasoning_effort, - "summary": "detailed", - } - } - return {"reasoning_effort": reasoning_effort} + return { + "reasoning_effort": LiteLLMAnthropicMessagesAdapter._apply_reasoning_summary_wrapping( + reasoning_effort, thinking + ) + } return {} + @staticmethod + def _apply_reasoning_summary_wrapping( + reasoning_effort: str, + thinking: Dict[str, Any], + ) -> Any: + """ + Apply the reasoning_effort/summary wrapping rules shared by every + thinking->reasoning_effort translation path. + + Disabled thinking always stays a plain string - there's no reasoning + trace to summarize, and non-Claude providers (e.g. Fireworks) expect + reasoning_effort as a plain string, not a summary dict. + """ + thinking_type = thinking.get("type") if isinstance(thinking, dict) else None + if thinking_type == "disabled": + return reasoning_effort + + summary = thinking.get("summary") if isinstance(thinking, dict) else None + if summary: + return cast(Any, {"effort": reasoning_effort, "summary": summary}) + if is_reasoning_auto_summary_enabled(): + return cast(Any, {"effort": reasoning_effort, "summary": "detailed"}) + return reasoning_effort + def translate_anthropic_tool_choice_to_openai( self, tool_choice: AnthropicMessagesToolChoice ) -> ChatCompletionToolChoiceValues: @@ -997,30 +1009,9 @@ class LiteLLMAnthropicMessagesAdapter: if isinstance(output_config, dict) and output_config.get("effort"): reasoning_effort = output_config["effort"] - if thinking_type == "disabled": - new_kwargs["reasoning_effort"] = reasoning_effort - return - - summary = thinking.get("summary") if isinstance(thinking, dict) else None - auto_summary = is_reasoning_auto_summary_enabled() - if summary: - new_kwargs["reasoning_effort"] = cast( - Any, - { - "effort": reasoning_effort, - "summary": summary, - }, - ) - elif auto_summary: - new_kwargs["reasoning_effort"] = cast( - Any, - { - "effort": reasoning_effort, - "summary": "detailed", - }, - ) - else: - new_kwargs["reasoning_effort"] = reasoning_effort + new_kwargs["reasoning_effort"] = self._apply_reasoning_summary_wrapping( + reasoning_effort, cast(Dict[str, Any], thinking) + ) def _translate_output_format_to_openai( self, diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index 8875a75e86f..df3db3d2c57 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -651,6 +651,26 @@ class TestThinkingSummaryPreservation: "reasoning_effort": {"effort": "high", "summary": "concise"} } + def test_translate_thinking_for_model_disabled_stays_plain_string_when_auto_summary_enabled(self): + """Disabled thinking must stay a plain string even when reasoning_auto_summary is on.""" + import litellm + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + LiteLLMAnthropicMessagesAdapter, + ) + + original = litellm.reasoning_auto_summary + try: + litellm.reasoning_auto_summary = True + thinking = {"type": "disabled"} + result = LiteLLMAnthropicMessagesAdapter.translate_thinking_for_model( + thinking=thinking, + model="openai/gpt-5.2", + ) + finally: + litellm.reasoning_auto_summary = original + + assert result == {"reasoning_effort": "none"} + # --------------------------------------------------------------------------- # Parity tests: redundant empty-text-block sanitization scan removal. From d478b9955e2fcd0baaf7e40eb6f5b35e805a918c Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Fri, 24 Jul 2026 20:01:43 -0700 Subject: [PATCH 5/5] fix(anthropic-adapter): drop redundant casts to satisfy type-discipline budget _apply_reasoning_summary_wrapping already returns Any, so wrapping its dict-literal returns in cast(Any, ...) was a no-op that only inflated the LIT006 cast-count budget the lint gate enforces. --- .../experimental_pass_through/adapters/transformation.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index d046ff1eaeb..86c9c1db481 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -710,9 +710,9 @@ class LiteLLMAnthropicMessagesAdapter: summary = thinking.get("summary") if isinstance(thinking, dict) else None if summary: - return cast(Any, {"effort": reasoning_effort, "summary": summary}) + return {"effort": reasoning_effort, "summary": summary} if is_reasoning_auto_summary_enabled(): - return cast(Any, {"effort": reasoning_effort, "summary": "detailed"}) + return {"effort": reasoning_effort, "summary": "detailed"} return reasoning_effort def translate_anthropic_tool_choice_to_openai(