diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index 4b6617fbeac..86c9c1db481 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -331,6 +331,7 @@ class LiteLLMAnthropicMessagesAdapter: "thinking", "output_format", "output_config", + "stop_sequences", ] def _is_web_search_tool(self, tool: Dict[str, Any]) -> bool: @@ -615,7 +616,7 @@ class LiteLLMAnthropicMessagesAdapter: thinking_type = thinking.get("type", "disabled") if thinking_type == "disabled": - return None + return "none" elif thinking_type == "enabled": return reasoning_effort_from_thinking_budget(thinking.get("budget_tokens", 0)) elif thinking_type == "adaptive": @@ -683,25 +684,37 @@ class LiteLLMAnthropicMessagesAdapter: thinking ) if reasoning_effort: - summary = thinking.get("summary") if isinstance(thinking, dict) else None - auto_summary = is_reasoning_auto_summary_enabled() - if summary: - return { - "reasoning_effort": { - "effort": reasoning_effort, - "summary": summary, - } - } - elif auto_summary: - return { - "reasoning_effort": { - "effort": reasoning_effort, - "summary": "detailed", - } - } - return {"reasoning_effort": reasoning_effort} + return { + "reasoning_effort": LiteLLMAnthropicMessagesAdapter._apply_reasoning_summary_wrapping( + reasoning_effort, thinking + ) + } return {} + @staticmethod + def _apply_reasoning_summary_wrapping( + reasoning_effort: str, + thinking: Dict[str, Any], + ) -> Any: + """ + Apply the reasoning_effort/summary wrapping rules shared by every + thinking->reasoning_effort translation path. + + Disabled thinking always stays a plain string - there's no reasoning + trace to summarize, and non-Claude providers (e.g. Fireworks) expect + reasoning_effort as a plain string, not a summary dict. + """ + thinking_type = thinking.get("type") if isinstance(thinking, dict) else None + if thinking_type == "disabled": + return reasoning_effort + + summary = thinking.get("summary") if isinstance(thinking, dict) else None + if summary: + return {"effort": reasoning_effort, "summary": summary} + if is_reasoning_auto_summary_enabled(): + return {"effort": reasoning_effort, "summary": "detailed"} + return reasoning_effort + def translate_anthropic_tool_choice_to_openai( self, tool_choice: AnthropicMessagesToolChoice ) -> ChatCompletionToolChoiceValues: @@ -919,6 +932,18 @@ class LiteLLMAnthropicMessagesAdapter: tool_choice=cast(AnthropicMessagesToolChoice, tool_choice) ) + def _translate_stop_sequences_to_openai( + self, + anthropic_message_request: AnthropicMessagesRequest, + new_kwargs: ChatCompletionRequest, + ) -> None: + if "stop_sequences" not in anthropic_message_request: + return + stop_sequences = anthropic_message_request["stop_sequences"] + if not stop_sequences: + return + new_kwargs["stop"] = stop_sequences + def _translate_tools_to_openai( self, anthropic_message_request: AnthropicMessagesRequest, @@ -976,32 +1001,17 @@ class LiteLLMAnthropicMessagesAdapter: if not reasoning_effort: return + thinking_type = thinking.get("type") if isinstance(thinking, dict) else None + # For adaptive thinking, override with output_config.effort if available - if isinstance(thinking, dict) and thinking.get("type") == "adaptive": + if thinking_type == "adaptive": output_config = anthropic_message_request.get("output_config") if isinstance(output_config, dict) and output_config.get("effort"): reasoning_effort = output_config["effort"] - summary = thinking.get("summary") if isinstance(thinking, dict) else None - auto_summary = is_reasoning_auto_summary_enabled() - if summary: - new_kwargs["reasoning_effort"] = cast( - Any, - { - "effort": reasoning_effort, - "summary": summary, - }, - ) - elif auto_summary: - new_kwargs["reasoning_effort"] = cast( - Any, - { - "effort": reasoning_effort, - "summary": "detailed", - }, - ) - else: - new_kwargs["reasoning_effort"] = reasoning_effort + new_kwargs["reasoning_effort"] = self._apply_reasoning_summary_wrapping( + reasoning_effort, cast(Dict[str, Any], thinking) + ) def _translate_output_format_to_openai( self, @@ -1098,6 +1108,11 @@ class LiteLLMAnthropicMessagesAdapter: anthropic_message_request=anthropic_message_request, new_kwargs=new_kwargs, ) + ## CONVERT STOP_SEQUENCES + self._translate_stop_sequences_to_openai( + anthropic_message_request=anthropic_message_request, + new_kwargs=new_kwargs, + ) ## CONVERT OUTPUT_FORMAT to RESPONSE_FORMAT self._translate_output_format_to_openai( anthropic_message_request=anthropic_message_request, diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py index dfe7e0c3a51..c0c6e315b5b 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py @@ -1582,6 +1582,67 @@ def test_thinking_still_translated_to_reasoning_effort_for_non_claude_model(): assert new_kwargs["reasoning_effort"] == "low" +def test_thinking_disabled_translated_to_reasoning_effort_none_for_non_claude_model(): + adapter = LiteLLMAnthropicMessagesAdapter() + thinking = {"type": "disabled"} + + new_kwargs = {"model": CACHE_CONTROL_NON_ANTHROPIC_MODEL} + adapter._translate_thinking_to_openai(cast(Any, {"thinking": thinking}), cast(Any, new_kwargs)) + + assert "thinking" not in new_kwargs + assert new_kwargs["reasoning_effort"] == "none" + + +def test_thinking_disabled_stays_plain_string_when_auto_summary_enabled(): + import litellm + + adapter = LiteLLMAnthropicMessagesAdapter() + thinking = {"type": "disabled"} + + original = litellm.reasoning_auto_summary + try: + litellm.reasoning_auto_summary = True + new_kwargs = {"model": CACHE_CONTROL_NON_ANTHROPIC_MODEL} + adapter._translate_thinking_to_openai(cast(Any, {"thinking": thinking}), cast(Any, new_kwargs)) + finally: + litellm.reasoning_auto_summary = original + + assert new_kwargs["reasoning_effort"] == "none" + + +def test_stop_sequences_translated_to_stop_for_non_claude_model(): + from litellm.types.llms.anthropic import AnthropicMessagesRequest + + anthropic_request = AnthropicMessagesRequest( + model=CACHE_CONTROL_NON_ANTHROPIC_MODEL, + max_tokens=1024, + messages=[{"role": "user", "content": "hi"}], + stop_sequences=[""], + ) + + adapter = LiteLLMAnthropicMessagesAdapter() + openai_request, _ = adapter.translate_anthropic_to_openai(anthropic_message_request=anthropic_request) + + assert openai_request["stop"] == [""] + assert "stop_sequences" not in openai_request + + +def test_empty_stop_sequences_does_not_set_stop(): + from litellm.types.llms.anthropic import AnthropicMessagesRequest + + anthropic_request = AnthropicMessagesRequest( + model=CACHE_CONTROL_NON_ANTHROPIC_MODEL, + max_tokens=1024, + messages=[{"role": "user", "content": "hi"}], + stop_sequences=[], + ) + + adapter = LiteLLMAnthropicMessagesAdapter() + openai_request, _ = adapter.translate_anthropic_to_openai(anthropic_message_request=anthropic_request) + + assert "stop" not in openai_request + + def test_cache_control_preserved_in_image_content_for_claude(): """Cache control should be preserved in image content for Claude models.""" anthropic_messages = [ diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index 8875a75e86f..df3db3d2c57 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -651,6 +651,26 @@ class TestThinkingSummaryPreservation: "reasoning_effort": {"effort": "high", "summary": "concise"} } + def test_translate_thinking_for_model_disabled_stays_plain_string_when_auto_summary_enabled(self): + """Disabled thinking must stay a plain string even when reasoning_auto_summary is on.""" + import litellm + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + LiteLLMAnthropicMessagesAdapter, + ) + + original = litellm.reasoning_auto_summary + try: + litellm.reasoning_auto_summary = True + thinking = {"type": "disabled"} + result = LiteLLMAnthropicMessagesAdapter.translate_thinking_for_model( + thinking=thinking, + model="openai/gpt-5.2", + ) + finally: + litellm.reasoning_auto_summary = original + + assert result == {"reasoning_effort": "none"} + # --------------------------------------------------------------------------- # Parity tests: redundant empty-text-block sanitization scan removal.