diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py index e6a73fdbb59..73e74c228ba 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/handler.py @@ -44,7 +44,8 @@ class LiteLLMMessagesToCompletionTransformationHandler: For OpenAI models, Chat Completions typically does not return reasoning text (only token accounting). To return a thinking-like content block in the - Anthropic response format, we route the request through OpenAI's Responses API. + Anthropic response format, we route the request through OpenAI's Responses API + and request a reasoning summary. """ custom_llm_provider = completion_kwargs.get("custom_llm_provider") if custom_llm_provider is None: @@ -79,7 +80,16 @@ class LiteLLMMessagesToCompletionTransformationHandler: if isinstance(reasoning_effort, str) and reasoning_effort: completion_kwargs["reasoning_effort"] = { "effort": reasoning_effort, + "summary": "detailed", } + elif isinstance(reasoning_effort, dict): + if ( + "summary" not in reasoning_effort + and "generate_summary" not in reasoning_effort + ): + updated_reasoning_effort = dict(reasoning_effort) + updated_reasoning_effort["summary"] = "detailed" + completion_kwargs["reasoning_effort"] = updated_reasoning_effort @staticmethod def _prepare_completion_kwargs( diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py index 05d14704905..935babe4380 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/transformation.py @@ -241,7 +241,7 @@ class LiteLLMAnthropicToResponsesAPIAdapter: effort = "low" else: effort = "minimal" - return {"effort": effort} + return {"effort": effort, "summary": "detailed"} def translate_request( self, diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index 24c3f24bc79..636e84fe796 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -180,8 +180,7 @@ def test_openai_model_with_thinking_converts_to_reasoning(): assert "reasoning" in call_kwargs, "reasoning should be passed to litellm.responses" # budget_tokens=1024 -> effort="minimal" (< 2000 threshold) - # summary should NOT be hardcoded — it's opt-in per the OpenAI spec - expected_reasoning = {"effort": "minimal"} + expected_reasoning = {"effort": "minimal", "summary": "detailed"} assert call_kwargs["reasoning"] == expected_reasoning, ( f"reasoning should be {expected_reasoning} for budget_tokens=1024, " f"got {call_kwargs.get('reasoning')}" @@ -223,38 +222,3 @@ class TestThinkingParameterTransformation: assert result == {"reasoning_effort": "minimal"} assert "thinking" not in result - - -class TestNoHardcodedReasoningSummary: - """Tests for issue #20998: adapter must not hardcode reasoning summary. - - Per OpenAI spec, reasoning.summary is opt-in. The adapter should not - inject summary='detailed' when the user didn't request it. - """ - - def test_no_summary_added_when_not_requested(self): - """reasoning_effort dict should only contain 'effort', no 'summary'.""" - from litellm.llms.anthropic.experimental_pass_through.adapters.handler import ( - LiteLLMMessagesToCompletionTransformationHandler, - ) - - thinking = {"type": "enabled", "budget_tokens": 5000} - completion_kwargs = {"model": "openai/gpt-5.1", "reasoning_effort": "medium"} - LiteLLMMessagesToCompletionTransformationHandler._route_openai_thinking_to_responses_api_if_needed( - completion_kwargs, thinking=thinking - ) - assert completion_kwargs["reasoning_effort"] == {"effort": "medium"} - assert "summary" not in completion_kwargs["reasoning_effort"] - - def test_model_prefixed_with_responses(self): - """Model should be prefixed with 'responses/' for Responses API routing.""" - from litellm.llms.anthropic.experimental_pass_through.adapters.handler import ( - LiteLLMMessagesToCompletionTransformationHandler, - ) - - thinking = {"type": "enabled", "budget_tokens": 5000} - completion_kwargs = {"model": "openai/gpt-5.1", "reasoning_effort": "medium"} - LiteLLMMessagesToCompletionTransformationHandler._route_openai_thinking_to_responses_api_if_needed( - completion_kwargs, thinking=thinking - ) - assert completion_kwargs["model"] == "responses/openai/gpt-5.1"