diff --git a/litellm/llms/deepseek/chat/transformation.py b/litellm/llms/deepseek/chat/transformation.py index b90b1e1aa21..c144286969c 100644 --- a/litellm/llms/deepseek/chat/transformation.py +++ b/litellm/llms/deepseek/chat/transformation.py @@ -49,14 +49,13 @@ class DeepSeekChatConfig(OpenAIGPTConfig): thinking_value = optional_params.pop("thinking", None) reasoning_effort = optional_params.pop("reasoning_effort", None) - # Handle thinking parameter - only accept {"type": "enabled"} + # DeepSeek only accepts the `type` key, ignore budget_tokens if thinking_value is not None: - if ( - isinstance(thinking_value, dict) - and thinking_value.get("type") == "enabled" + if isinstance(thinking_value, dict) and thinking_value.get("type") in ( + "enabled", + "disabled", ): - # DeepSeek only accepts {"type": "enabled"}, ignore budget_tokens - optional_params["thinking"] = {"type": "enabled"} + optional_params["thinking"] = {"type": thinking_value["type"]} # Handle reasoning_effort - map to thinking enabled elif reasoning_effort is not None and reasoning_effort != "none": @@ -137,14 +136,13 @@ class DeepSeekChatConfig(OpenAIGPTConfig): def _thinking_mode_active(self, model: str, optional_params: dict) -> bool: """ - Returns True only when thinking mode is actually active for this request: - - model supports reasoning (capability check) - - user explicitly passed thinking={"type": "enabled"} (opt-in check) + DeepSeek V4 enables thinking by default, so any reasoning-capable model + counts unless thinking is explicitly disabled. The API ignores + `reasoning_content` in non-thinking requests, so over-injecting is safe. """ - return ( - supports_reasoning(model=model, custom_llm_provider="deepseek") - and (optional_params.get("thinking") or {}).get("type") == "enabled" - ) + if (optional_params.get("thinking") or {}).get("type") == "disabled": + return False + return supports_reasoning(model=model, custom_llm_provider="deepseek") @staticmethod def _drop_unsupported_tools(optional_params: dict) -> dict: @@ -235,13 +233,8 @@ class DeepSeekChatConfig(OpenAIGPTConfig): headers: dict, ) -> dict: """ - Ensures `reasoning_content` is forwarded on assistant messages for - multi-turn thinking-mode conversations (issue #28045). - - Only runs when thinking mode is actually active - guarded by both - supports_reasoning() (model capability) and optional_params["thinking"] - (user explicitly enabled it), preventing spurious injection on models - like deepseek-v3.2 that support thinking as opt-in but not always-on. + Forwards `reasoning_content` on assistant messages for multi-turn + thinking-mode conversations. """ optional_params = self._drop_unsupported_tools(optional_params) if self._thinking_mode_active(model=model, optional_params=optional_params): diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6a65977e909..c72dc397286 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -11300,6 +11300,52 @@ "supports_system_messages": true, "supports_tool_choice": false }, + "deepseek-v4-flash": { + "cache_read_input_token_cost": 2.8e-09, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "deepseek-v4-pro": { + "cache_read_input_token_cost": 3.625e-09, + "input_cost_per_token": 4.35e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 8.7e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "dashscope/qwen-coder": { "input_cost_per_token": 3e-07, "litellm_provider": "dashscope", @@ -13892,6 +13938,52 @@ "supports_reasoning": true, "supports_tool_choice": true }, + "deepseek/deepseek-v4-flash": { + "cache_read_input_token_cost": 2.8e-09, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "deepseek/deepseek-v4-pro": { + "cache_read_input_token_cost": 3.625e-09, + "input_cost_per_token": 4.35e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 8.7e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "deepseek.v3-v1:0": { "input_cost_per_token": 5.8e-07, "litellm_provider": "bedrock_converse", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d5af7391439..8ea65d5af73 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -11300,6 +11300,52 @@ "supports_system_messages": true, "supports_tool_choice": false }, + "deepseek-v4-flash": { + "cache_read_input_token_cost": 2.8e-09, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "deepseek-v4-pro": { + "cache_read_input_token_cost": 3.625e-09, + "input_cost_per_token": 4.35e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 8.7e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "dashscope/qwen-coder": { "input_cost_per_token": 3e-07, "litellm_provider": "dashscope", @@ -13892,6 +13938,52 @@ "supports_reasoning": true, "supports_tool_choice": true }, + "deepseek/deepseek-v4-flash": { + "cache_read_input_token_cost": 2.8e-09, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "deepseek/deepseek-v4-pro": { + "cache_read_input_token_cost": 3.625e-09, + "input_cost_per_token": 4.35e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 8.7e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "deepseek.v3-v1:0": { "input_cost_per_token": 5.8e-07, "litellm_provider": "bedrock_converse", diff --git a/tests/llm_translation/test_deepseek_completion.py b/tests/llm_translation/test_deepseek_completion.py index 2ede5d3f3f8..d3cb913625a 100644 --- a/tests/llm_translation/test_deepseek_completion.py +++ b/tests/llm_translation/test_deepseek_completion.py @@ -233,13 +233,8 @@ def test_deepseek_fill_reasoning_content_multiturn(): def test_deepseek_fill_reasoning_content_guard_in_transform_request(): """ - _fill_reasoning_content must only run when BOTH conditions are true: - 1. supports_reasoning() is True for the model - 2. thinking mode is explicitly enabled in optional_params ({"type": "enabled"}) - - This prevents spurious injection on models like deepseek-v3.2 that support - thinking as opt-in but not always-on. Addresses oss-pr-review-agent feedback - on PR #28057. + _fill_reasoning_content runs for any reasoning-capable DeepSeek model + unless thinking is explicitly disabled (V4 enables thinking by default). """ from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig @@ -263,7 +258,8 @@ def test_deepseek_fill_reasoning_content_guard_in_transform_request(): "reasoning_content should be injected when thinking is enabled" ) - # Case 2: reasoning model + thinking NOT in optional_params -> no injection + # Case 2: reasoning model + thinking NOT in optional_params -> injection + # (thinking is on by default for DeepSeek V4 / deepseek-reasoner) result = config.transform_request( model="deepseek-reasoner", messages=messages, @@ -271,11 +267,23 @@ def test_deepseek_fill_reasoning_content_guard_in_transform_request(): litellm_params={}, headers={}, ) - assert "reasoning_content" not in result["messages"][1], ( - "reasoning_content should not be injected when thinking is not enabled" + assert result["messages"][1].get("reasoning_content") == " ", ( + "reasoning_content should be injected by default for reasoning models" ) - # Case 3: non-reasoning model + thinking enabled -> no injection + # Case 3: reasoning model + thinking explicitly disabled -> no injection + result = config.transform_request( + model="deepseek-reasoner", + messages=messages, + optional_params={"thinking": {"type": "disabled"}}, + litellm_params={}, + headers={}, + ) + assert "reasoning_content" not in result["messages"][1], ( + "reasoning_content should not be injected when thinking is disabled" + ) + + # Case 4: non-reasoning model -> no injection result = config.transform_request( model="deepseek-chat", messages=messages, diff --git a/tests/test_litellm/llms/deepseek/test_deepseek_chat_transformation.py b/tests/test_litellm/llms/deepseek/test_deepseek_chat_transformation.py new file mode 100644 index 00000000000..e64c6dbc59b --- /dev/null +++ b/tests/test_litellm/llms/deepseek/test_deepseek_chat_transformation.py @@ -0,0 +1,115 @@ +"""Tests for DeepSeek V4 default-on thinking mode / reasoning_content pass-back.""" + +import pytest + +from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig + + +class TestDeepSeekV4DefaultThinkingMode: + def setup_method(self): + self.config = DeepSeekChatConfig() + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_v4_models_registered_with_reasoning(self, model): + from litellm.utils import supports_reasoning + + assert supports_reasoning(model=model, custom_llm_provider="deepseek") + assert supports_reasoning(model=f"deepseek/{model}") + + def test_map_thinking_disabled_passed_through(self): + result = self.config.map_openai_params( + non_default_params={"thinking": {"type": "disabled"}}, + optional_params={}, + model="deepseek-v4-flash", + drop_params=False, + ) + + assert result["thinking"] == {"type": "disabled"} + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_thinking_active_by_default_for_v4(self, model): + assert self.config._thinking_mode_active(model=model, optional_params={}) + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_thinking_inactive_when_explicitly_disabled(self, model): + assert not self.config._thinking_mode_active( + model=model, optional_params={"thinking": {"type": "disabled"}} + ) + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_thinking_active_when_explicitly_enabled(self, model): + assert self.config._thinking_mode_active( + model=model, optional_params={"thinking": {"type": "enabled"}} + ) + + def test_non_reasoning_model_unaffected(self): + assert not self.config._thinking_mode_active( + model="deepseek-chat", optional_params={} + ) + + def _tool_call_history(self): + return [ + {"role": "user", "content": "What's the weather in Tokyo?"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_client_generated_id", + "type": "function", + "function": { + "name": "get_weather", + "arguments": '{"city": "Tokyo"}', + }, + } + ], + }, + { + "role": "tool", + "tool_call_id": "call_client_generated_id", + "content": '{"weather": "Sunny", "temp_c": 28}', + }, + ] + + def test_transform_request_injects_reasoning_content_for_v4_by_default(self): + body = self.config.transform_request( + model="deepseek-v4-flash", + messages=self._tool_call_history(), + optional_params={}, + litellm_params={}, + headers={}, + ) + assert body["messages"][1]["reasoning_content"] == " " + + def test_transform_request_no_injection_when_thinking_disabled(self): + body = self.config.transform_request( + model="deepseek-v4-flash", + messages=self._tool_call_history(), + optional_params={"thinking": {"type": "disabled"}}, + litellm_params={}, + headers={}, + ) + assert "reasoning_content" not in body["messages"][1] + + def test_transform_request_preserves_existing_reasoning_content(self): + messages = self._tool_call_history() + messages[1]["reasoning_content"] = "I should check the weather tool." + body = self.config.transform_request( + model="deepseek-v4-pro", + messages=messages, + optional_params={}, + litellm_params={}, + headers={}, + ) + assert body["messages"][1]["reasoning_content"] == "I should check the weather tool." + + @pytest.mark.asyncio + async def test_async_transform_request_injects_reasoning_content(self): + body = await self.config.async_transform_request( + model="deepseek-v4-flash", + messages=self._tool_call_history(), + optional_params={}, + litellm_params={}, + headers={}, + ) + assert body["messages"][1]["reasoning_content"] == " "