From 2ef49eed603c7b79e016294a3d8c2fe698a7beee Mon Sep 17 00:00:00 2001 From: Milan Date: Thu, 11 Jun 2026 22:51:39 +0300 Subject: [PATCH 1/4] fix(deepseek): treat thinking mode as default-on for DeepSeek V4 models DeepSeek V4 (deepseek-v4-flash / deepseek-v4-pro) runs in thinking mode by default and rejects multi-turn tool-call histories where an assistant message is missing reasoning_content and carries a tool_call id DeepSeek does not recognize (standard behavior for frontends/agent frameworks that regenerate ids), returning HTTP 400 "The reasoning_content in the thinking mode must be passed back to the API." The injection mechanism for this already exists (_fill_reasoning_content, #28080) but never fired for V4 because: 1. deepseek-v4-* models were missing from the model registry, so supports_reasoning() returned False 2. _thinking_mode_active() required an explicit thinking={"type": "enabled"} param, but V4 enables thinking by default Changes: - Add deepseek-v4-flash / deepseek-v4-pro registry entries (bare and deepseek/-prefixed) with official pricing - _thinking_mode_active(): treat the V4 family as thinking-mode by default unless thinking={"type": "disabled"} is passed; opt-in models (e.g. deepseek-v3.2) keep requiring explicit enablement - map_openai_params(): forward thinking={"type": "disabled"} (previously silently dropped) so users can opt out of V4 default-on thinking Fixes #26395 Co-authored-by: Cursor --- litellm/llms/deepseek/chat/transformation.py | 44 ++++-- ...odel_prices_and_context_window_backup.json | 92 ++++++++++++ model_prices_and_context_window.json | 92 ++++++++++++ .../chat/test_deepseek_chat_transformation.py | 133 ++++++++++++++++++ 4 files changed, 348 insertions(+), 13 deletions(-) diff --git a/litellm/llms/deepseek/chat/transformation.py b/litellm/llms/deepseek/chat/transformation.py index 7ed3e484535..d8e11c99323 100644 --- a/litellm/llms/deepseek/chat/transformation.py +++ b/litellm/llms/deepseek/chat/transformation.py @@ -49,14 +49,15 @@ class DeepSeekChatConfig(OpenAIGPTConfig): thinking_value = optional_params.pop("thinking", None) reasoning_effort = optional_params.pop("reasoning_effort", None) - # Handle thinking parameter - only accept {"type": "enabled"} + # Handle thinking parameter - accept {"type": "enabled"} and + # {"type": "disabled"} (the latter opts out of V4's default-on thinking) if thinking_value is not None: - if ( - isinstance(thinking_value, dict) - and thinking_value.get("type") == "enabled" + if isinstance(thinking_value, dict) and thinking_value.get("type") in ( + "enabled", + "disabled", ): - # DeepSeek only accepts {"type": "enabled"}, ignore budget_tokens - optional_params["thinking"] = {"type": "enabled"} + # DeepSeek only accepts the `type` key, ignore budget_tokens + optional_params["thinking"] = {"type": thinking_value["type"]} # Handle reasoning_effort - map to thinking enabled elif reasoning_effort is not None and reasoning_effort != "none": @@ -135,16 +136,33 @@ class DeepSeekChatConfig(OpenAIGPTConfig): messages=messages, model=model, is_async=False ) + # Model families where DeepSeek enables thinking mode BY DEFAULT (no + # `thinking` param required). Reference: + # https://api-docs.deepseek.com/guides/thinking_mode + DEFAULT_THINKING_MODEL_PREFIXES = ("deepseek-v4",) + + def _is_default_thinking_model(self, model: str) -> bool: + return any( + prefix in model for prefix in self.DEFAULT_THINKING_MODEL_PREFIXES + ) + def _thinking_mode_active(self, model: str, optional_params: dict) -> bool: """ - Returns True only when thinking mode is actually active for this request: - - model supports reasoning (capability check) - - user explicitly passed thinking={"type": "enabled"} (opt-in check) + Returns True when thinking mode is active for this request: + - user explicitly passed thinking={"type": "enabled"} on a model that + supports reasoning, OR + - the model runs in thinking mode by default (DeepSeek V4 family) and + the user did not explicitly disable it. + + Models like deepseek-v3.2 (supports_reasoning but opt-in thinking) + remain untouched unless thinking is explicitly enabled. """ - return ( - supports_reasoning(model=model, custom_llm_provider="deepseek") - and (optional_params.get("thinking") or {}).get("type") == "enabled" - ) + thinking_type = (optional_params.get("thinking") or {}).get("type") + if thinking_type == "disabled": + return False + if thinking_type == "enabled": + return supports_reasoning(model=model, custom_llm_provider="deepseek") + return self._is_default_thinking_model(model) def transform_request( self, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index aab0e4264d0..a0df50e481c 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -10854,6 +10854,52 @@ "supports_system_messages": true, "supports_tool_choice": false }, + "deepseek-v4-flash": { + "cache_read_input_token_cost": 2.8e-09, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "deepseek-v4-pro": { + "cache_read_input_token_cost": 3.625e-09, + "input_cost_per_token": 4.35e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 8.7e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "dashscope/qwen-coder": { "input_cost_per_token": 3e-07, "litellm_provider": "dashscope", @@ -13446,6 +13492,52 @@ "supports_reasoning": true, "supports_tool_choice": true }, + "deepseek/deepseek-v4-flash": { + "cache_read_input_token_cost": 2.8e-09, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "deepseek/deepseek-v4-pro": { + "cache_read_input_token_cost": 3.625e-09, + "input_cost_per_token": 4.35e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 8.7e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "deepseek.v3-v1:0": { "input_cost_per_token": 5.8e-07, "litellm_provider": "bedrock_converse", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index f0b2432ddc8..15c68c18254 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -10854,6 +10854,52 @@ "supports_system_messages": true, "supports_tool_choice": false }, + "deepseek-v4-flash": { + "cache_read_input_token_cost": 2.8e-09, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "deepseek-v4-pro": { + "cache_read_input_token_cost": 3.625e-09, + "input_cost_per_token": 4.35e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 8.7e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "dashscope/qwen-coder": { "input_cost_per_token": 3e-07, "litellm_provider": "dashscope", @@ -13446,6 +13492,52 @@ "supports_reasoning": true, "supports_tool_choice": true }, + "deepseek/deepseek-v4-flash": { + "cache_read_input_token_cost": 2.8e-09, + "input_cost_per_token": 1.4e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "deepseek/deepseek-v4-pro": { + "cache_read_input_token_cost": 3.625e-09, + "input_cost_per_token": 4.35e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1048576, + "max_output_tokens": 393216, + "max_tokens": 393216, + "mode": "chat", + "output_cost_per_token": 8.7e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "deepseek.v3-v1:0": { "input_cost_per_token": 5.8e-07, "litellm_provider": "bedrock_converse", diff --git a/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py b/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py index a2f45e7188b..01a2c522d8d 100644 --- a/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py +++ b/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py @@ -166,3 +166,136 @@ class TestDeepSeekThinkingParams: ) assert "thinking" not in result + + def test_map_thinking_disabled_passed_through(self): + """thinking={"type": "disabled"} must be forwarded so users can opt out + of DeepSeek V4's default-on thinking mode.""" + result = self.config.map_openai_params( + non_default_params={"thinking": {"type": "disabled"}}, + optional_params={}, + model="deepseek-v4-flash", + drop_params=False, + ) + + assert result["thinking"] == {"type": "disabled"} + + +class TestDeepSeekV4DefaultThinkingMode: + """ + DeepSeek V4 models run in thinking mode BY DEFAULT and require + `reasoning_content` to be passed back on assistant messages + (https://github.com/BerriAI/litellm/issues/26395). + """ + + def setup_method(self): + self.config = DeepSeekChatConfig() + + # --- registry --- + + @pytest.mark.parametrize( + "model", + ["deepseek-v4-flash", "deepseek-v4-pro"], + ) + def test_v4_models_registered_with_reasoning(self, model): + from litellm.utils import supports_reasoning + + assert supports_reasoning(model=model, custom_llm_provider="deepseek") + assert supports_reasoning(model=f"deepseek/{model}") + + # --- _thinking_mode_active guard --- + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_thinking_active_by_default_for_v4(self, model): + assert self.config._thinking_mode_active(model=model, optional_params={}) + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_thinking_inactive_when_explicitly_disabled(self, model): + assert not self.config._thinking_mode_active( + model=model, optional_params={"thinking": {"type": "disabled"}} + ) + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_thinking_active_when_explicitly_enabled(self, model): + assert self.config._thinking_mode_active( + model=model, optional_params={"thinking": {"type": "enabled"}} + ) + + def test_opt_in_models_unaffected_by_default(self): + """deepseek-v3.2 supports reasoning but thinking is opt-in: no thinking + param -> guard must stay off (no spurious injection).""" + assert not self.config._thinking_mode_active( + model="deepseek-v3.2", optional_params={} + ) + assert self.config._thinking_mode_active( + model="deepseek-v3.2", optional_params={"thinking": {"type": "enabled"}} + ) + + def test_non_reasoning_model_unaffected(self): + assert not self.config._thinking_mode_active( + model="deepseek-chat", optional_params={} + ) + + # --- end-to-end transform_request --- + + def _tool_call_history(self): + return [ + {"role": "user", "content": "What's the weather in Tokyo?"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_client_generated_id", + "type": "function", + "function": { + "name": "get_weather", + "arguments": '{"city": "Tokyo"}', + }, + } + ], + # reasoning_content stripped, tool_call id rewritten: + # standard behavior of frontends/agent frameworks, and the + # exact request shape DeepSeek rejects with + # "The `reasoning_content` in the thinking mode must be passed back to the API." + }, + { + "role": "tool", + "tool_call_id": "call_client_generated_id", + "content": '{"weather": "Sunny", "temp_c": 28}', + }, + ] + + def test_transform_request_injects_reasoning_content_for_v4_by_default(self): + body = self.config.transform_request( + model="deepseek-v4-flash", + messages=self._tool_call_history(), + optional_params={}, + litellm_params={}, + headers={}, + ) + assistant_msg = body["messages"][1] + assert assistant_msg["reasoning_content"] == " " + + def test_transform_request_no_injection_when_thinking_disabled(self): + body = self.config.transform_request( + model="deepseek-v4-flash", + messages=self._tool_call_history(), + optional_params={"thinking": {"type": "disabled"}}, + litellm_params={}, + headers={}, + ) + assistant_msg = body["messages"][1] + assert "reasoning_content" not in assistant_msg + + def test_transform_request_preserves_existing_reasoning_content(self): + messages = self._tool_call_history() + messages[1]["reasoning_content"] = "I should check the weather tool." + body = self.config.transform_request( + model="deepseek-v4-pro", + messages=messages, + optional_params={}, + litellm_params={}, + headers={}, + ) + assistant_msg = body["messages"][1] + assert assistant_msg["reasoning_content"] == "I should check the weather tool." From 6aa6d40c5dd1d2b81543d89fea02ef180eb7d373 Mon Sep 17 00:00:00 2001 From: Milan Date: Thu, 11 Jun 2026 23:04:35 +0300 Subject: [PATCH 2/4] refactor(deepseek): drop hardcoded V4 model list, guard on supports_reasoning Replace the DEFAULT_THINKING_MODEL_PREFIXES name list with the existing supports_reasoning registry flag (same approach as the Moonshot reasoning_content fix), so new DeepSeek models only need a cost map entry, not a code change. Safety verified against the live DeepSeek API: - reasoning_content is ignored in non-thinking requests (incl. explicit thinking={"type": "disabled"}), so injecting for any reasoning-capable model is harmless - deepseek-v3.2 (the opt-in model the stricter guard protected) is no longer served: the API only accepts deepseek-v4-pro / deepseek-v4-flash thinking={"type": "disabled"} still skips injection entirely. Co-authored-by: Cursor --- litellm/llms/deepseek/chat/transformation.py | 34 +++++++------------ .../chat/test_deepseek_chat_transformation.py | 10 +++--- .../test_deepseek_completion.py | 34 +++++++++++++------ 3 files changed, 42 insertions(+), 36 deletions(-) diff --git a/litellm/llms/deepseek/chat/transformation.py b/litellm/llms/deepseek/chat/transformation.py index d8e11c99323..c8ece202841 100644 --- a/litellm/llms/deepseek/chat/transformation.py +++ b/litellm/llms/deepseek/chat/transformation.py @@ -136,33 +136,23 @@ class DeepSeekChatConfig(OpenAIGPTConfig): messages=messages, model=model, is_async=False ) - # Model families where DeepSeek enables thinking mode BY DEFAULT (no - # `thinking` param required). Reference: - # https://api-docs.deepseek.com/guides/thinking_mode - DEFAULT_THINKING_MODEL_PREFIXES = ("deepseek-v4",) - - def _is_default_thinking_model(self, model: str) -> bool: - return any( - prefix in model for prefix in self.DEFAULT_THINKING_MODEL_PREFIXES - ) - def _thinking_mode_active(self, model: str, optional_params: dict) -> bool: """ - Returns True when thinking mode is active for this request: - - user explicitly passed thinking={"type": "enabled"} on a model that - supports reasoning, OR - - the model runs in thinking mode by default (DeepSeek V4 family) and - the user did not explicitly disable it. + Returns True when thinking mode may be active for this request. - Models like deepseek-v3.2 (supports_reasoning but opt-in thinking) - remain untouched unless thinking is explicitly enabled. + DeepSeek V4 models enable thinking BY DEFAULT (no `thinking` param + required - https://api-docs.deepseek.com/guides/thinking_mode), so any + reasoning-capable model counts unless the user explicitly disabled + thinking. Same approach as the Moonshot reasoning fix + (litellm/llms/moonshot/chat/transformation.py). + + Injecting `reasoning_content` when thinking is NOT active is harmless: + the DeepSeek API ignores the field in non-thinking requests (verified + against the live API, including thinking={"type": "disabled"}). """ - thinking_type = (optional_params.get("thinking") or {}).get("type") - if thinking_type == "disabled": + if (optional_params.get("thinking") or {}).get("type") == "disabled": return False - if thinking_type == "enabled": - return supports_reasoning(model=model, custom_llm_provider="deepseek") - return self._is_default_thinking_model(model) + return supports_reasoning(model=model, custom_llm_provider="deepseek") def transform_request( self, diff --git a/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py b/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py index 01a2c522d8d..ba3dfed2b60 100644 --- a/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py +++ b/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py @@ -220,10 +220,12 @@ class TestDeepSeekV4DefaultThinkingMode: model=model, optional_params={"thinking": {"type": "enabled"}} ) - def test_opt_in_models_unaffected_by_default(self): - """deepseek-v3.2 supports reasoning but thinking is opt-in: no thinking - param -> guard must stay off (no spurious injection).""" - assert not self.config._thinking_mode_active( + def test_reasoning_capable_models_active_by_default(self): + """Any reasoning-capable DeepSeek model counts as potentially + thinking-mode (V4 enables thinking by default). Injection is harmless + when thinking is not actually active: the live API ignores + reasoning_content in non-thinking requests.""" + assert self.config._thinking_mode_active( model="deepseek-v3.2", optional_params={} ) assert self.config._thinking_mode_active( diff --git a/tests/llm_translation/test_deepseek_completion.py b/tests/llm_translation/test_deepseek_completion.py index 2ede5d3f3f8..e574b7e1f63 100644 --- a/tests/llm_translation/test_deepseek_completion.py +++ b/tests/llm_translation/test_deepseek_completion.py @@ -233,13 +233,14 @@ def test_deepseek_fill_reasoning_content_multiturn(): def test_deepseek_fill_reasoning_content_guard_in_transform_request(): """ - _fill_reasoning_content must only run when BOTH conditions are true: - 1. supports_reasoning() is True for the model - 2. thinking mode is explicitly enabled in optional_params ({"type": "enabled"}) + _fill_reasoning_content runs for any reasoning-capable DeepSeek model + unless thinking is explicitly disabled. - This prevents spurious injection on models like deepseek-v3.2 that support - thinking as opt-in but not always-on. Addresses oss-pr-review-agent feedback - on PR #28057. + DeepSeek V4 enables thinking mode BY DEFAULT (no `thinking` param + required), so the guard cannot rely on an explicit opt-in (issue #26395). + Injecting reasoning_content when thinking is not actually active is + harmless: the DeepSeek API ignores the field in non-thinking requests + (verified against the live API, including thinking={"type": "disabled"}). """ from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig @@ -263,7 +264,8 @@ def test_deepseek_fill_reasoning_content_guard_in_transform_request(): "reasoning_content should be injected when thinking is enabled" ) - # Case 2: reasoning model + thinking NOT in optional_params -> no injection + # Case 2: reasoning model + thinking NOT in optional_params -> injection + # (thinking is on by default for DeepSeek V4 / deepseek-reasoner) result = config.transform_request( model="deepseek-reasoner", messages=messages, @@ -271,11 +273,23 @@ def test_deepseek_fill_reasoning_content_guard_in_transform_request(): litellm_params={}, headers={}, ) - assert "reasoning_content" not in result["messages"][1], ( - "reasoning_content should not be injected when thinking is not enabled" + assert result["messages"][1].get("reasoning_content") == " ", ( + "reasoning_content should be injected by default for reasoning models" ) - # Case 3: non-reasoning model + thinking enabled -> no injection + # Case 3: reasoning model + thinking explicitly disabled -> no injection + result = config.transform_request( + model="deepseek-reasoner", + messages=messages, + optional_params={"thinking": {"type": "disabled"}}, + litellm_params={}, + headers={}, + ) + assert "reasoning_content" not in result["messages"][1], ( + "reasoning_content should not be injected when thinking is disabled" + ) + + # Case 4: non-reasoning model -> no injection result = config.transform_request( model="deepseek-chat", messages=messages, From 7824dc07d3802071932dc6495368bf61233f6acd Mon Sep 17 00:00:00 2001 From: Milan Date: Thu, 11 Jun 2026 23:28:36 +0300 Subject: [PATCH 3/4] chore(deepseek): trim comments to essentials Co-authored-by: Cursor --- litellm/llms/deepseek/chat/transformation.py | 27 +++++-------------- .../chat/test_deepseek_chat_transformation.py | 13 ++------- .../test_deepseek_completion.py | 8 +----- 3 files changed, 9 insertions(+), 39 deletions(-) diff --git a/litellm/llms/deepseek/chat/transformation.py b/litellm/llms/deepseek/chat/transformation.py index c8ece202841..44abe5c6878 100644 --- a/litellm/llms/deepseek/chat/transformation.py +++ b/litellm/llms/deepseek/chat/transformation.py @@ -49,14 +49,12 @@ class DeepSeekChatConfig(OpenAIGPTConfig): thinking_value = optional_params.pop("thinking", None) reasoning_effort = optional_params.pop("reasoning_effort", None) - # Handle thinking parameter - accept {"type": "enabled"} and - # {"type": "disabled"} (the latter opts out of V4's default-on thinking) + # DeepSeek only accepts the `type` key, ignore budget_tokens if thinking_value is not None: if isinstance(thinking_value, dict) and thinking_value.get("type") in ( "enabled", "disabled", ): - # DeepSeek only accepts the `type` key, ignore budget_tokens optional_params["thinking"] = {"type": thinking_value["type"]} # Handle reasoning_effort - map to thinking enabled @@ -138,17 +136,9 @@ class DeepSeekChatConfig(OpenAIGPTConfig): def _thinking_mode_active(self, model: str, optional_params: dict) -> bool: """ - Returns True when thinking mode may be active for this request. - - DeepSeek V4 models enable thinking BY DEFAULT (no `thinking` param - required - https://api-docs.deepseek.com/guides/thinking_mode), so any - reasoning-capable model counts unless the user explicitly disabled - thinking. Same approach as the Moonshot reasoning fix - (litellm/llms/moonshot/chat/transformation.py). - - Injecting `reasoning_content` when thinking is NOT active is harmless: - the DeepSeek API ignores the field in non-thinking requests (verified - against the live API, including thinking={"type": "disabled"}). + DeepSeek V4 enables thinking by default, so any reasoning-capable model + counts unless thinking is explicitly disabled. The API ignores + `reasoning_content` in non-thinking requests, so over-injecting is safe. """ if (optional_params.get("thinking") or {}).get("type") == "disabled": return False @@ -163,13 +153,8 @@ class DeepSeekChatConfig(OpenAIGPTConfig): headers: dict, ) -> dict: """ - Ensures `reasoning_content` is forwarded on assistant messages for - multi-turn thinking-mode conversations (issue #28045). - - Only runs when thinking mode is actually active - guarded by both - supports_reasoning() (model capability) and optional_params["thinking"] - (user explicitly enabled it), preventing spurious injection on models - like deepseek-v3.2 that support thinking as opt-in but not always-on. + Forwards `reasoning_content` on assistant messages for multi-turn + thinking-mode conversations. """ if self._thinking_mode_active(model=model, optional_params=optional_params): messages = self._fill_reasoning_content(messages) diff --git a/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py b/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py index ba3dfed2b60..d74fb533e1a 100644 --- a/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py +++ b/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py @@ -168,8 +168,7 @@ class TestDeepSeekThinkingParams: assert "thinking" not in result def test_map_thinking_disabled_passed_through(self): - """thinking={"type": "disabled"} must be forwarded so users can opt out - of DeepSeek V4's default-on thinking mode.""" + """thinking={"type": "disabled"} is forwarded (opt-out of V4 default thinking).""" result = self.config.map_openai_params( non_default_params={"thinking": {"type": "disabled"}}, optional_params={}, @@ -181,11 +180,7 @@ class TestDeepSeekThinkingParams: class TestDeepSeekV4DefaultThinkingMode: - """ - DeepSeek V4 models run in thinking mode BY DEFAULT and require - `reasoning_content` to be passed back on assistant messages - (https://github.com/BerriAI/litellm/issues/26395). - """ + """DeepSeek V4 default-on thinking mode / reasoning_content pass-back.""" def setup_method(self): self.config = DeepSeekChatConfig() @@ -221,10 +216,6 @@ class TestDeepSeekV4DefaultThinkingMode: ) def test_reasoning_capable_models_active_by_default(self): - """Any reasoning-capable DeepSeek model counts as potentially - thinking-mode (V4 enables thinking by default). Injection is harmless - when thinking is not actually active: the live API ignores - reasoning_content in non-thinking requests.""" assert self.config._thinking_mode_active( model="deepseek-v3.2", optional_params={} ) diff --git a/tests/llm_translation/test_deepseek_completion.py b/tests/llm_translation/test_deepseek_completion.py index e574b7e1f63..d3cb913625a 100644 --- a/tests/llm_translation/test_deepseek_completion.py +++ b/tests/llm_translation/test_deepseek_completion.py @@ -234,13 +234,7 @@ def test_deepseek_fill_reasoning_content_multiturn(): def test_deepseek_fill_reasoning_content_guard_in_transform_request(): """ _fill_reasoning_content runs for any reasoning-capable DeepSeek model - unless thinking is explicitly disabled. - - DeepSeek V4 enables thinking mode BY DEFAULT (no `thinking` param - required), so the guard cannot rely on an explicit opt-in (issue #26395). - Injecting reasoning_content when thinking is not actually active is - harmless: the DeepSeek API ignores the field in non-thinking requests - (verified against the live API, including thinking={"type": "disabled"}). + unless thinking is explicitly disabled (V4 enables thinking by default). """ from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig From c000bf95671314849bf5852b5777bcd9f0919c8c Mon Sep 17 00:00:00 2001 From: Milan Date: Thu, 11 Jun 2026 23:41:55 +0300 Subject: [PATCH 4/4] test(deepseek): move V4 thinking-mode tests to tests/test_litellm for CI coverage Codecov patch coverage runs the tests/test_litellm suite; the new tests lived in tests/litellm and were not executed there. Also adds an async_transform_request test. Co-authored-by: Cursor --- .../chat/test_deepseek_chat_transformation.py | 126 ------------------ .../test_deepseek_chat_transformation.py | 115 ++++++++++++++++ 2 files changed, 115 insertions(+), 126 deletions(-) create mode 100644 tests/test_litellm/llms/deepseek/test_deepseek_chat_transformation.py diff --git a/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py b/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py index d74fb533e1a..a2f45e7188b 100644 --- a/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py +++ b/tests/litellm/llms/deepseek/chat/test_deepseek_chat_transformation.py @@ -166,129 +166,3 @@ class TestDeepSeekThinkingParams: ) assert "thinking" not in result - - def test_map_thinking_disabled_passed_through(self): - """thinking={"type": "disabled"} is forwarded (opt-out of V4 default thinking).""" - result = self.config.map_openai_params( - non_default_params={"thinking": {"type": "disabled"}}, - optional_params={}, - model="deepseek-v4-flash", - drop_params=False, - ) - - assert result["thinking"] == {"type": "disabled"} - - -class TestDeepSeekV4DefaultThinkingMode: - """DeepSeek V4 default-on thinking mode / reasoning_content pass-back.""" - - def setup_method(self): - self.config = DeepSeekChatConfig() - - # --- registry --- - - @pytest.mark.parametrize( - "model", - ["deepseek-v4-flash", "deepseek-v4-pro"], - ) - def test_v4_models_registered_with_reasoning(self, model): - from litellm.utils import supports_reasoning - - assert supports_reasoning(model=model, custom_llm_provider="deepseek") - assert supports_reasoning(model=f"deepseek/{model}") - - # --- _thinking_mode_active guard --- - - @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) - def test_thinking_active_by_default_for_v4(self, model): - assert self.config._thinking_mode_active(model=model, optional_params={}) - - @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) - def test_thinking_inactive_when_explicitly_disabled(self, model): - assert not self.config._thinking_mode_active( - model=model, optional_params={"thinking": {"type": "disabled"}} - ) - - @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) - def test_thinking_active_when_explicitly_enabled(self, model): - assert self.config._thinking_mode_active( - model=model, optional_params={"thinking": {"type": "enabled"}} - ) - - def test_reasoning_capable_models_active_by_default(self): - assert self.config._thinking_mode_active( - model="deepseek-v3.2", optional_params={} - ) - assert self.config._thinking_mode_active( - model="deepseek-v3.2", optional_params={"thinking": {"type": "enabled"}} - ) - - def test_non_reasoning_model_unaffected(self): - assert not self.config._thinking_mode_active( - model="deepseek-chat", optional_params={} - ) - - # --- end-to-end transform_request --- - - def _tool_call_history(self): - return [ - {"role": "user", "content": "What's the weather in Tokyo?"}, - { - "role": "assistant", - "content": None, - "tool_calls": [ - { - "id": "call_client_generated_id", - "type": "function", - "function": { - "name": "get_weather", - "arguments": '{"city": "Tokyo"}', - }, - } - ], - # reasoning_content stripped, tool_call id rewritten: - # standard behavior of frontends/agent frameworks, and the - # exact request shape DeepSeek rejects with - # "The `reasoning_content` in the thinking mode must be passed back to the API." - }, - { - "role": "tool", - "tool_call_id": "call_client_generated_id", - "content": '{"weather": "Sunny", "temp_c": 28}', - }, - ] - - def test_transform_request_injects_reasoning_content_for_v4_by_default(self): - body = self.config.transform_request( - model="deepseek-v4-flash", - messages=self._tool_call_history(), - optional_params={}, - litellm_params={}, - headers={}, - ) - assistant_msg = body["messages"][1] - assert assistant_msg["reasoning_content"] == " " - - def test_transform_request_no_injection_when_thinking_disabled(self): - body = self.config.transform_request( - model="deepseek-v4-flash", - messages=self._tool_call_history(), - optional_params={"thinking": {"type": "disabled"}}, - litellm_params={}, - headers={}, - ) - assistant_msg = body["messages"][1] - assert "reasoning_content" not in assistant_msg - - def test_transform_request_preserves_existing_reasoning_content(self): - messages = self._tool_call_history() - messages[1]["reasoning_content"] = "I should check the weather tool." - body = self.config.transform_request( - model="deepseek-v4-pro", - messages=messages, - optional_params={}, - litellm_params={}, - headers={}, - ) - assistant_msg = body["messages"][1] - assert assistant_msg["reasoning_content"] == "I should check the weather tool." diff --git a/tests/test_litellm/llms/deepseek/test_deepseek_chat_transformation.py b/tests/test_litellm/llms/deepseek/test_deepseek_chat_transformation.py new file mode 100644 index 00000000000..e64c6dbc59b --- /dev/null +++ b/tests/test_litellm/llms/deepseek/test_deepseek_chat_transformation.py @@ -0,0 +1,115 @@ +"""Tests for DeepSeek V4 default-on thinking mode / reasoning_content pass-back.""" + +import pytest + +from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig + + +class TestDeepSeekV4DefaultThinkingMode: + def setup_method(self): + self.config = DeepSeekChatConfig() + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_v4_models_registered_with_reasoning(self, model): + from litellm.utils import supports_reasoning + + assert supports_reasoning(model=model, custom_llm_provider="deepseek") + assert supports_reasoning(model=f"deepseek/{model}") + + def test_map_thinking_disabled_passed_through(self): + result = self.config.map_openai_params( + non_default_params={"thinking": {"type": "disabled"}}, + optional_params={}, + model="deepseek-v4-flash", + drop_params=False, + ) + + assert result["thinking"] == {"type": "disabled"} + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_thinking_active_by_default_for_v4(self, model): + assert self.config._thinking_mode_active(model=model, optional_params={}) + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_thinking_inactive_when_explicitly_disabled(self, model): + assert not self.config._thinking_mode_active( + model=model, optional_params={"thinking": {"type": "disabled"}} + ) + + @pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"]) + def test_thinking_active_when_explicitly_enabled(self, model): + assert self.config._thinking_mode_active( + model=model, optional_params={"thinking": {"type": "enabled"}} + ) + + def test_non_reasoning_model_unaffected(self): + assert not self.config._thinking_mode_active( + model="deepseek-chat", optional_params={} + ) + + def _tool_call_history(self): + return [ + {"role": "user", "content": "What's the weather in Tokyo?"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_client_generated_id", + "type": "function", + "function": { + "name": "get_weather", + "arguments": '{"city": "Tokyo"}', + }, + } + ], + }, + { + "role": "tool", + "tool_call_id": "call_client_generated_id", + "content": '{"weather": "Sunny", "temp_c": 28}', + }, + ] + + def test_transform_request_injects_reasoning_content_for_v4_by_default(self): + body = self.config.transform_request( + model="deepseek-v4-flash", + messages=self._tool_call_history(), + optional_params={}, + litellm_params={}, + headers={}, + ) + assert body["messages"][1]["reasoning_content"] == " " + + def test_transform_request_no_injection_when_thinking_disabled(self): + body = self.config.transform_request( + model="deepseek-v4-flash", + messages=self._tool_call_history(), + optional_params={"thinking": {"type": "disabled"}}, + litellm_params={}, + headers={}, + ) + assert "reasoning_content" not in body["messages"][1] + + def test_transform_request_preserves_existing_reasoning_content(self): + messages = self._tool_call_history() + messages[1]["reasoning_content"] = "I should check the weather tool." + body = self.config.transform_request( + model="deepseek-v4-pro", + messages=messages, + optional_params={}, + litellm_params={}, + headers={}, + ) + assert body["messages"][1]["reasoning_content"] == "I should check the weather tool." + + @pytest.mark.asyncio + async def test_async_transform_request_injects_reasoning_content(self): + body = await self.config.async_transform_request( + model="deepseek-v4-flash", + messages=self._tool_call_history(), + optional_params={}, + litellm_params={}, + headers={}, + ) + assert body["messages"][1]["reasoning_content"] == " "