This commit is contained in:
milan-berri 2026-06-26 23:40:44 -04:00 • committed by GitHub
commit d500b3d600
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 331 additions and 31 deletions

View file

@ -49,14 +49,13 @@ class DeepSeekChatConfig(OpenAIGPTConfig):
thinking_value = optional_params.pop("thinking", None)
reasoning_effort = optional_params.pop("reasoning_effort", None)
# Handle thinking parameter - only accept {"type": "enabled"}
# DeepSeek only accepts the `type` key, ignore budget_tokens
if thinking_value is not None:
if (
isinstance(thinking_value, dict)
and thinking_value.get("type") == "enabled"
if isinstance(thinking_value, dict) and thinking_value.get("type") in (
"enabled",
"disabled",
):
# DeepSeek only accepts {"type": "enabled"}, ignore budget_tokens
optional_params["thinking"] = {"type": "enabled"}
optional_params["thinking"] = {"type": thinking_value["type"]}
# Handle reasoning_effort - map to thinking enabled
elif reasoning_effort is not None and reasoning_effort != "none":
@ -137,14 +136,13 @@ class DeepSeekChatConfig(OpenAIGPTConfig):
def _thinking_mode_active(self, model: str, optional_params: dict) -> bool:
"""
Returns True only when thinking mode is actually active for this request:
- model supports reasoning (capability check)
- user explicitly passed thinking={"type": "enabled"} (opt-in check)
DeepSeek V4 enables thinking by default, so any reasoning-capable model
counts unless thinking is explicitly disabled. The API ignores
`reasoning_content` in non-thinking requests, so over-injecting is safe.
"""
return (
supports_reasoning(model=model, custom_llm_provider="deepseek")
and (optional_params.get("thinking") or {}).get("type") == "enabled"
)
if (optional_params.get("thinking") or {}).get("type") == "disabled":
return False
return supports_reasoning(model=model, custom_llm_provider="deepseek")
@staticmethod
def _drop_unsupported_tools(optional_params: dict) -> dict:
@ -235,13 +233,8 @@ class DeepSeekChatConfig(OpenAIGPTConfig):
headers: dict,
) -> dict:
"""
Ensures `reasoning_content` is forwarded on assistant messages for
multi-turn thinking-mode conversations (issue #28045).
Only runs when thinking mode is actually active - guarded by both
supports_reasoning() (model capability) and optional_params["thinking"]
(user explicitly enabled it), preventing spurious injection on models
like deepseek-v3.2 that support thinking as opt-in but not always-on.
Forwards `reasoning_content` on assistant messages for multi-turn
thinking-mode conversations.
"""
optional_params = self._drop_unsupported_tools(optional_params)
if self._thinking_mode_active(model=model, optional_params=optional_params):

View file

@ -11300,6 +11300,52 @@
"supports_system_messages": true,
"supports_tool_choice": false
},
"deepseek-v4-flash": {
"cache_read_input_token_cost": 2.8e-09,
"input_cost_per_token": 1.4e-07,
"litellm_provider": "deepseek",
"max_input_tokens": 1048576,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"output_cost_per_token": 2.8e-07,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"deepseek-v4-pro": {
"cache_read_input_token_cost": 3.625e-09,
"input_cost_per_token": 4.35e-07,
"litellm_provider": "deepseek",
"max_input_tokens": 1048576,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"output_cost_per_token": 8.7e-07,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"dashscope/qwen-coder": {
"input_cost_per_token": 3e-07,
"litellm_provider": "dashscope",
@ -13892,6 +13938,52 @@
"supports_reasoning": true,
"supports_tool_choice": true
},
"deepseek/deepseek-v4-flash": {
"cache_read_input_token_cost": 2.8e-09,
"input_cost_per_token": 1.4e-07,
"litellm_provider": "deepseek",
"max_input_tokens": 1048576,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"output_cost_per_token": 2.8e-07,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"deepseek/deepseek-v4-pro": {
"cache_read_input_token_cost": 3.625e-09,
"input_cost_per_token": 4.35e-07,
"litellm_provider": "deepseek",
"max_input_tokens": 1048576,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"output_cost_per_token": 8.7e-07,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"deepseek.v3-v1:0": {
"input_cost_per_token": 5.8e-07,
"litellm_provider": "bedrock_converse",

View file

@ -11300,6 +11300,52 @@
"supports_system_messages": true,
"supports_tool_choice": false
},
"deepseek-v4-flash": {
"cache_read_input_token_cost": 2.8e-09,
"input_cost_per_token": 1.4e-07,
"litellm_provider": "deepseek",
"max_input_tokens": 1048576,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"output_cost_per_token": 2.8e-07,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"deepseek-v4-pro": {
"cache_read_input_token_cost": 3.625e-09,
"input_cost_per_token": 4.35e-07,
"litellm_provider": "deepseek",
"max_input_tokens": 1048576,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"output_cost_per_token": 8.7e-07,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"dashscope/qwen-coder": {
"input_cost_per_token": 3e-07,
"litellm_provider": "dashscope",
@ -13892,6 +13938,52 @@
"supports_reasoning": true,
"supports_tool_choice": true
},
"deepseek/deepseek-v4-flash": {
"cache_read_input_token_cost": 2.8e-09,
"input_cost_per_token": 1.4e-07,
"litellm_provider": "deepseek",
"max_input_tokens": 1048576,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"output_cost_per_token": 2.8e-07,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"deepseek/deepseek-v4-pro": {
"cache_read_input_token_cost": 3.625e-09,
"input_cost_per_token": 4.35e-07,
"litellm_provider": "deepseek",
"max_input_tokens": 1048576,
"max_output_tokens": 393216,
"max_tokens": 393216,
"mode": "chat",
"output_cost_per_token": 8.7e-07,
"source": "https://api-docs.deepseek.com/quick_start/pricing",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_assistant_prefill": true,
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true
},
"deepseek.v3-v1:0": {
"input_cost_per_token": 5.8e-07,
"litellm_provider": "bedrock_converse",

View file

@ -233,13 +233,8 @@ def test_deepseek_fill_reasoning_content_multiturn():
def test_deepseek_fill_reasoning_content_guard_in_transform_request():
"""
_fill_reasoning_content must only run when BOTH conditions are true:
1. supports_reasoning() is True for the model
2. thinking mode is explicitly enabled in optional_params ({"type": "enabled"})
This prevents spurious injection on models like deepseek-v3.2 that support
thinking as opt-in but not always-on. Addresses oss-pr-review-agent feedback
on PR #28057.
_fill_reasoning_content runs for any reasoning-capable DeepSeek model
unless thinking is explicitly disabled (V4 enables thinking by default).
"""
from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig
@ -263,7 +258,8 @@ def test_deepseek_fill_reasoning_content_guard_in_transform_request():
"reasoning_content should be injected when thinking is enabled"
)
# Case 2: reasoning model + thinking NOT in optional_params -> no injection
# Case 2: reasoning model + thinking NOT in optional_params -> injection
# (thinking is on by default for DeepSeek V4 / deepseek-reasoner)
result = config.transform_request(
model="deepseek-reasoner",
messages=messages,
@ -271,11 +267,23 @@ def test_deepseek_fill_reasoning_content_guard_in_transform_request():
litellm_params={},
headers={},
)
assert "reasoning_content" not in result["messages"][1], (
"reasoning_content should not be injected when thinking is not enabled"
assert result["messages"][1].get("reasoning_content") == " ", (
"reasoning_content should be injected by default for reasoning models"
)
# Case 3: non-reasoning model + thinking enabled -> no injection
# Case 3: reasoning model + thinking explicitly disabled -> no injection
result = config.transform_request(
model="deepseek-reasoner",
messages=messages,
optional_params={"thinking": {"type": "disabled"}},
litellm_params={},
headers={},
)
assert "reasoning_content" not in result["messages"][1], (
"reasoning_content should not be injected when thinking is disabled"
)
# Case 4: non-reasoning model -> no injection
result = config.transform_request(
model="deepseek-chat",
messages=messages,

View file

@ -0,0 +1,115 @@
"""Tests for DeepSeek V4 default-on thinking mode / reasoning_content pass-back."""
import pytest
from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig
class TestDeepSeekV4DefaultThinkingMode:
def setup_method(self):
self.config = DeepSeekChatConfig()
@pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"])
def test_v4_models_registered_with_reasoning(self, model):
from litellm.utils import supports_reasoning
assert supports_reasoning(model=model, custom_llm_provider="deepseek")
assert supports_reasoning(model=f"deepseek/{model}")
def test_map_thinking_disabled_passed_through(self):
result = self.config.map_openai_params(
non_default_params={"thinking": {"type": "disabled"}},
optional_params={},
model="deepseek-v4-flash",
drop_params=False,
)
assert result["thinking"] == {"type": "disabled"}
@pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"])
def test_thinking_active_by_default_for_v4(self, model):
assert self.config._thinking_mode_active(model=model, optional_params={})
@pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"])
def test_thinking_inactive_when_explicitly_disabled(self, model):
assert not self.config._thinking_mode_active(
model=model, optional_params={"thinking": {"type": "disabled"}}
)
@pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"])
def test_thinking_active_when_explicitly_enabled(self, model):
assert self.config._thinking_mode_active(
model=model, optional_params={"thinking": {"type": "enabled"}}
)
def test_non_reasoning_model_unaffected(self):
assert not self.config._thinking_mode_active(
model="deepseek-chat", optional_params={}
)
def _tool_call_history(self):
return [
{"role": "user", "content": "What's the weather in Tokyo?"},
{
"role": "assistant",
"content": None,
"tool_calls": [
{
"id": "call_client_generated_id",
"type": "function",
"function": {
"name": "get_weather",
"arguments": '{"city": "Tokyo"}',
},
}
],
},
{
"role": "tool",
"tool_call_id": "call_client_generated_id",
"content": '{"weather": "Sunny", "temp_c": 28}',
},
]
def test_transform_request_injects_reasoning_content_for_v4_by_default(self):
body = self.config.transform_request(
model="deepseek-v4-flash",
messages=self._tool_call_history(),
optional_params={},
litellm_params={},
headers={},
)
assert body["messages"][1]["reasoning_content"] == " "
def test_transform_request_no_injection_when_thinking_disabled(self):
body = self.config.transform_request(
model="deepseek-v4-flash",
messages=self._tool_call_history(),
optional_params={"thinking": {"type": "disabled"}},
litellm_params={},
headers={},
)
assert "reasoning_content" not in body["messages"][1]
def test_transform_request_preserves_existing_reasoning_content(self):
messages = self._tool_call_history()
messages[1]["reasoning_content"] = "I should check the weather tool."
body = self.config.transform_request(
model="deepseek-v4-pro",
messages=messages,
optional_params={},
litellm_params={},
headers={},
)
assert body["messages"][1]["reasoning_content"] == "I should check the weather tool."
@pytest.mark.asyncio
async def test_async_transform_request_injects_reasoning_content(self):
body = await self.config.async_transform_request(
model="deepseek-v4-flash",
messages=self._tool_call_history(),
optional_params={},
litellm_params={},
headers={},
)
assert body["messages"][1]["reasoning_content"] == " "