mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
Merge c000bf9567 into 0216c969b8
This commit is contained in:
commit
d500b3d600
5 changed files with 331 additions and 31 deletions
|
|
@ -49,14 +49,13 @@ class DeepSeekChatConfig(OpenAIGPTConfig):
|
|||
thinking_value = optional_params.pop("thinking", None)
|
||||
reasoning_effort = optional_params.pop("reasoning_effort", None)
|
||||
|
||||
# Handle thinking parameter - only accept {"type": "enabled"}
|
||||
# DeepSeek only accepts the `type` key, ignore budget_tokens
|
||||
if thinking_value is not None:
|
||||
if (
|
||||
isinstance(thinking_value, dict)
|
||||
and thinking_value.get("type") == "enabled"
|
||||
if isinstance(thinking_value, dict) and thinking_value.get("type") in (
|
||||
"enabled",
|
||||
"disabled",
|
||||
):
|
||||
# DeepSeek only accepts {"type": "enabled"}, ignore budget_tokens
|
||||
optional_params["thinking"] = {"type": "enabled"}
|
||||
optional_params["thinking"] = {"type": thinking_value["type"]}
|
||||
|
||||
# Handle reasoning_effort - map to thinking enabled
|
||||
elif reasoning_effort is not None and reasoning_effort != "none":
|
||||
|
|
@ -137,14 +136,13 @@ class DeepSeekChatConfig(OpenAIGPTConfig):
|
|||
|
||||
def _thinking_mode_active(self, model: str, optional_params: dict) -> bool:
|
||||
"""
|
||||
Returns True only when thinking mode is actually active for this request:
|
||||
- model supports reasoning (capability check)
|
||||
- user explicitly passed thinking={"type": "enabled"} (opt-in check)
|
||||
DeepSeek V4 enables thinking by default, so any reasoning-capable model
|
||||
counts unless thinking is explicitly disabled. The API ignores
|
||||
`reasoning_content` in non-thinking requests, so over-injecting is safe.
|
||||
"""
|
||||
return (
|
||||
supports_reasoning(model=model, custom_llm_provider="deepseek")
|
||||
and (optional_params.get("thinking") or {}).get("type") == "enabled"
|
||||
)
|
||||
if (optional_params.get("thinking") or {}).get("type") == "disabled":
|
||||
return False
|
||||
return supports_reasoning(model=model, custom_llm_provider="deepseek")
|
||||
|
||||
@staticmethod
|
||||
def _drop_unsupported_tools(optional_params: dict) -> dict:
|
||||
|
|
@ -235,13 +233,8 @@ class DeepSeekChatConfig(OpenAIGPTConfig):
|
|||
headers: dict,
|
||||
) -> dict:
|
||||
"""
|
||||
Ensures `reasoning_content` is forwarded on assistant messages for
|
||||
multi-turn thinking-mode conversations (issue #28045).
|
||||
|
||||
Only runs when thinking mode is actually active - guarded by both
|
||||
supports_reasoning() (model capability) and optional_params["thinking"]
|
||||
(user explicitly enabled it), preventing spurious injection on models
|
||||
like deepseek-v3.2 that support thinking as opt-in but not always-on.
|
||||
Forwards `reasoning_content` on assistant messages for multi-turn
|
||||
thinking-mode conversations.
|
||||
"""
|
||||
optional_params = self._drop_unsupported_tools(optional_params)
|
||||
if self._thinking_mode_active(model=model, optional_params=optional_params):
|
||||
|
|
|
|||
|
|
@ -11300,6 +11300,52 @@
|
|||
"supports_system_messages": true,
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepseek-v4-flash": {
|
||||
"cache_read_input_token_cost": 2.8e-09,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.8e-07,
|
||||
"source": "https://api-docs.deepseek.com/quick_start/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek-v4-pro": {
|
||||
"cache_read_input_token_cost": 3.625e-09,
|
||||
"input_cost_per_token": 4.35e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 8.7e-07,
|
||||
"source": "https://api-docs.deepseek.com/quick_start/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"dashscope/qwen-coder": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"litellm_provider": "dashscope",
|
||||
|
|
@ -13892,6 +13938,52 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-v4-flash": {
|
||||
"cache_read_input_token_cost": 2.8e-09,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.8e-07,
|
||||
"source": "https://api-docs.deepseek.com/quick_start/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-v4-pro": {
|
||||
"cache_read_input_token_cost": 3.625e-09,
|
||||
"input_cost_per_token": 4.35e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 8.7e-07,
|
||||
"source": "https://api-docs.deepseek.com/quick_start/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek.v3-v1:0": {
|
||||
"input_cost_per_token": 5.8e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
|
|
|
|||
|
|
@ -11300,6 +11300,52 @@
|
|||
"supports_system_messages": true,
|
||||
"supports_tool_choice": false
|
||||
},
|
||||
"deepseek-v4-flash": {
|
||||
"cache_read_input_token_cost": 2.8e-09,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.8e-07,
|
||||
"source": "https://api-docs.deepseek.com/quick_start/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek-v4-pro": {
|
||||
"cache_read_input_token_cost": 3.625e-09,
|
||||
"input_cost_per_token": 4.35e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 8.7e-07,
|
||||
"source": "https://api-docs.deepseek.com/quick_start/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"dashscope/qwen-coder": {
|
||||
"input_cost_per_token": 3e-07,
|
||||
"litellm_provider": "dashscope",
|
||||
|
|
@ -13892,6 +13938,52 @@
|
|||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-v4-flash": {
|
||||
"cache_read_input_token_cost": 2.8e-09,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.8e-07,
|
||||
"source": "https://api-docs.deepseek.com/quick_start/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-v4-pro": {
|
||||
"cache_read_input_token_cost": 3.625e-09,
|
||||
"input_cost_per_token": 4.35e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 393216,
|
||||
"max_tokens": 393216,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 8.7e-07,
|
||||
"source": "https://api-docs.deepseek.com/quick_start/pricing",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supports_assistant_prefill": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek.v3-v1:0": {
|
||||
"input_cost_per_token": 5.8e-07,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
|
|
|
|||
|
|
@ -233,13 +233,8 @@ def test_deepseek_fill_reasoning_content_multiturn():
|
|||
|
||||
def test_deepseek_fill_reasoning_content_guard_in_transform_request():
|
||||
"""
|
||||
_fill_reasoning_content must only run when BOTH conditions are true:
|
||||
1. supports_reasoning() is True for the model
|
||||
2. thinking mode is explicitly enabled in optional_params ({"type": "enabled"})
|
||||
|
||||
This prevents spurious injection on models like deepseek-v3.2 that support
|
||||
thinking as opt-in but not always-on. Addresses oss-pr-review-agent feedback
|
||||
on PR #28057.
|
||||
_fill_reasoning_content runs for any reasoning-capable DeepSeek model
|
||||
unless thinking is explicitly disabled (V4 enables thinking by default).
|
||||
"""
|
||||
from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig
|
||||
|
||||
|
|
@ -263,7 +258,8 @@ def test_deepseek_fill_reasoning_content_guard_in_transform_request():
|
|||
"reasoning_content should be injected when thinking is enabled"
|
||||
)
|
||||
|
||||
# Case 2: reasoning model + thinking NOT in optional_params -> no injection
|
||||
# Case 2: reasoning model + thinking NOT in optional_params -> injection
|
||||
# (thinking is on by default for DeepSeek V4 / deepseek-reasoner)
|
||||
result = config.transform_request(
|
||||
model="deepseek-reasoner",
|
||||
messages=messages,
|
||||
|
|
@ -271,11 +267,23 @@ def test_deepseek_fill_reasoning_content_guard_in_transform_request():
|
|||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert "reasoning_content" not in result["messages"][1], (
|
||||
"reasoning_content should not be injected when thinking is not enabled"
|
||||
assert result["messages"][1].get("reasoning_content") == " ", (
|
||||
"reasoning_content should be injected by default for reasoning models"
|
||||
)
|
||||
|
||||
# Case 3: non-reasoning model + thinking enabled -> no injection
|
||||
# Case 3: reasoning model + thinking explicitly disabled -> no injection
|
||||
result = config.transform_request(
|
||||
model="deepseek-reasoner",
|
||||
messages=messages,
|
||||
optional_params={"thinking": {"type": "disabled"}},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert "reasoning_content" not in result["messages"][1], (
|
||||
"reasoning_content should not be injected when thinking is disabled"
|
||||
)
|
||||
|
||||
# Case 4: non-reasoning model -> no injection
|
||||
result = config.transform_request(
|
||||
model="deepseek-chat",
|
||||
messages=messages,
|
||||
|
|
|
|||
|
|
@ -0,0 +1,115 @@
|
|||
"""Tests for DeepSeek V4 default-on thinking mode / reasoning_content pass-back."""
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig
|
||||
|
||||
|
||||
class TestDeepSeekV4DefaultThinkingMode:
|
||||
def setup_method(self):
|
||||
self.config = DeepSeekChatConfig()
|
||||
|
||||
@pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"])
|
||||
def test_v4_models_registered_with_reasoning(self, model):
|
||||
from litellm.utils import supports_reasoning
|
||||
|
||||
assert supports_reasoning(model=model, custom_llm_provider="deepseek")
|
||||
assert supports_reasoning(model=f"deepseek/{model}")
|
||||
|
||||
def test_map_thinking_disabled_passed_through(self):
|
||||
result = self.config.map_openai_params(
|
||||
non_default_params={"thinking": {"type": "disabled"}},
|
||||
optional_params={},
|
||||
model="deepseek-v4-flash",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
assert result["thinking"] == {"type": "disabled"}
|
||||
|
||||
@pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"])
|
||||
def test_thinking_active_by_default_for_v4(self, model):
|
||||
assert self.config._thinking_mode_active(model=model, optional_params={})
|
||||
|
||||
@pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"])
|
||||
def test_thinking_inactive_when_explicitly_disabled(self, model):
|
||||
assert not self.config._thinking_mode_active(
|
||||
model=model, optional_params={"thinking": {"type": "disabled"}}
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize("model", ["deepseek-v4-flash", "deepseek-v4-pro"])
|
||||
def test_thinking_active_when_explicitly_enabled(self, model):
|
||||
assert self.config._thinking_mode_active(
|
||||
model=model, optional_params={"thinking": {"type": "enabled"}}
|
||||
)
|
||||
|
||||
def test_non_reasoning_model_unaffected(self):
|
||||
assert not self.config._thinking_mode_active(
|
||||
model="deepseek-chat", optional_params={}
|
||||
)
|
||||
|
||||
def _tool_call_history(self):
|
||||
return [
|
||||
{"role": "user", "content": "What's the weather in Tokyo?"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_client_generated_id",
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"arguments": '{"city": "Tokyo"}',
|
||||
},
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"role": "tool",
|
||||
"tool_call_id": "call_client_generated_id",
|
||||
"content": '{"weather": "Sunny", "temp_c": 28}',
|
||||
},
|
||||
]
|
||||
|
||||
def test_transform_request_injects_reasoning_content_for_v4_by_default(self):
|
||||
body = self.config.transform_request(
|
||||
model="deepseek-v4-flash",
|
||||
messages=self._tool_call_history(),
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert body["messages"][1]["reasoning_content"] == " "
|
||||
|
||||
def test_transform_request_no_injection_when_thinking_disabled(self):
|
||||
body = self.config.transform_request(
|
||||
model="deepseek-v4-flash",
|
||||
messages=self._tool_call_history(),
|
||||
optional_params={"thinking": {"type": "disabled"}},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert "reasoning_content" not in body["messages"][1]
|
||||
|
||||
def test_transform_request_preserves_existing_reasoning_content(self):
|
||||
messages = self._tool_call_history()
|
||||
messages[1]["reasoning_content"] = "I should check the weather tool."
|
||||
body = self.config.transform_request(
|
||||
model="deepseek-v4-pro",
|
||||
messages=messages,
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert body["messages"][1]["reasoning_content"] == "I should check the weather tool."
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_async_transform_request_injects_reasoning_content(self):
|
||||
body = await self.config.async_transform_request(
|
||||
model="deepseek-v4-flash",
|
||||
messages=self._tool_call_history(),
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
assert body["messages"][1]["reasoning_content"] == " "
|
||||
Loading…
Add table
Reference in a new issue