From 00a1c311e36b3d4ab193ceda2cfdd46bf0357456 Mon Sep 17 00:00:00 2001 From: Hazelhof <110658790+Hazelhof@users.noreply.github.com> Date: Sat, 29 Aug 2026 17:04:08 +1000 Subject: [PATCH 1/3] fix(deepseek): pass reasoning_effort through to the API DeepSeekChatConfig.map_openai_params mapped reasoning_effort to a binary thinking {type: enabled/disabled} toggle and discarded the effort value, so DeepSeek always ran default high effort. High-effort reasoning can run away and exhaust max_tokens during reasoning, returning empty content. Preserve the reasoning_effort value into the request body when thinking is enabled so callers can bound effort (e.g. reasoning_effort="low"). Backward-compatible. See https://api-docs.deepseek.com/guides/thinking_mode --- litellm/llms/deepseek/chat/transformation.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/litellm/llms/deepseek/chat/transformation.py b/litellm/llms/deepseek/chat/transformation.py index ea19a7c7ddf..80775889118 100644 --- a/litellm/llms/deepseek/chat/transformation.py +++ b/litellm/llms/deepseek/chat/transformation.py @@ -51,13 +51,28 @@ class DeepSeekChatConfig(OpenAIGPTConfig): thinking_value: Final = optional_params.pop("thinking", None) reasoning_effort: Final = optional_params.pop("reasoning_effort", None) + thinking_enabled = False # Handle thinking parameter - accept both enabled and disabled, ignore budget_tokens if isinstance(thinking_value, dict) and thinking_value.get("type") in ("enabled", "disabled"): optional_params["thinking"] = {"type": thinking_value["type"]} + thinking_enabled = thinking_value["type"] == "enabled" # Otherwise fall back to reasoning_effort: "none" disables, anything else enables elif reasoning_effort is not None: optional_params["thinking"] = {"type": "disabled" if reasoning_effort == "none" else "enabled"} + thinking_enabled = reasoning_effort != "none" + else: + thinking_enabled = True # DeepSeek thinking mode is on by default + + # reasoning_effort is a real DeepSeek top-level param (default "high"; valid + # values low/medium/high/xhigh/max). The branch above only used it as an on/off + # toggle and DROPPED the value, so DeepSeek always ran default HIGH effort — which + # can run away, exhausting max_tokens during reasoning and returning empty content + # (finish_reason="length", content:null). Preserve the value into the request body + # when thinking is on so callers can bound effort (e.g. reasoning_effort="low"). + # See https://api-docs.deepseek.com/guides/thinking_mode + if thinking_enabled and reasoning_effort is not None: + optional_params["reasoning_effort"] = reasoning_effort return optional_params From 43f319b57fe8c3144413f296933842e8698af242 Mon Sep 17 00:00:00 2001 From: Hazelhof <110658790+Hazelhof@users.noreply.github.com> Date: Sat, 29 Aug 2026 17:04:48 +1000 Subject: [PATCH 2/3] test(deepseek): reasoning_effort passthrough coverage --- .../test_deepseek_reasoning_effort.py | 41 +++++++++++++++++++ 1 file changed, 41 insertions(+) create mode 100644 tests/llm_translation/test_deepseek_reasoning_effort.py diff --git a/tests/llm_translation/test_deepseek_reasoning_effort.py b/tests/llm_translation/test_deepseek_reasoning_effort.py new file mode 100644 index 00000000000..f9d3f29d2ab --- /dev/null +++ b/tests/llm_translation/test_deepseek_reasoning_effort.py @@ -0,0 +1,41 @@ +"""Verify DeepSeek `reasoning_effort` passes through the transformation. + +Covers the fix that preserves the effort value (low/medium/high/xhigh/max) into +the request body when thinking is on, instead of mapping it to a binary thinking +toggle and discarding it. See https://api-docs.deepseek.com/guides/thinking_mode +""" +from litellm.llms.deepseek.chat.transformation import DeepSeekChatConfig + + +def _map(non_default_params, drop_params=True): + cfg = DeepSeekChatConfig() + return cfg.map_openai_params(non_default_params, {}, "deepseek/deepseek-v4-flash", drop_params) + + +def test_reasoning_effort_low_passes_through(): + out = _map({"reasoning_effort": "low"}) + assert out["reasoning_effort"] == "low" + assert out["thinking"] == {"type": "enabled"} + + +def test_reasoning_effort_max_passes_through(): + out = _map({"reasoning_effort": "max"}) + assert out["reasoning_effort"] == "max" + assert out["thinking"] == {"type": "enabled"} + + +def test_reasoning_effort_none_disables_thinking(): + out = _map({"reasoning_effort": "none"}) + assert out["thinking"] == {"type": "disabled"} + assert "reasoning_effort" not in out + + +def test_explicit_thinking_disabled_wins_and_drops_effort(): + out = _map({"thinking": {"type": "disabled"}, "reasoning_effort": "low"}) + assert out["thinking"] == {"type": "disabled"} + assert "reasoning_effort" not in out + + +def test_no_reasoning_effort_leaves_no_key(): + out = _map({}) + assert out.get("reasoning_effort") is None From b84eff2c1fe9908b9cf86b2320dd089b760210c5 Mon Sep 17 00:00:00 2001 From: Hazelhof <110658790+Hazelhof@users.noreply.github.com> Date: Sat, 29 Aug 2026 17:23:30 +1000 Subject: [PATCH 3/3] fix(deepseek): exclude reasoning_effort=none sentinel from passthrough "none" is the thinking-OFF sentinel, not a real effort value. Passing it alongside explicit thinking:enabled would send contradictory instructions to DeepSeek. Only forward real effort values (low/medium/high/xhigh/max). Addresses review feedback on the reasoning_effort passthrough change. --- litellm/llms/deepseek/chat/transformation.py | 6 ++++-- tests/llm_translation/test_deepseek_reasoning_effort.py | 8 ++++++++ 2 files changed, 12 insertions(+), 2 deletions(-) diff --git a/litellm/llms/deepseek/chat/transformation.py b/litellm/llms/deepseek/chat/transformation.py index 80775889118..4c3eb55cb54 100644 --- a/litellm/llms/deepseek/chat/transformation.py +++ b/litellm/llms/deepseek/chat/transformation.py @@ -70,8 +70,10 @@ class DeepSeekChatConfig(OpenAIGPTConfig): # can run away, exhausting max_tokens during reasoning and returning empty content # (finish_reason="length", content:null). Preserve the value into the request body # when thinking is on so callers can bound effort (e.g. reasoning_effort="low"). - # See https://api-docs.deepseek.com/guides/thinking_mode - if thinking_enabled and reasoning_effort is not None: + # Exclude the "none" sentinel: it is an OpenAI-style thinking-OFF switch, not a real + # effort value, so passing it alongside explicit thinking:enabled would be + # contradictory. See https://api-docs.deepseek.com/guides/thinking_mode + if thinking_enabled and reasoning_effort not in (None, "none"): optional_params["reasoning_effort"] = reasoning_effort return optional_params diff --git a/tests/llm_translation/test_deepseek_reasoning_effort.py b/tests/llm_translation/test_deepseek_reasoning_effort.py index f9d3f29d2ab..6e1bb8f8281 100644 --- a/tests/llm_translation/test_deepseek_reasoning_effort.py +++ b/tests/llm_translation/test_deepseek_reasoning_effort.py @@ -39,3 +39,11 @@ def test_explicit_thinking_disabled_wins_and_drops_effort(): def test_no_reasoning_effort_leaves_no_key(): out = _map({}) assert out.get("reasoning_effort") is None + + +def test_explicit_enabled_thinking_with_none_effort_not_contradictory(): + # "none" is a thinking-OFF sentinel, not a real effort value: it must not be + # passed alongside explicit thinking:enabled (would contradict the provider). + out = _map({"thinking": {"type": "enabled"}, "reasoning_effort": "none"}) + assert out["thinking"] == {"type": "enabled"} + assert "reasoning_effort" not in out