mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(tencent): route thinking through extra_body in chat completions
Tencent chat completions route through the OpenAI SDK's
chat.completions.create(), which raises TypeError on unknown kwargs -
so a top-level 'thinking' optional param crashed every reasoning
request with a 500 before any HTTP call was made.
Nest the resolved thinking object in extra_body instead: the SDK merges
extra_body into the top-level JSON payload, so TokenHub still receives
the documented thinking field (type/budget_tokens) in the request body.
Also align the param mapping with TokenHub's documented behavior:
- reasoning_effort="none" now maps to thinking={"type": "disabled"}
instead of being dropped (deepseek-v4-* default to thinking enabled,
so dropping it never actually disabled thinking)
- MiniMax models only accept thinking.type "adaptive"/"disabled",
so "enabled" is coerced to "adaptive" instead of returning a 400
Refs: https://www.tencentcloud.com/document/product/1300/82345
This commit is contained in:
parent
947dbbf029
commit
6a0e7fe10f
3 changed files with 171 additions and 13 deletions
|
|
@ -30,14 +30,38 @@ class TencentChatConfig(OpenAIGPTConfig):
|
|||
thinking_value: Final = optional_params.pop("thinking", None)
|
||||
reasoning_effort: Final = optional_params.pop("reasoning_effort", None)
|
||||
|
||||
if thinking_value is not None:
|
||||
if isinstance(thinking_value, dict):
|
||||
optional_params["thinking"] = thinking_value
|
||||
elif reasoning_effort is not None and reasoning_effort != "none":
|
||||
optional_params["thinking"] = {"type": "enabled"}
|
||||
thinking: dict | None = None
|
||||
if isinstance(thinking_value, dict):
|
||||
thinking = thinking_value
|
||||
elif reasoning_effort is not None:
|
||||
# TokenHub recommends explicitly disabling thinking instead of
|
||||
# relying on per-model defaults (deepseek-v4-* default to enabled).
|
||||
thinking = {"type": "disabled" if reasoning_effort == "none" else "enabled"}
|
||||
|
||||
if thinking is not None:
|
||||
thinking = self._normalize_thinking_type_for_model(model=model, thinking=thinking)
|
||||
# Tencent TokenHub expects `thinking` in the request JSON body, but
|
||||
# the OpenAI SDK's chat.completions.create() rejects unknown
|
||||
# top-level kwargs. Route it through `extra_body` so it is merged
|
||||
# into the payload instead of passed as a keyword argument.
|
||||
extra_body: Final = optional_params.setdefault("extra_body", {})
|
||||
extra_body["thinking"] = thinking
|
||||
|
||||
return optional_params
|
||||
|
||||
@staticmethod
|
||||
def _normalize_thinking_type_for_model(model: str, thinking: dict) -> dict:
|
||||
"""Coerce `thinking.type` values the model does not accept.
|
||||
|
||||
MiniMax models on TokenHub only accept "adaptive" or "disabled" —
|
||||
sending "enabled" returns a 400. "adaptive" is the closest semantic
|
||||
(the model decides when to think), so "enabled" is coerced to it.
|
||||
Ref: https://www.tencentcloud.com/document/product/1300/82345
|
||||
"""
|
||||
if thinking.get("type") == "enabled" and "minimax" in model.lower():
|
||||
return {**thinking, "type": "adaptive"}
|
||||
return thinking
|
||||
|
||||
def _get_openai_compatible_provider_info(
|
||||
self, api_base: str | None, api_key: str | None
|
||||
) -> tuple[str | None, str | None]:
|
||||
|
|
|
|||
|
|
@ -45,7 +45,8 @@ def test_map_openai_params_passes_thinking_dict_through():
|
|||
drop_params=False,
|
||||
)
|
||||
|
||||
assert result["thinking"] == {"type": "enabled", "budget_tokens": 1024}
|
||||
assert "thinking" not in result
|
||||
assert result["extra_body"]["thinking"] == {"type": "enabled", "budget_tokens": 1024}
|
||||
|
||||
|
||||
def test_map_openai_params_converts_reasoning_effort_to_thinking():
|
||||
|
|
@ -61,10 +62,11 @@ def test_map_openai_params_converts_reasoning_effort_to_thinking():
|
|||
drop_params=False,
|
||||
)
|
||||
|
||||
assert result["thinking"] == {"type": "enabled"}
|
||||
assert "thinking" not in result
|
||||
assert result["extra_body"]["thinking"] == {"type": "enabled"}
|
||||
|
||||
|
||||
def test_map_openai_params_drops_none_reasoning_effort():
|
||||
def test_map_openai_params_none_reasoning_effort_disables_thinking():
|
||||
config = TencentChatConfig()
|
||||
with patch(
|
||||
"litellm.llms.tencent.chat.transformation.supports_reasoning",
|
||||
|
|
@ -78,6 +80,7 @@ def test_map_openai_params_drops_none_reasoning_effort():
|
|||
)
|
||||
|
||||
assert "thinking" not in result
|
||||
assert result["extra_body"]["thinking"] == {"type": "disabled"}
|
||||
assert "reasoning_effort" not in result
|
||||
|
||||
|
||||
|
|
@ -97,7 +100,8 @@ def test_map_openai_params_thinking_priority_over_reasoning_effort():
|
|||
drop_params=False,
|
||||
)
|
||||
|
||||
assert result["thinking"] == {"type": "enabled", "budget_tokens": 2048}
|
||||
assert "thinking" not in result
|
||||
assert result["extra_body"]["thinking"] == {"type": "enabled", "budget_tokens": 2048}
|
||||
|
||||
|
||||
def test_map_openai_params_extracts_thinking_and_effort_from_optional_params():
|
||||
|
|
@ -109,10 +113,134 @@ def test_map_openai_params_extracts_thinking_and_effort_from_optional_params():
|
|||
drop_params=False,
|
||||
)
|
||||
|
||||
assert "thinking" in result
|
||||
assert "thinking" not in result
|
||||
assert result["extra_body"]["thinking"] == {"type": "enabled"}
|
||||
assert "reasoning_effort" not in result
|
||||
|
||||
|
||||
def test_map_openai_params_merges_into_existing_extra_body():
|
||||
config = TencentChatConfig()
|
||||
result = config.map_openai_params(
|
||||
non_default_params={},
|
||||
optional_params={
|
||||
"thinking": {"type": "enabled"},
|
||||
"extra_body": {"custom_flag": True},
|
||||
},
|
||||
model="tencent/deepseek-v4-pro",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
assert result["extra_body"] == {"custom_flag": True, "thinking": {"type": "enabled"}}
|
||||
|
||||
|
||||
def test_transform_request_never_passes_thinking_as_top_level_kwarg():
|
||||
"""
|
||||
Regression test: tencent routes through the OpenAI SDK's
|
||||
chat.completions.create(**data), which raises TypeError on unknown kwargs.
|
||||
`thinking` must be nested inside extra_body, never top-level.
|
||||
"""
|
||||
config = TencentChatConfig()
|
||||
optional_params = config.map_openai_params(
|
||||
non_default_params={"thinking": {"type": "enabled", "budget_tokens": 1024}},
|
||||
optional_params={},
|
||||
model="tencent/deepseek-v4-pro",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
data = config.transform_request(
|
||||
model="deepseek-v4-pro",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
optional_params=optional_params,
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert "thinking" not in data
|
||||
assert data["extra_body"]["thinking"] == {"type": "enabled", "budget_tokens": 1024}
|
||||
|
||||
|
||||
class TestMinimaxThinkingCoercion:
|
||||
"""
|
||||
MiniMax models on TokenHub only accept thinking.type "adaptive"/"disabled" —
|
||||
"enabled" returns a 400. Ref: https://www.tencentcloud.com/document/product/1300/82345
|
||||
"""
|
||||
|
||||
def test_reasoning_effort_maps_to_adaptive_for_minimax(self):
|
||||
config = TencentChatConfig()
|
||||
with patch(
|
||||
"litellm.llms.tencent.chat.transformation.supports_reasoning",
|
||||
return_value=True,
|
||||
):
|
||||
result = config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": "medium"},
|
||||
optional_params={},
|
||||
model="tencent/minimax-m3",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
assert result["extra_body"]["thinking"] == {"type": "adaptive"}
|
||||
|
||||
def test_explicit_enabled_thinking_coerced_to_adaptive_for_minimax(self):
|
||||
config = TencentChatConfig()
|
||||
with patch(
|
||||
"litellm.llms.tencent.chat.transformation.supports_reasoning",
|
||||
return_value=True,
|
||||
):
|
||||
result = config.map_openai_params(
|
||||
non_default_params={"thinking": {"type": "enabled", "budget_tokens": 4096}},
|
||||
optional_params={},
|
||||
model="minimax-m3",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
assert result["extra_body"]["thinking"] == {"type": "adaptive", "budget_tokens": 4096}
|
||||
|
||||
def test_disabled_thinking_kept_for_minimax(self):
|
||||
config = TencentChatConfig()
|
||||
with patch(
|
||||
"litellm.llms.tencent.chat.transformation.supports_reasoning",
|
||||
return_value=True,
|
||||
):
|
||||
result = config.map_openai_params(
|
||||
non_default_params={"thinking": {"type": "disabled"}},
|
||||
optional_params={},
|
||||
model="tencent/minimax-m3",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
assert result["extra_body"]["thinking"] == {"type": "disabled"}
|
||||
|
||||
def test_none_reasoning_effort_disables_thinking_for_minimax(self):
|
||||
config = TencentChatConfig()
|
||||
with patch(
|
||||
"litellm.llms.tencent.chat.transformation.supports_reasoning",
|
||||
return_value=True,
|
||||
):
|
||||
result = config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": "none"},
|
||||
optional_params={},
|
||||
model="tencent/minimax-m3",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
assert result["extra_body"]["thinking"] == {"type": "disabled"}
|
||||
|
||||
def test_non_minimax_model_keeps_enabled(self):
|
||||
config = TencentChatConfig()
|
||||
with patch(
|
||||
"litellm.llms.tencent.chat.transformation.supports_reasoning",
|
||||
return_value=True,
|
||||
):
|
||||
result = config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": "high"},
|
||||
optional_params={},
|
||||
model="tencent/kimi-k3",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
assert result["extra_body"]["thinking"] == {"type": "enabled"}
|
||||
|
||||
|
||||
def test_get_complete_url_default():
|
||||
config = TencentChatConfig()
|
||||
|
||||
|
|
|
|||
|
|
@ -4244,7 +4244,11 @@ class TestGetOptionalParamsTencent:
|
|||
"""Tests that tencent provider uses TencentChatConfig for parameter mapping."""
|
||||
|
||||
def test_tencent_supports_thinking_param(self):
|
||||
"""Verify get_optional_params for tencent accepts the 'thinking' param."""
|
||||
"""Verify get_optional_params for tencent accepts the 'thinking' param.
|
||||
|
||||
`thinking` must be nested in extra_body: tencent routes through the
|
||||
OpenAI SDK's chat.completions.create(), which rejects unknown kwargs.
|
||||
"""
|
||||
from unittest.mock import patch
|
||||
|
||||
from litellm.utils import get_optional_params
|
||||
|
|
@ -4258,7 +4262,8 @@ class TestGetOptionalParamsTencent:
|
|||
custom_llm_provider="tencent",
|
||||
thinking={"type": "enabled"},
|
||||
)
|
||||
assert result.get("thinking") == {"type": "enabled"}
|
||||
assert "thinking" not in result
|
||||
assert result["extra_body"]["thinking"] == {"type": "enabled"}
|
||||
|
||||
def test_tencent_supports_reasoning_effort(self):
|
||||
"""Verify get_optional_params for tencent converts reasoning_effort to thinking."""
|
||||
|
|
@ -4275,7 +4280,8 @@ class TestGetOptionalParamsTencent:
|
|||
custom_llm_provider="tencent",
|
||||
reasoning_effort="medium",
|
||||
)
|
||||
assert result.get("thinking") == {"type": "enabled"}
|
||||
assert "thinking" not in result
|
||||
assert result["extra_body"]["thinking"] == {"type": "enabled"}
|
||||
|
||||
def test_tencent_supported_params_includes_thinking_and_reasoning_effort(self):
|
||||
"""Verify get_supported_openai_params for tencent includes custom params."""
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue