fix(tencent): route thinking through extra_body in chat completions

Tencent chat completions route through the OpenAI SDK's
chat.completions.create(), which raises TypeError on unknown kwargs -
so a top-level 'thinking' optional param crashed every reasoning
request with a 500 before any HTTP call was made.

Nest the resolved thinking object in extra_body instead: the SDK merges
extra_body into the top-level JSON payload, so TokenHub still receives
the documented thinking field (type/budget_tokens) in the request body.

Also align the param mapping with TokenHub's documented behavior:
- reasoning_effort="none" now maps to thinking={"type": "disabled"}
  instead of being dropped (deepseek-v4-* default to thinking enabled,
  so dropping it never actually disabled thinking)
- MiniMax models only accept thinking.type "adaptive"/"disabled",
  so "enabled" is coerced to "adaptive" instead of returning a 400

Refs: https://www.tencentcloud.com/document/product/1300/82345
This commit is contained in:
Felipe Rodrigues Gare Carnielli 2026-08-24 14:58:13 -03:00
parent 947dbbf029
commit 6a0e7fe10f
3 changed files with 171 additions and 13 deletions

View file

@ -30,14 +30,38 @@ class TencentChatConfig(OpenAIGPTConfig):
thinking_value: Final = optional_params.pop("thinking", None)
reasoning_effort: Final = optional_params.pop("reasoning_effort", None)
if thinking_value is not None:
if isinstance(thinking_value, dict):
optional_params["thinking"] = thinking_value
elif reasoning_effort is not None and reasoning_effort != "none":
optional_params["thinking"] = {"type": "enabled"}
thinking: dict | None = None
if isinstance(thinking_value, dict):
thinking = thinking_value
elif reasoning_effort is not None:
# TokenHub recommends explicitly disabling thinking instead of
# relying on per-model defaults (deepseek-v4-* default to enabled).
thinking = {"type": "disabled" if reasoning_effort == "none" else "enabled"}
if thinking is not None:
thinking = self._normalize_thinking_type_for_model(model=model, thinking=thinking)
# Tencent TokenHub expects `thinking` in the request JSON body, but
# the OpenAI SDK's chat.completions.create() rejects unknown
# top-level kwargs. Route it through `extra_body` so it is merged
# into the payload instead of passed as a keyword argument.
extra_body: Final = optional_params.setdefault("extra_body", {})
extra_body["thinking"] = thinking
return optional_params
@staticmethod
def _normalize_thinking_type_for_model(model: str, thinking: dict) -> dict:
"""Coerce `thinking.type` values the model does not accept.
MiniMax models on TokenHub only accept "adaptive" or "disabled" —
sending "enabled" returns a 400. "adaptive" is the closest semantic
(the model decides when to think), so "enabled" is coerced to it.
Ref: https://www.tencentcloud.com/document/product/1300/82345
"""
if thinking.get("type") == "enabled" and "minimax" in model.lower():
return {**thinking, "type": "adaptive"}
return thinking
def _get_openai_compatible_provider_info(
self, api_base: str | None, api_key: str | None
) -> tuple[str | None, str | None]:

View file

@ -45,7 +45,8 @@ def test_map_openai_params_passes_thinking_dict_through():
drop_params=False,
)
assert result["thinking"] == {"type": "enabled", "budget_tokens": 1024}
assert "thinking" not in result
assert result["extra_body"]["thinking"] == {"type": "enabled", "budget_tokens": 1024}
def test_map_openai_params_converts_reasoning_effort_to_thinking():
@ -61,10 +62,11 @@ def test_map_openai_params_converts_reasoning_effort_to_thinking():
drop_params=False,
)
assert result["thinking"] == {"type": "enabled"}
assert "thinking" not in result
assert result["extra_body"]["thinking"] == {"type": "enabled"}
def test_map_openai_params_drops_none_reasoning_effort():
def test_map_openai_params_none_reasoning_effort_disables_thinking():
config = TencentChatConfig()
with patch(
"litellm.llms.tencent.chat.transformation.supports_reasoning",
@ -78,6 +80,7 @@ def test_map_openai_params_drops_none_reasoning_effort():
)
assert "thinking" not in result
assert result["extra_body"]["thinking"] == {"type": "disabled"}
assert "reasoning_effort" not in result
@ -97,7 +100,8 @@ def test_map_openai_params_thinking_priority_over_reasoning_effort():
drop_params=False,
)
assert result["thinking"] == {"type": "enabled", "budget_tokens": 2048}
assert "thinking" not in result
assert result["extra_body"]["thinking"] == {"type": "enabled", "budget_tokens": 2048}
def test_map_openai_params_extracts_thinking_and_effort_from_optional_params():
@ -109,10 +113,134 @@ def test_map_openai_params_extracts_thinking_and_effort_from_optional_params():
drop_params=False,
)
assert "thinking" in result
assert "thinking" not in result
assert result["extra_body"]["thinking"] == {"type": "enabled"}
assert "reasoning_effort" not in result
def test_map_openai_params_merges_into_existing_extra_body():
config = TencentChatConfig()
result = config.map_openai_params(
non_default_params={},
optional_params={
"thinking": {"type": "enabled"},
"extra_body": {"custom_flag": True},
},
model="tencent/deepseek-v4-pro",
drop_params=False,
)
assert result["extra_body"] == {"custom_flag": True, "thinking": {"type": "enabled"}}
def test_transform_request_never_passes_thinking_as_top_level_kwarg():
"""
Regression test: tencent routes through the OpenAI SDK's
chat.completions.create(**data), which raises TypeError on unknown kwargs.
`thinking` must be nested inside extra_body, never top-level.
"""
config = TencentChatConfig()
optional_params = config.map_openai_params(
non_default_params={"thinking": {"type": "enabled", "budget_tokens": 1024}},
optional_params={},
model="tencent/deepseek-v4-pro",
drop_params=False,
)
data = config.transform_request(
model="deepseek-v4-pro",
messages=[{"role": "user", "content": "hi"}],
optional_params=optional_params,
litellm_params={},
headers={},
)
assert "thinking" not in data
assert data["extra_body"]["thinking"] == {"type": "enabled", "budget_tokens": 1024}
class TestMinimaxThinkingCoercion:
"""
MiniMax models on TokenHub only accept thinking.type "adaptive"/"disabled" —
"enabled" returns a 400. Ref: https://www.tencentcloud.com/document/product/1300/82345
"""
def test_reasoning_effort_maps_to_adaptive_for_minimax(self):
config = TencentChatConfig()
with patch(
"litellm.llms.tencent.chat.transformation.supports_reasoning",
return_value=True,
):
result = config.map_openai_params(
non_default_params={"reasoning_effort": "medium"},
optional_params={},
model="tencent/minimax-m3",
drop_params=False,
)
assert result["extra_body"]["thinking"] == {"type": "adaptive"}
def test_explicit_enabled_thinking_coerced_to_adaptive_for_minimax(self):
config = TencentChatConfig()
with patch(
"litellm.llms.tencent.chat.transformation.supports_reasoning",
return_value=True,
):
result = config.map_openai_params(
non_default_params={"thinking": {"type": "enabled", "budget_tokens": 4096}},
optional_params={},
model="minimax-m3",
drop_params=False,
)
assert result["extra_body"]["thinking"] == {"type": "adaptive", "budget_tokens": 4096}
def test_disabled_thinking_kept_for_minimax(self):
config = TencentChatConfig()
with patch(
"litellm.llms.tencent.chat.transformation.supports_reasoning",
return_value=True,
):
result = config.map_openai_params(
non_default_params={"thinking": {"type": "disabled"}},
optional_params={},
model="tencent/minimax-m3",
drop_params=False,
)
assert result["extra_body"]["thinking"] == {"type": "disabled"}
def test_none_reasoning_effort_disables_thinking_for_minimax(self):
config = TencentChatConfig()
with patch(
"litellm.llms.tencent.chat.transformation.supports_reasoning",
return_value=True,
):
result = config.map_openai_params(
non_default_params={"reasoning_effort": "none"},
optional_params={},
model="tencent/minimax-m3",
drop_params=False,
)
assert result["extra_body"]["thinking"] == {"type": "disabled"}
def test_non_minimax_model_keeps_enabled(self):
config = TencentChatConfig()
with patch(
"litellm.llms.tencent.chat.transformation.supports_reasoning",
return_value=True,
):
result = config.map_openai_params(
non_default_params={"reasoning_effort": "high"},
optional_params={},
model="tencent/kimi-k3",
drop_params=False,
)
assert result["extra_body"]["thinking"] == {"type": "enabled"}
def test_get_complete_url_default():
config = TencentChatConfig()

View file

@ -4244,7 +4244,11 @@ class TestGetOptionalParamsTencent:
"""Tests that tencent provider uses TencentChatConfig for parameter mapping."""
def test_tencent_supports_thinking_param(self):
"""Verify get_optional_params for tencent accepts the 'thinking' param."""
"""Verify get_optional_params for tencent accepts the 'thinking' param.
`thinking` must be nested in extra_body: tencent routes through the
OpenAI SDK's chat.completions.create(), which rejects unknown kwargs.
"""
from unittest.mock import patch
from litellm.utils import get_optional_params
@ -4258,7 +4262,8 @@ class TestGetOptionalParamsTencent:
custom_llm_provider="tencent",
thinking={"type": "enabled"},
)
assert result.get("thinking") == {"type": "enabled"}
assert "thinking" not in result
assert result["extra_body"]["thinking"] == {"type": "enabled"}
def test_tencent_supports_reasoning_effort(self):
"""Verify get_optional_params for tencent converts reasoning_effort to thinking."""
@ -4275,7 +4280,8 @@ class TestGetOptionalParamsTencent:
custom_llm_provider="tencent",
reasoning_effort="medium",
)
assert result.get("thinking") == {"type": "enabled"}
assert "thinking" not in result
assert result["extra_body"]["thinking"] == {"type": "enabled"}
def test_tencent_supported_params_includes_thinking_and_reasoning_effort(self):
"""Verify get_supported_openai_params for tencent includes custom params."""