mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(zai): wrap thinking and reasoning_effort into extra_body
ZAIChatConfig inherited OpenAIGPTConfig._map_openai_params, which copies
whitelisted params straight to optional_params at the top level. The
OpenAI Python SDK then rejects them as unknown kwargs, e.g.
`AsyncCompletions.create() got an unexpected keyword argument
'thinking'`. ZAI request-body fields like `thinking` and
`reasoning_effort` must travel inside `extra_body` so the SDK flattens
them into the JSON body.
On a single hop the bug hid behind the extra_body salvage in
add_provider_specific_params_to_optional_params, which preserved an
extra_body the client sent. On a chained-proxy topology
(litellm proxy -> litellm proxy -> ZAI) hop 1's SDK had already
flattened the client's extra_body into a top-level `thinking` kwarg by
the time hop 2 saw the request; hop 2's get_optional_params re-emitted
that as a top-level kwarg and the SDK crashed.
The fix overrides ZAIChatConfig._map_openai_params to route both
`thinking` and `reasoning_effort` into optional_params["extra_body"],
matching the existing pattern in LiteLLMProxyChatConfig
(litellm/llms/litellm_proxy/chat/transformation.py) and VolcEngineConfig
(litellm/llms/volcengine/chat/transformation.py).
Also dropped the `if supports_reasoning(model): base_params.append("thinking")`
gate; per docs.z.ai, `thinking` is supported by every GLM-4.5+ model
unconditionally, and the registry's `supports_reasoning` flag was
incorrectly false for the entire GLM-4.5 family. The patch corrects the
registry too: glm-4.5, glm-4.5v, glm-4.5-x, glm-4.5-air, glm-4.5-airx,
and glm-4.5-flash now carry `supports_reasoning: true`.
Tests in tests/test_litellm/llms/zai/test_zai_provider.py cover the
whitelist contract, the _map_openai_params wrapping, and end-to-end
boundary tests that capture the outgoing HTTP body via respx and assert
the JSON shape the upstream actually receives.
This commit is contained in:
parent
69b0dd2da0
commit
0af6ba42e2
3 changed files with 261 additions and 17 deletions
|
|
@ -7,6 +7,8 @@ from ...openai.chat.gpt_transformation import OpenAIGPTConfig
|
|||
|
||||
ZAI_API_BASE = "https://api.z.ai/api/paas/v4"
|
||||
|
||||
_REASONING_PARAMS = ("thinking", "reasoning_effort")
|
||||
|
||||
|
||||
class ZAIChatConfig(OpenAIGPTConfig):
|
||||
@property
|
||||
|
|
@ -26,15 +28,10 @@ class ZAIChatConfig(OpenAIGPTConfig):
|
|||
messages: List[AllMessageValues],
|
||||
tools: Optional[List[ChatCompletionToolParam]] = None,
|
||||
) -> Tuple[List[AllMessageValues], Optional[List[ChatCompletionToolParam]]]:
|
||||
"""
|
||||
Override to preserve cache_control for GLM/ZAI.
|
||||
GLM supports cache_control - don't strip it.
|
||||
"""
|
||||
# GLM/ZAI supports cache_control, so return messages and tools unchanged
|
||||
return messages, tools
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> list:
|
||||
base_params = [
|
||||
return [
|
||||
"max_tokens",
|
||||
"stream",
|
||||
"stream_options",
|
||||
|
|
@ -43,16 +40,21 @@ class ZAIChatConfig(OpenAIGPTConfig):
|
|||
"stop",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"thinking",
|
||||
"reasoning_effort",
|
||||
]
|
||||
|
||||
import litellm
|
||||
|
||||
try:
|
||||
if litellm.supports_reasoning(
|
||||
model=model, custom_llm_provider=self.custom_llm_provider
|
||||
):
|
||||
base_params.append("thinking")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return base_params
|
||||
def _map_openai_params(
|
||||
self,
|
||||
non_default_params: dict,
|
||||
optional_params: dict,
|
||||
model: str,
|
||||
drop_params: bool,
|
||||
) -> dict:
|
||||
supported = self.get_supported_openai_params(model)
|
||||
for param, value in non_default_params.items():
|
||||
if param in _REASONING_PARAMS:
|
||||
optional_params.setdefault("extra_body", {})[param] = value
|
||||
elif param in supported:
|
||||
optional_params[param] = value
|
||||
return optional_params
|
||||
|
|
|
|||
|
|
@ -37618,6 +37618,7 @@
|
|||
"max_output_tokens": 32000,
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://docs.z.ai/guides/overview/pricing"
|
||||
},
|
||||
|
|
@ -37629,6 +37630,7 @@
|
|||
"max_output_tokens": 32000,
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"source": "https://docs.z.ai/guides/overview/pricing"
|
||||
|
|
@ -37641,6 +37643,7 @@
|
|||
"max_output_tokens": 32000,
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://docs.z.ai/guides/overview/pricing"
|
||||
},
|
||||
|
|
@ -37652,6 +37655,7 @@
|
|||
"max_output_tokens": 32000,
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://docs.z.ai/guides/overview/pricing"
|
||||
},
|
||||
|
|
@ -37663,6 +37667,7 @@
|
|||
"max_output_tokens": 32000,
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://docs.z.ai/guides/overview/pricing"
|
||||
},
|
||||
|
|
@ -37685,6 +37690,7 @@
|
|||
"max_output_tokens": 32000,
|
||||
"mode": "chat",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_tool_choice": true,
|
||||
"source": "https://docs.z.ai/guides/overview/pricing"
|
||||
},
|
||||
|
|
|
|||
|
|
@ -186,3 +186,239 @@ def test_zai_sync_completion(respx_mock, zai_response, monkeypatch):
|
|||
|
||||
assert response.choices[0].message.content == "Hello! How can I help you today?"
|
||||
assert response.usage.total_tokens == 25
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def zai_thinking_response():
|
||||
return {
|
||||
"id": "chatcmpl-zai-thinking",
|
||||
"object": "chat.completion",
|
||||
"created": 1700000000,
|
||||
"model": "glm-4.6",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": "hi"},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
"usage": {"prompt_tokens": 5, "completion_tokens": 2, "total_tokens": 7},
|
||||
}
|
||||
|
||||
|
||||
def _captured_body(respx_mock):
|
||||
assert len(respx_mock.calls) == 1
|
||||
return json.loads(respx_mock.calls[0].request.content.decode("utf-8"))
|
||||
|
||||
|
||||
class TestZaiSupportedParamsWhitelistReasoning:
|
||||
"""`thinking` and `reasoning_effort` must be in the whitelist for every
|
||||
GLM-4.5+ model, independently of the registry's `supports_reasoning`
|
||||
flag. The prior gate `if litellm.supports_reasoning(model): base_params.append("thinking")`
|
||||
silently broke any model whose registry entry was incomplete; the
|
||||
entire GLM-4.5 family in `model_prices_and_context_window_backup.json`
|
||||
was missing the flag despite docs.z.ai listing GLM-4.5 as the first
|
||||
model with `thinking` support
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
"glm-4.5",
|
||||
"glm-4.5v",
|
||||
"glm-4.5-air",
|
||||
"glm-4.5-airx",
|
||||
"glm-4.5-x",
|
||||
"glm-4.5-flash",
|
||||
"glm-4.6",
|
||||
"glm-4.7",
|
||||
"glm-5",
|
||||
],
|
||||
)
|
||||
def test_reasoning_params_in_whitelist(self, model):
|
||||
from litellm.llms.zai.chat.transformation import ZAIChatConfig
|
||||
|
||||
params = ZAIChatConfig().get_supported_openai_params(model=model)
|
||||
assert "thinking" in params
|
||||
assert "reasoning_effort" in params
|
||||
|
||||
|
||||
class TestZaiReasoningParamsLandInExtraBody:
|
||||
"""The OpenAI Python SDK rejects unknown top-level kwargs (e.g.
|
||||
`AsyncCompletions.create() got an unexpected keyword argument 'thinking'`),
|
||||
so ZAI-specific reasoning fields must travel inside `extra_body` and
|
||||
let the SDK flatten them into the HTTP body. Without this wrapping a
|
||||
chained-proxy topology (LiteLLM A -> LiteLLM B -> ZAI) drops the
|
||||
field on hop 2: hop 1's SDK flattens `extra_body` into a top-level
|
||||
`thinking` kwarg, hop 2 re-emits that as a top-level kwarg, and the
|
||||
SDK rejects it
|
||||
"""
|
||||
|
||||
def test_thinking_wraps_into_extra_body(self):
|
||||
from litellm.llms.zai.chat.transformation import ZAIChatConfig
|
||||
|
||||
result = ZAIChatConfig()._map_openai_params(
|
||||
non_default_params={"thinking": {"type": "disabled"}},
|
||||
optional_params={},
|
||||
model="glm-4.6",
|
||||
drop_params=False,
|
||||
)
|
||||
assert "thinking" not in result
|
||||
assert result["extra_body"]["thinking"] == {"type": "disabled"}
|
||||
|
||||
def test_reasoning_effort_wraps_into_extra_body(self):
|
||||
from litellm.llms.zai.chat.transformation import ZAIChatConfig
|
||||
|
||||
result = ZAIChatConfig()._map_openai_params(
|
||||
non_default_params={"reasoning_effort": "none"},
|
||||
optional_params={},
|
||||
model="glm-5",
|
||||
drop_params=False,
|
||||
)
|
||||
assert "reasoning_effort" not in result
|
||||
assert result["extra_body"]["reasoning_effort"] == "none"
|
||||
|
||||
def test_thinking_merges_into_existing_extra_body(self):
|
||||
from litellm.llms.zai.chat.transformation import ZAIChatConfig
|
||||
|
||||
result = ZAIChatConfig()._map_openai_params(
|
||||
non_default_params={"thinking": {"type": "disabled"}},
|
||||
optional_params={"extra_body": {"already_here": True}},
|
||||
model="glm-4.6",
|
||||
drop_params=False,
|
||||
)
|
||||
assert result["extra_body"]["already_here"] is True
|
||||
assert result["extra_body"]["thinking"] == {"type": "disabled"}
|
||||
|
||||
def test_standard_params_stay_top_level_alongside_thinking(self):
|
||||
from litellm.llms.zai.chat.transformation import ZAIChatConfig
|
||||
|
||||
result = ZAIChatConfig()._map_openai_params(
|
||||
non_default_params={
|
||||
"max_tokens": 100,
|
||||
"temperature": 0.7,
|
||||
"thinking": {"type": "enabled"},
|
||||
},
|
||||
optional_params={},
|
||||
model="glm-4.6",
|
||||
drop_params=False,
|
||||
)
|
||||
assert result["max_tokens"] == 100
|
||||
assert result["temperature"] == 0.7
|
||||
assert result["extra_body"]["thinking"] == {"type": "enabled"}
|
||||
assert "thinking" not in result
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_top_level_thinking_kwarg_reaches_http_body(
|
||||
self, respx_mock, zai_thinking_response, monkeypatch
|
||||
):
|
||||
"""Regression for the hop-2 SDK crash. Pre-fix this raised
|
||||
`AsyncCompletions.create() got an unexpected keyword argument
|
||||
'thinking'`. Post-fix the boundary HTTP body carries `thinking`
|
||||
as a top-level field (the OpenAI SDK flattened `extra_body`)
|
||||
"""
|
||||
monkeypatch.setenv("ZAI_API_KEY", "test-key")
|
||||
litellm.disable_aiohttp_transport = True
|
||||
respx_mock.post("https://api.z.ai/api/paas/v4/chat/completions").respond(
|
||||
json=zai_thinking_response
|
||||
)
|
||||
|
||||
await litellm.acompletion(
|
||||
model="zai/glm-4.6",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
thinking={"type": "disabled"},
|
||||
)
|
||||
|
||||
body = _captured_body(respx_mock)
|
||||
assert body["thinking"] == {"type": "disabled"}
|
||||
assert "extra_body" not in body
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extra_body_thinking_reaches_http_body(
|
||||
self, respx_mock, zai_thinking_response, monkeypatch
|
||||
):
|
||||
monkeypatch.setenv("ZAI_API_KEY", "test-key")
|
||||
litellm.disable_aiohttp_transport = True
|
||||
respx_mock.post("https://api.z.ai/api/paas/v4/chat/completions").respond(
|
||||
json=zai_thinking_response
|
||||
)
|
||||
|
||||
await litellm.acompletion(
|
||||
model="zai/glm-4.6",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
extra_body={"thinking": {"type": "disabled"}},
|
||||
)
|
||||
|
||||
body = _captured_body(respx_mock)
|
||||
assert body["thinking"] == {"type": "disabled"}
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_top_level_reasoning_effort_reaches_http_body(
|
||||
self, respx_mock, zai_thinking_response, monkeypatch
|
||||
):
|
||||
monkeypatch.setenv("ZAI_API_KEY", "test-key")
|
||||
litellm.disable_aiohttp_transport = True
|
||||
respx_mock.post("https://api.z.ai/api/paas/v4/chat/completions").respond(
|
||||
json=zai_thinking_response
|
||||
)
|
||||
|
||||
await litellm.acompletion(
|
||||
model="zai/glm-5",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
reasoning_effort="none",
|
||||
)
|
||||
|
||||
body = _captured_body(respx_mock)
|
||||
assert body["reasoning_effort"] == "none"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_thinking_works_on_glm_4_5_without_registry_flag(
|
||||
self, respx_mock, zai_thinking_response, monkeypatch
|
||||
):
|
||||
"""Even when a registry entry is missing `supports_reasoning`,
|
||||
ZAIChatConfig must still allow `thinking`. The prior gate
|
||||
silently dropped the field for the entire GLM-4.5 family in
|
||||
the registry; this test pins the unconditional contract
|
||||
"""
|
||||
monkeypatch.setenv("ZAI_API_KEY", "test-key")
|
||||
litellm.disable_aiohttp_transport = True
|
||||
respx_mock.post("https://api.z.ai/api/paas/v4/chat/completions").respond(
|
||||
json=zai_thinking_response
|
||||
)
|
||||
|
||||
await litellm.acompletion(
|
||||
model="zai/glm-4.5",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
thinking={"type": "disabled"},
|
||||
)
|
||||
|
||||
body = _captured_body(respx_mock)
|
||||
assert body["thinking"] == {"type": "disabled"}
|
||||
|
||||
|
||||
class TestGlm45FamilyRegistrySupportsReasoning:
|
||||
"""docs.z.ai lists GLM-4.5 as the first model family that supports
|
||||
`thinking`. The registry had every GLM-4.5 entry marked
|
||||
`supports_reasoning: false`, which is the source-of-truth bug that
|
||||
let the SDK-kwarg crash hide for so long
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_key",
|
||||
[
|
||||
"zai/glm-4.5",
|
||||
"zai/glm-4.5v",
|
||||
"zai/glm-4.5-x",
|
||||
"zai/glm-4.5-air",
|
||||
"zai/glm-4.5-airx",
|
||||
"zai/glm-4.5-flash",
|
||||
],
|
||||
)
|
||||
def test_supports_reasoning_true(self, model_key):
|
||||
import os
|
||||
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
assert model_key in litellm.model_cost
|
||||
assert litellm.model_cost[model_key].get("supports_reasoning") is True
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue