mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix: Qwen3.8 2.4T now supports response schema on runinfra
JSON mode went live on the host on 2026-08-17, verified on the public endpoint: 3 of 3 json_object responses parsed, 3 of 3 strict json_schema responses validated, warm constrained requests at 0.35x baseline latency. The test that pinned the deliberate absence now pins the verified presence.
This commit is contained in:
parent
4509ebabb8
commit
323c90e91f
3 changed files with 3 additions and 1 deletions
|
|
@ -48017,6 +48017,7 @@
|
|||
"supports_native_streaming": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"runinfra/Qwen/Qwen3.8-27B": {
|
||||
|
|
|
|||
|
|
@ -48017,6 +48017,7 @@
|
|||
"supports_native_streaming": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"runinfra/Qwen/Qwen3.8-27B": {
|
||||
|
|
|
|||
|
|
@ -445,7 +445,7 @@ class TestRuninfra:
|
|||
|
||||
assert model_cost["runinfra/deepseek-ai/DeepSeek-V4-Flash-0731"]["max_input_tokens"] == 1048576
|
||||
assert model_cost["runinfra/Qwen/Qwen3.8-27B"]["max_input_tokens"] == 262144
|
||||
assert "supports_response_schema" not in model_cost["runinfra/Inferact/Qwen3.8-2.4T-A95B-NVFP4"]
|
||||
assert model_cost["runinfra/Inferact/Qwen3.8-2.4T-A95B-NVFP4"]["supports_response_schema"] is True
|
||||
assert model_cost["runinfra/Qwen/Qwen3.8-27B"]["supports_response_schema"] is True
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue