fix: Qwen3.8 2.4T now supports response schema on runinfra

JSON mode went live on the host on 2026-08-17, verified on the public endpoint: 3 of 3 json_object responses parsed, 3 of 3 strict json_schema responses validated, warm constrained requests at 0.35x baseline latency. The test that pinned the deliberate absence now pins the verified presence.
This commit is contained in:
jaberjaber23 2026-08-17 12:29:50 +03:00
parent 4509ebabb8
commit 323c90e91f
3 changed files with 3 additions and 1 deletions

View file

@ -48017,6 +48017,7 @@
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
},
"runinfra/Qwen/Qwen3.8-27B": {

View file

@ -48017,6 +48017,7 @@
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
},
"runinfra/Qwen/Qwen3.8-27B": {

View file

@ -445,7 +445,7 @@ class TestRuninfra:
assert model_cost["runinfra/deepseek-ai/DeepSeek-V4-Flash-0731"]["max_input_tokens"] == 1048576
assert model_cost["runinfra/Qwen/Qwen3.8-27B"]["max_input_tokens"] == 262144
assert "supports_response_schema" not in model_cost["runinfra/Inferact/Qwen3.8-2.4T-A95B-NVFP4"]
assert model_cost["runinfra/Inferact/Qwen3.8-2.4T-A95B-NVFP4"]["supports_response_schema"] is True
assert model_cost["runinfra/Qwen/Qwen3.8-27B"]["supports_response_schema"] is True