From 323c90e91f0fea0d14e4aba4c8c4996894ca5a7e Mon Sep 17 00:00:00 2001 From: jaberjaber23 Date: Mon, 17 Aug 2026 12:29:50 +0300 Subject: [PATCH] fix: Qwen3.8 2.4T now supports response schema on runinfra JSON mode went live on the host on 2026-08-17, verified on the public endpoint: 3 of 3 json_object responses parsed, 3 of 3 strict json_schema responses validated, warm constrained requests at 0.35x baseline latency. The test that pinned the deliberate absence now pins the verified presence. --- litellm/model_prices_and_context_window_backup.json | 1 + model_prices_and_context_window.json | 1 + tests/test_litellm/llms/openai_like/test_json_providers.py | 2 +- 3 files changed, 3 insertions(+), 1 deletion(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 04121620479..db9e3e4c711 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -48017,6 +48017,7 @@ "supports_native_streaming": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_response_schema": true, "supports_tool_choice": true }, "runinfra/Qwen/Qwen3.8-27B": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 04121620479..db9e3e4c711 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -48017,6 +48017,7 @@ "supports_native_streaming": true, "supports_prompt_caching": true, "supports_reasoning": true, + "supports_response_schema": true, "supports_tool_choice": true }, "runinfra/Qwen/Qwen3.8-27B": { diff --git a/tests/test_litellm/llms/openai_like/test_json_providers.py b/tests/test_litellm/llms/openai_like/test_json_providers.py index cc8c570640b..7c9b2f1a98b 100644 --- a/tests/test_litellm/llms/openai_like/test_json_providers.py +++ b/tests/test_litellm/llms/openai_like/test_json_providers.py @@ -445,7 +445,7 @@ class TestRuninfra: assert model_cost["runinfra/deepseek-ai/DeepSeek-V4-Flash-0731"]["max_input_tokens"] == 1048576 assert model_cost["runinfra/Qwen/Qwen3.8-27B"]["max_input_tokens"] == 262144 - assert "supports_response_schema" not in model_cost["runinfra/Inferact/Qwen3.8-2.4T-A95B-NVFP4"] + assert model_cost["runinfra/Inferact/Qwen3.8-2.4T-A95B-NVFP4"]["supports_response_schema"] is True assert model_cost["runinfra/Qwen/Qwen3.8-27B"]["supports_response_schema"] is True