feat: add DeepSeek V4 Pro to the runinfra provider

Fifth hosted model, USD per token from the published per-1M rates (0.60 input, 1.90 output, 0.03 cached input read), 1,048,576 context, 32,768 output cap. Verified live on the public endpoint on 2026-08-18 before this commit: json_object and strict json_schema 3 of 3 each, one tool call round-tripped, a prefix-cache hit on a repeated identical prompt, reasoning content present on the wire when enabled (reasoning is toggleable on this host), and the output cap refusing over-cap requests with 400.
This commit is contained in:
jaberjaber23 2026-08-18 19:00:12 +03:00
parent 323c90e91f
commit 79e2537cbb
3 changed files with 42 additions and 0 deletions

View file

@ -48060,6 +48060,26 @@
"supports_response_schema": true,
"supports_tool_choice": true
},
"runinfra/deepseek-ai/DeepSeek-V4-Pro-0813": {
"cache_read_input_token_cost": 3e-08,
"input_cost_per_token": 6e-07,
"litellm_provider": "runinfra",
"max_input_tokens": 1048576,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 1.9e-06,
"source": "https://runinfra.ai/inference-api/deepseek-v4-pro",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
},
"runinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16": {
"input_cost_per_token": 5e-08,
"litellm_provider": "runinfra",

View file

@ -48060,6 +48060,26 @@
"supports_response_schema": true,
"supports_tool_choice": true
},
"runinfra/deepseek-ai/DeepSeek-V4-Pro-0813": {
"cache_read_input_token_cost": 3e-08,
"input_cost_per_token": 6e-07,
"litellm_provider": "runinfra",
"max_input_tokens": 1048576,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"output_cost_per_token": 1.9e-06,
"source": "https://runinfra.ai/inference-api/deepseek-v4-pro",
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true
},
"runinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16": {
"input_cost_per_token": 5e-08,
"litellm_provider": "runinfra",

View file

@ -419,6 +419,7 @@ class TestRuninfra:
expected_models = {
"runinfra/deepseek-ai/DeepSeek-V4-Flash-0731": (1.3e-07, 2.7e-07, 1e-08),
"runinfra/deepseek-ai/DeepSeek-V4-Pro-0813": (6e-07, 1.9e-06, 3e-08),
"runinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16": (
5e-08,
1.5e-07,
@ -444,6 +445,7 @@ class TestRuninfra:
assert model_cost[model]["supports_prompt_caching"] is True
assert model_cost["runinfra/deepseek-ai/DeepSeek-V4-Flash-0731"]["max_input_tokens"] == 1048576
assert model_cost["runinfra/deepseek-ai/DeepSeek-V4-Pro-0813"]["max_input_tokens"] == 1048576
assert model_cost["runinfra/Qwen/Qwen3.8-27B"]["max_input_tokens"] == 262144
assert model_cost["runinfra/Inferact/Qwen3.8-2.4T-A95B-NVFP4"]["supports_response_schema"] is True
assert model_cost["runinfra/Qwen/Qwen3.8-27B"]["supports_response_schema"] is True