mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
feat: add DeepSeek V4 Pro to the runinfra provider
Fifth hosted model, USD per token from the published per-1M rates (0.60 input, 1.90 output, 0.03 cached input read), 1,048,576 context, 32,768 output cap. Verified live on the public endpoint on 2026-08-18 before this commit: json_object and strict json_schema 3 of 3 each, one tool call round-tripped, a prefix-cache hit on a repeated identical prompt, reasoning content present on the wire when enabled (reasoning is toggleable on this host), and the output cap refusing over-cap requests with 400.
This commit is contained in:
parent
323c90e91f
commit
79e2537cbb
3 changed files with 42 additions and 0 deletions
|
|
@ -48060,6 +48060,26 @@
|
|||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"runinfra/deepseek-ai/DeepSeek-V4-Pro-0813": {
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "runinfra",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.9e-06,
|
||||
"source": "https://runinfra.ai/inference-api/deepseek-v4-pro",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"runinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16": {
|
||||
"input_cost_per_token": 5e-08,
|
||||
"litellm_provider": "runinfra",
|
||||
|
|
|
|||
|
|
@ -48060,6 +48060,26 @@
|
|||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"runinfra/deepseek-ai/DeepSeek-V4-Pro-0813": {
|
||||
"cache_read_input_token_cost": 3e-08,
|
||||
"input_cost_per_token": 6e-07,
|
||||
"litellm_provider": "runinfra",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.9e-06,
|
||||
"source": "https://runinfra.ai/inference-api/deepseek-v4-pro",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions"
|
||||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"runinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16": {
|
||||
"input_cost_per_token": 5e-08,
|
||||
"litellm_provider": "runinfra",
|
||||
|
|
|
|||
|
|
@ -419,6 +419,7 @@ class TestRuninfra:
|
|||
|
||||
expected_models = {
|
||||
"runinfra/deepseek-ai/DeepSeek-V4-Flash-0731": (1.3e-07, 2.7e-07, 1e-08),
|
||||
"runinfra/deepseek-ai/DeepSeek-V4-Pro-0813": (6e-07, 1.9e-06, 3e-08),
|
||||
"runinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16": (
|
||||
5e-08,
|
||||
1.5e-07,
|
||||
|
|
@ -444,6 +445,7 @@ class TestRuninfra:
|
|||
assert model_cost[model]["supports_prompt_caching"] is True
|
||||
|
||||
assert model_cost["runinfra/deepseek-ai/DeepSeek-V4-Flash-0731"]["max_input_tokens"] == 1048576
|
||||
assert model_cost["runinfra/deepseek-ai/DeepSeek-V4-Pro-0813"]["max_input_tokens"] == 1048576
|
||||
assert model_cost["runinfra/Qwen/Qwen3.8-27B"]["max_input_tokens"] == 262144
|
||||
assert model_cost["runinfra/Inferact/Qwen3.8-2.4T-A95B-NVFP4"]["supports_response_schema"] is True
|
||||
assert model_cost["runinfra/Qwen/Qwen3.8-27B"]["supports_response_schema"] is True
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue