From 9482e771ad103a57f092b9d27b35cde1bbf84b10 Mon Sep 17 00:00:00 2001 From: Dushyant Acharya Date: Sun, 3 May 2026 00:03:52 +0530 Subject: [PATCH] feat(cost): add cost mapping for deepseek-v4-flash and deepseek-v4-pro Adds pricing entries for the two new DeepSeek V4 models released on 2026-04-24, for both bare model names and the deepseek/ provider prefix. Prices sourced from https://api-docs.deepseek.com/quick_start/pricing: - deepseek-v4-flash: $0.14/M input, $0.28/M output - deepseek-v4-pro: $1.74/M input, $3.48/M output Cache hit price set to 1/10 of input (per DeepSeek docs). Context window: 1M tokens for both models. Closes #26709 --- model_prices_and_context_window.json | 48 ++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index a3b8499d8a5..6ddcdc7a961 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -12573,6 +12573,54 @@ "supports_system_messages": true, "supports_tool_choice": false }, + "deepseek/deepseek-v4-flash": { + "cache_creation_input_token_cost": 0.0, + "cache_read_input_token_cost": 1.4e-08, + "input_cost_per_token": 1.4e-07, + "input_cost_per_token_cache_hit": 1.4e-08, + "litellm_provider": "deepseek", + "max_input_tokens": 1000000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 2.8e-07, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, + "deepseek/deepseek-v4-pro": { + "cache_creation_input_token_cost": 0.0, + "cache_read_input_token_cost": 1.74e-07, + "input_cost_per_token": 1.74e-06, + "input_cost_per_token_cache_hit": 1.74e-07, + "litellm_provider": "deepseek", + "max_input_tokens": 1000000, + "max_output_tokens": 8192, + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 3.48e-06, + "source": "https://api-docs.deepseek.com/quick_start/pricing", + "supported_endpoints": [ + "/v1/chat/completions" + ], + "supports_assistant_prefill": true, + "supports_function_calling": true, + "supports_native_streaming": true, + "supports_parallel_function_calling": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true + }, "deepseek/deepseek-v3": { "cache_creation_input_token_cost": 0.0, "cache_read_input_token_cost": 7e-08,