From acab4761a210cd883653c04083925c9d927d6747 Mon Sep 17 00:00:00 2001 From: "berriai-litellm-provider-info-sync[bot]" <328147090+berriai-litellm-provider-info-sync[bot]@users.noreply.github.com> Date: Thu, 8 Oct 2026 17:23:42 -0700 Subject: [PATCH] fix(cost-map): update baseten DeepSeek-V4.1-Flash-Fast cache read price and max output (#45468) Price-Sync: litellm-providers Co-authored-by: berriai-litellm-provider-info-sync[bot] <328147090+berriai-litellm-provider-info-sync[bot]@users.noreply.github.com> --- litellm/model_prices_and_context_window_backup.json | 6 +++--- model_prices_and_context_window.json | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index bbd0d9ab989..3b2e0af5f49 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -80491,12 +80491,12 @@ "supports_web_search": true }, "baseten/deepseek-ai/DeepSeek-V4.1-Flash-Fast": { - "cache_read_input_token_cost": 1.4e-07, + "cache_read_input_token_cost": 1.4e-08, "input_cost_per_token": 6e-07, "litellm_provider": "baseten", "max_input_tokens": 1048576, - "max_output_tokens": 32768, - "max_tokens": 32768, + "max_output_tokens": 131072, + "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 2.4e-06, "source": "https://inference.baseten.co/v1/models", diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index bbd0d9ab989..3b2e0af5f49 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -80491,12 +80491,12 @@ "supports_web_search": true }, "baseten/deepseek-ai/DeepSeek-V4.1-Flash-Fast": { - "cache_read_input_token_cost": 1.4e-07, + "cache_read_input_token_cost": 1.4e-08, "input_cost_per_token": 6e-07, "litellm_provider": "baseten", "max_input_tokens": 1048576, - "max_output_tokens": 32768, - "max_tokens": 32768, + "max_output_tokens": 131072, + "max_tokens": 131072, "mode": "chat", "output_cost_per_token": 2.4e-06, "source": "https://inference.baseten.co/v1/models",