From 184687157e053e106721f14b25cb22f896efe499 Mon Sep 17 00:00:00 2001 From: Krish Dholakia Date: Sun, 10 Aug 2025 07:38:35 -0700 Subject: [PATCH] Litellm model cost map fixes (#13480) * build(model_prices_and_context_window.json): fix max token values * build(model_prices_and_context_window.json): fix max token values * build(model_prices_and_context_window.json): fix azure gpt-5-chat pricing --- .../model_prices_and_context_window_backup.json | 16 ++++++++-------- model_prices_and_context_window.json | 16 ++++++++-------- 2 files changed, 16 insertions(+), 16 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 7cfc9dea5ac..28dec7cce90 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -2457,7 +2457,7 @@ }, "azure/gpt-5-chat": { "max_tokens": 128000, - "max_input_tokens": 128000, + "max_input_tokens": 400000, "max_output_tokens": 128000, "input_cost_per_token": 1.25e-06, "output_cost_per_token": 1e-05, @@ -2490,7 +2490,7 @@ }, "azure/gpt-5-chat-latest": { "max_tokens": 128000, - "max_input_tokens": 128000, + "max_input_tokens": 400000, "max_output_tokens": 128000, "input_cost_per_token": 1.25e-06, "output_cost_per_token": 1e-05, @@ -12582,8 +12582,8 @@ }, "openai.gpt-oss-20b-1:0": { "max_tokens": 128000, - "max_input_tokens": 200000, - "max_output_tokens": 32000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, "input_cost_per_token": 7e-08, "output_cost_per_token": 3e-07, "litellm_provider": "bedrock_converse", @@ -12596,8 +12596,8 @@ }, "openai.gpt-oss-120b-1:0": { "max_tokens": 128000, - "max_input_tokens": 200000, - "max_output_tokens": 32000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, "input_cost_per_token": 1.5e-07, "output_cost_per_token": 6e-07, "litellm_provider": "bedrock_converse", @@ -15605,7 +15605,7 @@ "supports_tool_choice": false }, "fireworks_ai/accounts/fireworks/models/glm-4p5": { - "max_tokens": 128000, + "max_tokens": 96000, "max_input_tokens": 128000, "max_output_tokens": 96000, "input_cost_per_token": 5.5e-07, @@ -15618,7 +15618,7 @@ "source": "https://fireworks.ai/models/fireworks/glm-4p5" }, "fireworks_ai/accounts/fireworks/models/glm-4p5-air": { - "max_tokens": 128000, + "max_tokens": 96000, "max_input_tokens": 128000, "max_output_tokens": 96000, "input_cost_per_token": 2.2e-07, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 7cfc9dea5ac..28dec7cce90 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -2457,7 +2457,7 @@ }, "azure/gpt-5-chat": { "max_tokens": 128000, - "max_input_tokens": 128000, + "max_input_tokens": 400000, "max_output_tokens": 128000, "input_cost_per_token": 1.25e-06, "output_cost_per_token": 1e-05, @@ -2490,7 +2490,7 @@ }, "azure/gpt-5-chat-latest": { "max_tokens": 128000, - "max_input_tokens": 128000, + "max_input_tokens": 400000, "max_output_tokens": 128000, "input_cost_per_token": 1.25e-06, "output_cost_per_token": 1e-05, @@ -12582,8 +12582,8 @@ }, "openai.gpt-oss-20b-1:0": { "max_tokens": 128000, - "max_input_tokens": 200000, - "max_output_tokens": 32000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, "input_cost_per_token": 7e-08, "output_cost_per_token": 3e-07, "litellm_provider": "bedrock_converse", @@ -12596,8 +12596,8 @@ }, "openai.gpt-oss-120b-1:0": { "max_tokens": 128000, - "max_input_tokens": 200000, - "max_output_tokens": 32000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, "input_cost_per_token": 1.5e-07, "output_cost_per_token": 6e-07, "litellm_provider": "bedrock_converse", @@ -15605,7 +15605,7 @@ "supports_tool_choice": false }, "fireworks_ai/accounts/fireworks/models/glm-4p5": { - "max_tokens": 128000, + "max_tokens": 96000, "max_input_tokens": 128000, "max_output_tokens": 96000, "input_cost_per_token": 5.5e-07, @@ -15618,7 +15618,7 @@ "source": "https://fireworks.ai/models/fireworks/glm-4p5" }, "fireworks_ai/accounts/fireworks/models/glm-4p5-air": { - "max_tokens": 128000, + "max_tokens": 96000, "max_input_tokens": 128000, "max_output_tokens": 96000, "input_cost_per_token": 2.2e-07,