From f4a7c04d992b48f8cbdd0945087d506739bed9fc Mon Sep 17 00:00:00 2001 From: "berriai-litellm-provider-info-sync[bot]" <328147090+berriai-litellm-provider-info-sync[bot]@users.noreply.github.com> Date: Tue, 29 Sep 2026 19:05:57 +0000 Subject: [PATCH] chore(cost-map): add openai gpt-6-astra ultrafast tier prices from the pricing page (#43745) * chore(cost-map): add openai gpt-6-astra ultrafast tier prices from the pricing page Price-Sync: litellm-providers * feat(cost): support openai ultrafast tier fields in the model catalog --------- Co-authored-by: berriai-litellm-provider-info-sync[bot] <328147090+berriai-litellm-provider-info-sync[bot]@users.noreply.github.com> Co-authored-by: Kerry --- .../crates/model-catalog/src/model_info.rs | 24 +++++++++++++ ...odel_prices_and_context_window_backup.json | 8 +++++ model_prices_and_context_window.json | 8 +++++ model_prices_and_context_window.schema.json | 36 +++++++++++++++++++ 4 files changed, 76 insertions(+) diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 75f16e0c00d..380f6713d7a 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -57,6 +57,9 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(skip_serializing_if = "Option::is_none")] pub cache_creation_input_token_cost_above_272k_tokens_priority: Option, + /// Ultrafast service-tier rate for the same-named base field. + #[serde(skip_serializing_if = "Option::is_none")] + pub cache_creation_input_token_cost_above_272k_tokens_ultrafast: Option, #[serde(skip_serializing_if = "Option::is_none")] pub cache_creation_input_token_cost_batches: Option, /// Flex service-tier rate for the same-named base field. @@ -65,6 +68,9 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(skip_serializing_if = "Option::is_none")] pub cache_creation_input_token_cost_priority: Option, + /// Ultrafast service-tier rate for the same-named base field. + #[serde(skip_serializing_if = "Option::is_none")] + pub cache_creation_input_token_cost_ultrafast: Option, #[serde(skip_serializing_if = "Option::is_none")] pub cache_read_input_audio_token_cost: Option, #[serde(skip_serializing_if = "Option::is_none")] @@ -101,6 +107,9 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(skip_serializing_if = "Option::is_none")] pub cache_read_input_token_cost_above_272k_tokens_priority: Option, + /// Ultrafast service-tier rate for the same-named base field. + #[serde(skip_serializing_if = "Option::is_none")] + pub cache_read_input_token_cost_above_272k_tokens_ultrafast: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(skip_serializing_if = "Option::is_none")] pub cache_read_input_token_cost_above_512k_tokens: Option, @@ -115,6 +124,9 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(skip_serializing_if = "Option::is_none")] pub cache_read_input_token_cost_priority: Option, + /// Ultrafast service-tier rate for the same-named base field. + #[serde(skip_serializing_if = "Option::is_none")] + pub cache_read_input_token_cost_ultrafast: Option, #[serde(skip_serializing_if = "Option::is_none")] pub citation_cost_per_token: Option, #[serde(skip_serializing_if = "Option::is_none")] @@ -213,6 +225,9 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(skip_serializing_if = "Option::is_none")] pub input_cost_per_token_above_272k_tokens_priority: Option, + /// Ultrafast service-tier rate for the same-named base field. + #[serde(skip_serializing_if = "Option::is_none")] + pub input_cost_per_token_above_272k_tokens_ultrafast: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(skip_serializing_if = "Option::is_none")] pub input_cost_per_token_above_512k_tokens: Option, @@ -230,6 +245,9 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(skip_serializing_if = "Option::is_none")] pub input_cost_per_token_priority: Option, + /// Ultrafast service-tier rate for the same-named base field. + #[serde(skip_serializing_if = "Option::is_none")] + pub input_cost_per_token_ultrafast: Option, #[serde(skip_serializing_if = "Option::is_none")] pub input_cost_per_video_per_second: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. @@ -362,6 +380,9 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(skip_serializing_if = "Option::is_none")] pub output_cost_per_token_above_272k_tokens_priority: Option, + /// Ultrafast service-tier rate for the same-named base field. + #[serde(skip_serializing_if = "Option::is_none")] + pub output_cost_per_token_above_272k_tokens_ultrafast: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(skip_serializing_if = "Option::is_none")] pub output_cost_per_token_above_512k_tokens: Option, @@ -377,6 +398,9 @@ pub struct ModelInfo { /// Priority service-tier rate for the same-named base field. #[serde(skip_serializing_if = "Option::is_none")] pub output_cost_per_token_priority: Option, + /// Ultrafast service-tier rate for the same-named base field. + #[serde(skip_serializing_if = "Option::is_none")] + pub output_cost_per_token_ultrafast: Option, #[serde(skip_serializing_if = "Option::is_none")] pub output_cost_per_video_per_second: Option, #[serde(skip_serializing_if = "Option::is_none")] diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 9ae270243d2..c147f4bf94b 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -33870,16 +33870,22 @@ "cache_read_input_token_cost_above_272k_tokens_batches": 1e-06, "cache_creation_input_token_cost_batches": 6.25e-06, "cache_creation_input_token_cost_above_272k_tokens_batches": 1.25e-05, + "cache_creation_input_token_cost_above_272k_tokens_ultrafast": 0.00015, + "cache_creation_input_token_cost_ultrafast": 7.5e-05, + "cache_read_input_token_cost_above_272k_tokens_ultrafast": 1.2e-05, "cache_read_input_token_cost_flex": 5e-07, "cache_read_input_token_cost_priority": 2e-06, + "cache_read_input_token_cost_ultrafast": 6e-06, "input_cost_per_token": 1e-05, "input_cost_per_token_above_272k_tokens": 2e-05, "input_cost_per_token_above_272k_tokens_flex": 1e-05, "input_cost_per_token_above_272k_tokens_priority": 4e-05, "input_cost_per_token_batches": 5e-06, "input_cost_per_token_above_272k_tokens_batches": 1e-05, + "input_cost_per_token_above_272k_tokens_ultrafast": 0.00012, "input_cost_per_token_flex": 5e-06, "input_cost_per_token_priority": 2e-05, + "input_cost_per_token_ultrafast": 6e-05, "litellm_provider": "openai", "max_input_tokens": 922000, "max_output_tokens": 128000, @@ -33891,8 +33897,10 @@ "output_cost_per_token_above_272k_tokens_priority": 0.00015, "output_cost_per_token_batches": 2.5e-05, "output_cost_per_token_above_272k_tokens_batches": 3.75e-05, + "output_cost_per_token_above_272k_tokens_ultrafast": 0.00045, "output_cost_per_token_flex": 2.5e-05, "output_cost_per_token_priority": 0.0001, + "output_cost_per_token_ultrafast": 0.0003, "regional_processing_uplift_multiplier_eu": 1.1, "regional_processing_uplift_multiplier_us": 1.1, "search_context_cost_per_query": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 9ae270243d2..c147f4bf94b 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -33870,16 +33870,22 @@ "cache_read_input_token_cost_above_272k_tokens_batches": 1e-06, "cache_creation_input_token_cost_batches": 6.25e-06, "cache_creation_input_token_cost_above_272k_tokens_batches": 1.25e-05, + "cache_creation_input_token_cost_above_272k_tokens_ultrafast": 0.00015, + "cache_creation_input_token_cost_ultrafast": 7.5e-05, + "cache_read_input_token_cost_above_272k_tokens_ultrafast": 1.2e-05, "cache_read_input_token_cost_flex": 5e-07, "cache_read_input_token_cost_priority": 2e-06, + "cache_read_input_token_cost_ultrafast": 6e-06, "input_cost_per_token": 1e-05, "input_cost_per_token_above_272k_tokens": 2e-05, "input_cost_per_token_above_272k_tokens_flex": 1e-05, "input_cost_per_token_above_272k_tokens_priority": 4e-05, "input_cost_per_token_batches": 5e-06, "input_cost_per_token_above_272k_tokens_batches": 1e-05, + "input_cost_per_token_above_272k_tokens_ultrafast": 0.00012, "input_cost_per_token_flex": 5e-06, "input_cost_per_token_priority": 2e-05, + "input_cost_per_token_ultrafast": 6e-05, "litellm_provider": "openai", "max_input_tokens": 922000, "max_output_tokens": 128000, @@ -33891,8 +33897,10 @@ "output_cost_per_token_above_272k_tokens_priority": 0.00015, "output_cost_per_token_batches": 2.5e-05, "output_cost_per_token_above_272k_tokens_batches": 3.75e-05, + "output_cost_per_token_above_272k_tokens_ultrafast": 0.00045, "output_cost_per_token_flex": 2.5e-05, "output_cost_per_token_priority": 0.0001, + "output_cost_per_token_ultrafast": 0.0003, "regional_processing_uplift_multiplier_eu": 1.1, "regional_processing_uplift_multiplier_us": 1.1, "search_context_cost_per_query": { diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index c4b424a26b4..cdf023e71ef 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -133,6 +133,11 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "cache_creation_input_token_cost_above_272k_tokens_ultrafast": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "cache_creation_input_token_cost_above_32k_tokens": { "type": "number", "minimum": 0, @@ -152,6 +157,10 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "cache_creation_input_token_cost_ultrafast": { + "type": "number", + "minimum": 0 + }, "cache_read_input_audio_token_cost": { "type": "number", "minimum": 0 @@ -210,6 +219,11 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "cache_read_input_token_cost_above_272k_tokens_ultrafast": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "cache_read_input_token_cost_above_32k_tokens": { "type": "number", "minimum": 0, @@ -238,6 +252,10 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "cache_read_input_token_cost_ultrafast": { + "type": "number", + "minimum": 0 + }, "citation_cost_per_token": { "type": "number", "minimum": 0 @@ -404,6 +422,11 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "input_cost_per_token_above_272k_tokens_ultrafast": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "input_cost_per_token_above_32k_tokens": { "type": "number", "minimum": 0, @@ -437,6 +460,10 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "input_cost_per_token_ultrafast": { + "type": "number", + "minimum": 0 + }, "input_cost_per_video_per_second": { "type": "number", "minimum": 0 @@ -770,6 +797,11 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "output_cost_per_token_above_272k_tokens_ultrafast": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "output_cost_per_token_above_32k_tokens": { "type": "number", "minimum": 0, @@ -799,6 +831,10 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "output_cost_per_token_ultrafast": { + "type": "number", + "minimum": 0 + }, "output_cost_per_video_per_second": { "type": "number", "minimum": 0