From 46d440ae9604f81b39d66152b1aceed87e0f9d30 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 7 Oct 2026 14:14:47 -0700 Subject: [PATCH] fix(model_prices): consolidate claude-haiku-5-5 over-100k pricing and capability flags (#45151) * feat(types): declare above_100k_tokens price fields on ModelInfoBase Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit 186f6c81a06a7c80b3496a09aedb86f4fdafc9ba) * fix(router): mirror above_100k pricing fields Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit 75c9c3ff1e19024385ef5ef57211d22153d7b9eb) * chore(ui): regenerate schema.d.ts for above_100k pricing fields Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit 53163d21843e331840683e1f45024ca6b372dac8) * refactor(types): mark above_100k ModelInfoBase fields ReadOnly Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit febed3199a3b8018af70ddae82616d5daf80241d) * fix(model_prices): bill claude-haiku-5-5 long prompts on every provider and in batch Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit 218c00cf5a6d1e1b6361e64615401b23dee583bd) * fix(cost): pass *_above_Nk_tokens_batches rates through get_model_info Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit 37b6f2fba0a3063df796ea22dd4611a9b34d01a2) * fix(model_prices): allow disabling thinking and forced tool use on claude-haiku-5-5 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit 6e91e0ed363d0657e94e6d8fd00ef5f90b2bc97d) * test(model_prices): cite the vendor source for claude-haiku-5-5 capability flags Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit 1f97bb27b49b9cd4764d662bf5bfcc81cfd45398) * test(model_prices): type and tidy the claude-haiku-5-5 config tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit 3d2526035f588acf7aa106a9535eebe9ff09d6e1) * feat(bedrock): add claude haiku 5.5 over 100k token tier Price-Sync: litellm-providers (cherry picked from commit ee3822c29c8513858bf2b513198b82d0f6a5b10e) * feat(model_prices): add openrouter claude-haiku-5.5 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(model_prices): dedupe above_100k pricing keys from text merge Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(model_prices): bill vertex claude-haiku-5-5 prompts over 100k tokens Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(model_prices): add adaptive thinking and cache minimum to openrouter claude-haiku-5.5 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: berriai-litellm-provider-info-sync[bot] <328147090+berriai-litellm-provider-info-sync[bot]@users.noreply.github.com> --- .../crates/model-catalog/src/model_info.rs | 18 ++ .../tests/registry_validation.rs | 17 ++ ...odel_prices_and_context_window_backup.json | 183 ++++++++++++++---- litellm/types/utils.py | 20 +- model_prices_and_context_window.json | 183 ++++++++++++++---- model_prices_and_context_window.schema.json | 20 ++ .../llm_cost_calc/test_llm_cost_calc_utils.py | 135 +++++++++++++ tests/unit/test_claude_haiku_5_5_config.py | 128 ++++++++++++ tests/unit/test_utils.py | 4 + ui/litellm-dashboard/src/lib/http/schema.d.ts | 36 ++++ 10 files changed, 675 insertions(+), 69 deletions(-) create mode 100644 tests/unit/test_claude_haiku_5_5_config.py diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 96dec84de84..ebee2c64140 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -29,10 +29,16 @@ pub struct ModelInfo { #[serde(skip_serializing_if = "Option::is_none")] pub cache_creation_input_token_cost_above_32k_tokens: Option, #[serde(skip_serializing_if = "Option::is_none")] + pub cache_creation_input_token_cost_above_100k_tokens: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub cache_creation_input_token_cost_above_100k_tokens_batches: Option, + #[serde(skip_serializing_if = "Option::is_none")] pub cache_creation_input_token_cost_above_128k_tokens: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(skip_serializing_if = "Option::is_none")] pub cache_creation_input_token_cost_above_1hr: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub cache_creation_input_token_cost_above_1hr_above_100k_tokens: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(skip_serializing_if = "Option::is_none")] pub cache_creation_input_token_cost_above_1hr_above_200k_tokens: Option, @@ -82,6 +88,10 @@ pub struct ModelInfo { #[serde(skip_serializing_if = "Option::is_none")] pub cache_read_input_token_cost_above_32k_tokens: Option, #[serde(skip_serializing_if = "Option::is_none")] + pub cache_read_input_token_cost_above_100k_tokens: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub cache_read_input_token_cost_above_100k_tokens_batches: Option, + #[serde(skip_serializing_if = "Option::is_none")] pub cache_read_input_token_cost_above_128k_tokens: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(skip_serializing_if = "Option::is_none")] @@ -200,6 +210,10 @@ pub struct ModelInfo { #[serde(skip_serializing_if = "Option::is_none")] pub input_cost_per_token_above_32k_tokens: Option, #[serde(skip_serializing_if = "Option::is_none")] + pub input_cost_per_token_above_100k_tokens: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub input_cost_per_token_above_100k_tokens_batches: Option, + #[serde(skip_serializing_if = "Option::is_none")] pub input_cost_per_token_above_128k_tokens: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(skip_serializing_if = "Option::is_none")] @@ -355,6 +369,10 @@ pub struct ModelInfo { #[serde(skip_serializing_if = "Option::is_none")] pub output_cost_per_token_above_32k_tokens: Option, #[serde(skip_serializing_if = "Option::is_none")] + pub output_cost_per_token_above_100k_tokens: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub output_cost_per_token_above_100k_tokens_batches: Option, + #[serde(skip_serializing_if = "Option::is_none")] pub output_cost_per_token_above_128k_tokens: Option, /// Rate applied once the prompt exceeds the token threshold in the field name. #[serde(skip_serializing_if = "Option::is_none")] diff --git a/litellm-rust/crates/model-catalog/tests/registry_validation.rs b/litellm-rust/crates/model-catalog/tests/registry_validation.rs index 8f555e1f884..67589010aba 100644 --- a/litellm-rust/crates/model-catalog/tests/registry_validation.rs +++ b/litellm-rust/crates/model-catalog/tests/registry_validation.rs @@ -74,3 +74,20 @@ fn checked_in_catalog_and_backup_match() { "invalid registry aliases" ); } + +#[rstest] +#[case::input("input_cost_per_token_above_100k_tokens")] +#[case::input_batches("input_cost_per_token_above_100k_tokens_batches")] +#[case::output("output_cost_per_token_above_100k_tokens")] +#[case::output_batches("output_cost_per_token_above_100k_tokens_batches")] +#[case::cache_creation("cache_creation_input_token_cost_above_100k_tokens")] +#[case::cache_creation_batches("cache_creation_input_token_cost_above_100k_tokens_batches")] +#[case::cache_creation_1hr("cache_creation_input_token_cost_above_1hr_above_100k_tokens")] +#[case::cache_read("cache_read_input_token_cost_above_100k_tokens")] +#[case::cache_read_batches("cache_read_input_token_cost_above_100k_tokens_batches")] +fn registry_validation_keeps_above_100k_tier_rates(#[case] field: &str) { + let mut entry = Map::new(); + entry.insert("litellm_provider".into(), "anthropic".into()); + entry.insert(field.into(), 5e-7.into()); + validate_model_entry("test", &Value::Object(entry)).unwrap(); +} diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index cecc7d0856f..c2aa83a9216 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -42380,6 +42380,40 @@ "supports_response_schema": true, "supports_web_search": true }, + "openrouter/anthropic/claude-haiku-5.5": { + "input_cost_per_token": 1e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 1e-08, + "cache_creation_input_token_cost": 1.25e-07, + "cache_creation_input_token_cost_above_1hr": 2e-07, + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "litellm_provider": "openrouter", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "source": "https://openrouter.ai/api/v1/models", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_web_search": true, + "supports_adaptive_thinking": true, + "prompt_cache_min_tokens": 512, + "supports_sampling_params": false + }, "openrouter/anthropic/claude-haiku-4.5": { "cache_creation_input_token_cost": 1.25e-06, "cache_creation_input_token_cost_above_1hr": 2e-06, @@ -80770,6 +80804,10 @@ "supports_vision": true }, "claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens_batches": 2.5e-07, + "output_cost_per_token_above_100k_tokens_batches": 1.25e-06, + "cache_creation_input_token_cost_above_100k_tokens_batches": 3.125e-07, + "cache_read_input_token_cost_above_100k_tokens_batches": 2.5e-08, "supports_anthropic_compaction": true, "cache_creation_input_token_cost": 1.25e-07, "cache_creation_input_token_cost_above_1hr": 2e-07, @@ -80809,8 +80847,8 @@ "us": 1.1 }, "supports_output_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, "source": "https://platform.claude.com/docs/en/models/haiku-5-5/overview", "supports_web_search": true, @@ -80821,6 +80859,11 @@ "cache_read_input_token_cost_above_100k_tokens": 5e-08 }, "bedrock_mantle/anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_1hr": 2.2e-07, @@ -80856,12 +80899,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-haiku-5-5.html" }, "bedrock_mantle/us-gov-west-1/anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 6e-07, + "output_cost_per_token_above_100k_tokens": 3e-06, + "cache_creation_input_token_cost_above_100k_tokens": 7.5e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06, + "cache_read_input_token_cost_above_100k_tokens": 6e-08, "bedrock_converse_supports_strict_tools": false, "bedrock_output_config_effort_ceiling": "xhigh", "cache_creation_input_token_cost": 1.5e-07, @@ -80875,8 +80923,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 6e-07, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, "supports_adaptive_thinking": true, "supports_assistant_prefill": false, @@ -80898,6 +80946,11 @@ "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-haiku-5-5.html" }, "anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.25e-07, "cache_creation_input_token_cost_above_1hr": 2e-07, @@ -80933,12 +80986,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "apac.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "litellm_provider": "bedrock_converse", "supports_tool_search": true, @@ -80964,8 +81022,8 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, "cache_creation_input_token_cost_above_1hr": 2.2e-07, "cache_creation_input_token_cost": 1.375e-07, @@ -80975,6 +81033,11 @@ "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "au.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_1hr": 2.2e-07, @@ -81010,12 +81073,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "azure_ai/claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, "supports_mid_conversation_system": true, "cache_creation_input_token_cost": 1.25e-07, "cache_creation_input_token_cost_above_1hr": 2e-07, @@ -81052,6 +81120,11 @@ "source": "https://platform.claude.com/docs/en/models/haiku-5-5/migration-guide" }, "bedrock/us-gov-east-1/anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 6e-07, + "output_cost_per_token_above_100k_tokens": 3e-06, + "cache_creation_input_token_cost_above_100k_tokens": 7.5e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06, + "cache_read_input_token_cost_above_100k_tokens": 6e-08, "bedrock_converse_supports_strict_tools": false, "bedrock_output_config_effort_ceiling": "xhigh", "cache_creation_input_token_cost": 1.5e-07, @@ -81065,9 +81138,10 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 6e-07, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json", "supports_adaptive_thinking": true, "supports_assistant_prefill": false, "supports_computer_use": true, @@ -81087,6 +81161,11 @@ "supports_xhigh_reasoning_effort": true }, "bedrock/us-gov-west-1/anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 6e-07, + "output_cost_per_token_above_100k_tokens": 3e-06, + "cache_creation_input_token_cost_above_100k_tokens": 7.5e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06, + "cache_read_input_token_cost_above_100k_tokens": 6e-08, "bedrock_converse_supports_strict_tools": false, "bedrock_output_config_effort_ceiling": "xhigh", "cache_creation_input_token_cost": 1.5e-07, @@ -81100,9 +81179,10 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 6e-07, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json", "supports_adaptive_thinking": true, "supports_assistant_prefill": false, "supports_computer_use": true, @@ -81122,6 +81202,11 @@ "supports_xhigh_reasoning_effort": true }, "eu.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_1hr": 2.2e-07, @@ -81157,12 +81242,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.25e-07, "cache_creation_input_token_cost_above_1hr": 2e-07, @@ -81198,12 +81288,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "jp.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_1hr": 2.2e-07, @@ -81239,10 +81334,10 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "perplexity/anthropic/claude-haiku-5-5": { "litellm_provider": "perplexity", @@ -81256,6 +81351,11 @@ "source": "https://docs.perplexity.ai/docs/agent-api/models" }, "us-gov.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 6e-07, + "output_cost_per_token_above_100k_tokens": 3e-06, + "cache_creation_input_token_cost_above_100k_tokens": 7.5e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06, + "cache_read_input_token_cost_above_100k_tokens": 6e-08, "bedrock_converse_supports_strict_tools": false, "bedrock_output_config_effort_ceiling": "xhigh", "cache_creation_input_token_cost": 1.5e-07, @@ -81269,10 +81369,10 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 6e-07, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/", + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json", "supports_adaptive_thinking": true, "supports_assistant_prefill": false, "supports_computer_use": true, @@ -81292,6 +81392,11 @@ "supports_xhigh_reasoning_effort": true }, "us.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_1hr": 2.2e-07, @@ -81327,12 +81432,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "vertex_ai/claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, "regional_endpoint_uplift_multiplier": 1.1, "supports_mid_conversation_system": true, "cache_creation_input_token_cost": 1.25e-07, @@ -81370,9 +81480,14 @@ "supports_forced_tool_use": true, "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://platform.claude.com/docs/en/about-claude/pricing" + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, "vertex_ai/claude-haiku-5-5@default": { + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, "regional_endpoint_uplift_multiplier": 1.1, "supports_mid_conversation_system": true, "cache_creation_input_token_cost": 1.25e-07, @@ -81410,6 +81525,6 @@ "supports_forced_tool_use": true, "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://platform.claude.com/docs/en/about-claude/pricing" + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" } } diff --git a/litellm/types/utils.py b/litellm/types/utils.py index f8f8388b6f7..507bfa6c5e1 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -291,6 +291,8 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False): input_cost_per_token_ultrafast: ReadOnly[float | None] # OpenAI ultrafast service tier pricing cache_creation_input_token_cost: float | None cache_creation_input_token_cost_above_200k_tokens: float | None + cache_creation_input_token_cost_above_100k_tokens: ReadOnly[float | None] + cache_creation_input_token_cost_above_1hr_above_100k_tokens: ReadOnly[float | None] cache_creation_input_token_cost_above_272k_tokens: float | None cache_creation_input_token_cost_above_272k_tokens_priority: float | None cache_creation_input_token_cost_above_272k_tokens_flex: float | None @@ -307,6 +309,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False): cache_read_input_token_cost_balanced: ReadOnly[float | None] cache_read_input_token_cost_ultrafast: ReadOnly[float | None] # OpenAI ultrafast service tier pricing cache_read_input_token_cost_above_200k_tokens: float | None + cache_read_input_token_cost_above_100k_tokens: ReadOnly[float | None] cache_read_input_token_cost_above_200k_tokens_priority: float | None cache_read_input_token_cost_above_272k_tokens: float | None cache_read_input_token_cost_above_272k_tokens_priority: float | None @@ -315,9 +318,11 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False): cache_read_input_token_cost_above_512k_tokens: float | None cache_read_input_token_cost_batches: ReadOnly[float | None] cache_read_input_token_cost_above_200k_tokens_batches: ReadOnly[float | None] + cache_read_input_token_cost_above_100k_tokens_batches: ReadOnly[float | None] cache_read_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None] cache_creation_input_token_cost_batches: ReadOnly[float | None] cache_creation_input_token_cost_above_200k_tokens_batches: ReadOnly[float | None] + cache_creation_input_token_cost_above_100k_tokens_batches: ReadOnly[float | None] cache_creation_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None] # Smallest prefix this model will actually cache, whatever caching mechanism its provider uses. # Absent means the provider-agnostic default applies; see MINIMUM_PROMPT_CACHE_TOKEN_COUNT. @@ -327,6 +332,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False): input_cost_per_audio_token: float | None input_cost_per_token_above_128k_tokens: float | None # only for vertex ai models input_cost_per_token_above_200k_tokens: float | None # only for vertex ai gemini-2.5-pro models + input_cost_per_token_above_100k_tokens: ReadOnly[float | None] input_cost_per_token_above_200k_tokens_priority: float | None input_cost_per_token_above_272k_tokens: float | None # GPT-5.4/5.4-pro: prompts >272K priced at 2x input input_cost_per_token_above_272k_tokens_priority: float | None @@ -347,9 +353,11 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False): input_cost_per_token_batches: float | None input_cost_per_video_token_batches: ReadOnly[float | None] input_cost_per_token_above_200k_tokens_batches: ReadOnly[float | None] + input_cost_per_token_above_100k_tokens_batches: ReadOnly[float | None] input_cost_per_token_above_272k_tokens_batches: ReadOnly[float | None] output_cost_per_token_batches: float | None output_cost_per_token_above_200k_tokens_batches: ReadOnly[float | None] + output_cost_per_token_above_100k_tokens_batches: ReadOnly[float | None] output_cost_per_token_above_272k_tokens_batches: ReadOnly[float | None] output_cost_per_token: Required[float | None] output_cost_per_token_flex: float | None # OpenAI flex service tier pricing @@ -369,6 +377,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False): output_cost_per_audio_token: float | None output_cost_per_token_above_128k_tokens: float | None # only for vertex ai models output_cost_per_token_above_200k_tokens: float | None # only for vertex ai gemini-2.5-pro models + output_cost_per_token_above_100k_tokens: ReadOnly[float | None] output_cost_per_token_above_200k_tokens_priority: float | None output_cost_per_token_above_272k_tokens: float | None # GPT-5.4/5.4-pro: prompts >272K priced at 1.5x output output_cost_per_token_above_272k_tokens_priority: float | None @@ -3816,6 +3825,8 @@ class CustomPricingLiteLLMParams(MirroredPricingParams): input_cost_per_token_balanced: float | None = None input_cost_per_token_ultrafast: float | None = None cache_creation_input_token_cost_above_1hr: float | None = None + cache_creation_input_token_cost_above_100k_tokens: float | None = None + cache_creation_input_token_cost_above_1hr_above_100k_tokens: float | None = None cache_creation_input_token_cost_above_200k_tokens: float | None = None cache_creation_input_token_cost_above_272k_tokens: float | None = None cache_creation_input_token_cost_above_272k_tokens_priority: float | None = None @@ -3829,15 +3840,18 @@ class CustomPricingLiteLLMParams(MirroredPricingParams): cache_read_input_token_cost_priority: float | None = None cache_read_input_token_cost_balanced: float | None = None cache_read_input_token_cost_ultrafast: float | None = None + cache_read_input_token_cost_above_100k_tokens: float | None = None cache_read_input_token_cost_above_200k_tokens: float | None = None cache_read_input_token_cost_above_200k_tokens_priority: float | None = None cache_read_input_token_cost_above_272k_tokens_priority: float | None = None cache_read_input_token_cost_above_272k_tokens_flex: float | None = None cache_read_input_token_cost_above_272k_tokens_ultrafast: float | None = None cache_read_input_token_cost_batches: float | None = None + cache_read_input_token_cost_above_100k_tokens_batches: float | None = None cache_read_input_token_cost_above_200k_tokens_batches: float | None = None cache_read_input_token_cost_above_272k_tokens_batches: float | None = None cache_creation_input_token_cost_batches: float | None = None + cache_creation_input_token_cost_above_100k_tokens_batches: float | None = None cache_creation_input_token_cost_above_200k_tokens_batches: float | None = None cache_creation_input_token_cost_above_272k_tokens_batches: float | None = None cache_read_input_audio_token_cost: float | None = None @@ -3846,12 +3860,14 @@ class CustomPricingLiteLLMParams(MirroredPricingParams): input_cost_per_audio_token: float | None = None input_cost_per_token_cache_hit: float | None = None input_cost_per_token_above_128k_tokens: float | None = None + input_cost_per_token_above_100k_tokens: float | None = None input_cost_per_token_above_200k_tokens: float | None = None input_cost_per_token_above_200k_tokens_priority: float | None = None input_cost_per_token_above_272k_tokens_priority: float | None = None input_cost_per_token_above_272k_tokens_flex: float | None = None input_cost_per_token_above_272k_tokens_ultrafast: float | None = None input_cost_per_token_above_200k_tokens_batches: float | None = None + input_cost_per_token_above_100k_tokens_batches: float | None = None input_cost_per_token_above_272k_tokens_batches: float | None = None input_cost_per_query: float | None = None input_cost_per_image: float | None = None @@ -3873,12 +3889,14 @@ class CustomPricingLiteLLMParams(MirroredPricingParams): output_cost_per_token_ultrafast: float | None = None output_cost_per_audio_token: float | None = None output_cost_per_token_above_128k_tokens: float | None = None + output_cost_per_token_above_100k_tokens: float | None = None output_cost_per_token_above_200k_tokens: float | None = None output_cost_per_token_above_200k_tokens_priority: float | None = None output_cost_per_token_above_272k_tokens_priority: float | None = None output_cost_per_token_above_272k_tokens_flex: float | None = None output_cost_per_token_above_272k_tokens_ultrafast: float | None = None output_cost_per_token_above_200k_tokens_batches: float | None = None + output_cost_per_token_above_100k_tokens_batches: float | None = None output_cost_per_token_above_272k_tokens_batches: float | None = None output_cost_per_character_above_128k_tokens: float | None = None output_cost_per_image: float | None = None @@ -3946,7 +3964,7 @@ def shared_backend_model_info(model_info: dict[str, Any]) -> dict[str, Any]: return {k: v for k, v in model_info.items() if k in SHARED_BACKEND_MODEL_INFO_FIELDS} -ABOVE_THRESHOLD_COST_KEY_PATTERN: Final = re.compile(r"_above_\d+k?_tokens$") +ABOVE_THRESHOLD_COST_KEY_PATTERN: Final = re.compile(r"_above_\d+k?_tokens(?:_batches)?$") _PRICING_FIELD_EXEMPTIONS: Final[frozenset[str]] = frozenset({"output_vector_size"}) diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index cecc7d0856f..c2aa83a9216 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -42380,6 +42380,40 @@ "supports_response_schema": true, "supports_web_search": true }, + "openrouter/anthropic/claude-haiku-5.5": { + "input_cost_per_token": 1e-07, + "output_cost_per_token": 5e-07, + "cache_read_input_token_cost": 1e-08, + "cache_creation_input_token_cost": 1.25e-07, + "cache_creation_input_token_cost_above_1hr": 2e-07, + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "litellm_provider": "openrouter", + "max_input_tokens": 1000000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "source": "https://openrouter.ai/api/v1/models", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_pdf_input": true, + "supports_prompt_caching": true, + "supports_web_search": true, + "supports_adaptive_thinking": true, + "prompt_cache_min_tokens": 512, + "supports_sampling_params": false + }, "openrouter/anthropic/claude-haiku-4.5": { "cache_creation_input_token_cost": 1.25e-06, "cache_creation_input_token_cost_above_1hr": 2e-06, @@ -80770,6 +80804,10 @@ "supports_vision": true }, "claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens_batches": 2.5e-07, + "output_cost_per_token_above_100k_tokens_batches": 1.25e-06, + "cache_creation_input_token_cost_above_100k_tokens_batches": 3.125e-07, + "cache_read_input_token_cost_above_100k_tokens_batches": 2.5e-08, "supports_anthropic_compaction": true, "cache_creation_input_token_cost": 1.25e-07, "cache_creation_input_token_cost_above_1hr": 2e-07, @@ -80809,8 +80847,8 @@ "us": 1.1 }, "supports_output_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, "source": "https://platform.claude.com/docs/en/models/haiku-5-5/overview", "supports_web_search": true, @@ -80821,6 +80859,11 @@ "cache_read_input_token_cost_above_100k_tokens": 5e-08 }, "bedrock_mantle/anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_1hr": 2.2e-07, @@ -80856,12 +80899,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-haiku-5-5.html" }, "bedrock_mantle/us-gov-west-1/anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 6e-07, + "output_cost_per_token_above_100k_tokens": 3e-06, + "cache_creation_input_token_cost_above_100k_tokens": 7.5e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06, + "cache_read_input_token_cost_above_100k_tokens": 6e-08, "bedrock_converse_supports_strict_tools": false, "bedrock_output_config_effort_ceiling": "xhigh", "cache_creation_input_token_cost": 1.5e-07, @@ -80875,8 +80923,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 6e-07, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, "supports_adaptive_thinking": true, "supports_assistant_prefill": false, @@ -80898,6 +80946,11 @@ "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-haiku-5-5.html" }, "anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.25e-07, "cache_creation_input_token_cost_above_1hr": 2e-07, @@ -80933,12 +80986,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "apac.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "litellm_provider": "bedrock_converse", "supports_tool_search": true, @@ -80964,8 +81022,8 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, "cache_creation_input_token_cost_above_1hr": 2.2e-07, "cache_creation_input_token_cost": 1.375e-07, @@ -80975,6 +81033,11 @@ "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "au.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_1hr": 2.2e-07, @@ -81010,12 +81073,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "azure_ai/claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, "supports_mid_conversation_system": true, "cache_creation_input_token_cost": 1.25e-07, "cache_creation_input_token_cost_above_1hr": 2e-07, @@ -81052,6 +81120,11 @@ "source": "https://platform.claude.com/docs/en/models/haiku-5-5/migration-guide" }, "bedrock/us-gov-east-1/anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 6e-07, + "output_cost_per_token_above_100k_tokens": 3e-06, + "cache_creation_input_token_cost_above_100k_tokens": 7.5e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06, + "cache_read_input_token_cost_above_100k_tokens": 6e-08, "bedrock_converse_supports_strict_tools": false, "bedrock_output_config_effort_ceiling": "xhigh", "cache_creation_input_token_cost": 1.5e-07, @@ -81065,9 +81138,10 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 6e-07, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json", "supports_adaptive_thinking": true, "supports_assistant_prefill": false, "supports_computer_use": true, @@ -81087,6 +81161,11 @@ "supports_xhigh_reasoning_effort": true }, "bedrock/us-gov-west-1/anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 6e-07, + "output_cost_per_token_above_100k_tokens": 3e-06, + "cache_creation_input_token_cost_above_100k_tokens": 7.5e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06, + "cache_read_input_token_cost_above_100k_tokens": 6e-08, "bedrock_converse_supports_strict_tools": false, "bedrock_output_config_effort_ceiling": "xhigh", "cache_creation_input_token_cost": 1.5e-07, @@ -81100,9 +81179,10 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 6e-07, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json", "supports_adaptive_thinking": true, "supports_assistant_prefill": false, "supports_computer_use": true, @@ -81122,6 +81202,11 @@ "supports_xhigh_reasoning_effort": true }, "eu.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_1hr": 2.2e-07, @@ -81157,12 +81242,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "global.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.25e-07, "cache_creation_input_token_cost_above_1hr": 2e-07, @@ -81198,12 +81288,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "jp.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_1hr": 2.2e-07, @@ -81239,10 +81334,10 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "perplexity/anthropic/claude-haiku-5-5": { "litellm_provider": "perplexity", @@ -81256,6 +81351,11 @@ "source": "https://docs.perplexity.ai/docs/agent-api/models" }, "us-gov.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 6e-07, + "output_cost_per_token_above_100k_tokens": 3e-06, + "cache_creation_input_token_cost_above_100k_tokens": 7.5e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06, + "cache_read_input_token_cost_above_100k_tokens": 6e-08, "bedrock_converse_supports_strict_tools": false, "bedrock_output_config_effort_ceiling": "xhigh", "cache_creation_input_token_cost": 1.5e-07, @@ -81269,10 +81369,10 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 6e-07, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/", + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json", "supports_adaptive_thinking": true, "supports_assistant_prefill": false, "supports_computer_use": true, @@ -81292,6 +81392,11 @@ "supports_xhigh_reasoning_effort": true }, "us.anthropic.claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5.5e-07, + "output_cost_per_token_above_100k_tokens": 2.75e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.875e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5.5e-08, "bedrock_converse_supports_strict_tools": false, "cache_creation_input_token_cost": 1.375e-07, "cache_creation_input_token_cost_above_1hr": 2.2e-07, @@ -81327,12 +81432,17 @@ "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh", "supports_parallel_tool_use_config": true, - "supports_forced_tool_use": false, - "thinking_always_on": true, + "supports_forced_tool_use": true, + "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://aws.amazon.com/bedrock/pricing/" + "source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json" }, "vertex_ai/claude-haiku-5-5": { + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, "regional_endpoint_uplift_multiplier": 1.1, "supports_mid_conversation_system": true, "cache_creation_input_token_cost": 1.25e-07, @@ -81370,9 +81480,14 @@ "supports_forced_tool_use": true, "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://platform.claude.com/docs/en/about-claude/pricing" + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, "vertex_ai/claude-haiku-5-5@default": { + "input_cost_per_token_above_100k_tokens": 5e-07, + "output_cost_per_token_above_100k_tokens": 2.5e-06, + "cache_creation_input_token_cost_above_100k_tokens": 6.25e-07, + "cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06, + "cache_read_input_token_cost_above_100k_tokens": 5e-08, "regional_endpoint_uplift_multiplier": 1.1, "supports_mid_conversation_system": true, "cache_creation_input_token_cost": 1.25e-07, @@ -81410,6 +81525,6 @@ "supports_forced_tool_use": true, "thinking_always_on": false, "prompt_cache_min_tokens": 512, - "source": "https://platform.claude.com/docs/en/about-claude/pricing" + "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" } } diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index 41b616a6b1f..e51ae0453ea 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -88,6 +88,11 @@ "minimum": 0, "description": "Rate applied once the prompt exceeds the token threshold in the field name." }, + "cache_creation_input_token_cost_above_100k_tokens_batches": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "cache_creation_input_token_cost_above_128k_tokens": { "type": "number", "minimum": 0, @@ -189,6 +194,11 @@ "minimum": 0, "description": "Rate applied once the prompt exceeds the token threshold in the field name." }, + "cache_read_input_token_cost_above_100k_tokens_batches": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "cache_read_input_token_cost_above_128k_tokens": { "type": "number", "minimum": 0, @@ -397,6 +407,11 @@ "minimum": 0, "description": "Rate applied once the prompt exceeds the token threshold in the field name." }, + "input_cost_per_token_above_100k_tokens_batches": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "input_cost_per_token_above_128k_tokens": { "type": "number", "minimum": 0, @@ -781,6 +796,11 @@ "minimum": 0, "description": "Rate applied once the prompt exceeds the token threshold in the field name." }, + "output_cost_per_token_above_100k_tokens_batches": { + "type": "number", + "minimum": 0, + "description": "Rate applied once the prompt exceeds the token threshold in the field name." + }, "output_cost_per_token_above_128k_tokens": { "type": "number", "minimum": 0, diff --git a/tests/unit/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/unit/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index ea628794c2c..2ce354b546f 100644 --- a/tests/unit/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/unit/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -3830,3 +3830,138 @@ def test_azure_gpt_5_6_alias_matches_sol_pricing(_local_model_cost_map, region_p assert shared_cost_fields for field in shared_cost_fields: assert alias[field] == sol[field], field + + +# Per-token rates read 2026-10-07 from https://platform.claude.com/docs/en/about-claude/pricing (direct and +# azure_ai, which Microsoft bills at Anthropic's rates per +# https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/claude-models-billing) and from the +# AmazonBedrockFoundationModels price list at +# https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json (Bedrock) +@pytest.mark.parametrize( + ("model", "custom_llm_provider", "prompt_tokens", "input_rate", "cache_read_rate", "output_rate"), + [ + ("claude-haiku-5-5", "anthropic", 100_000, 1e-07, 1e-08, 5e-07), + ("claude-haiku-5-5", "anthropic", 100_001, 5e-07, 5e-08, 2.5e-06), + ("azure_ai/claude-haiku-5-5", "azure_ai", 100_000, 1e-07, 1e-08, 5e-07), + ("azure_ai/claude-haiku-5-5", "azure_ai", 100_001, 5e-07, 5e-08, 2.5e-06), + ("global.anthropic.claude-haiku-5-5", "bedrock", 100_000, 1e-07, 1e-08, 5e-07), + ("global.anthropic.claude-haiku-5-5", "bedrock", 100_001, 5e-07, 5e-08, 2.5e-06), + ("us.anthropic.claude-haiku-5-5", "bedrock", 100_000, 1.1e-07, 1.1e-08, 5.5e-07), + ("us.anthropic.claude-haiku-5-5", "bedrock", 100_001, 5.5e-07, 5.5e-08, 2.75e-06), + ("us-gov.anthropic.claude-haiku-5-5", "bedrock", 100_000, 1.2e-07, 1.2e-08, 6e-07), + ("us-gov.anthropic.claude-haiku-5-5", "bedrock", 100_001, 6e-07, 6e-08, 3e-06), + ("bedrock_mantle/anthropic.claude-haiku-5-5", "bedrock_mantle", 100_001, 5.5e-07, 5.5e-08, 2.75e-06), + ("bedrock/us-gov-west-1/anthropic.claude-haiku-5-5", "bedrock", 100_001, 6e-07, 6e-08, 3e-06), + ("vertex_ai/claude-haiku-5-5", "vertex_ai", 100_000, 1e-07, 1e-08, 5e-07), + ("vertex_ai/claude-haiku-5-5", "vertex_ai", 100_001, 5e-07, 5e-08, 2.5e-06), + ], +) +def test_generic_cost_per_token_claude_haiku_5_5_prompt_length_tiers( + _local_model_cost_map: None, + model: str, + custom_llm_provider: str, + prompt_tokens: int, + input_rate: float, + cache_read_rate: float, + output_rate: float, +) -> None: + """Claude Haiku 5.5 bills every token at 5x the base rates once the prompt is over 100,000 tokens.""" + cached_tokens: Final = 10_000 + completion_tokens: Final = 1_000 + usage: Final = Usage( + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + total_tokens=prompt_tokens + completion_tokens, + prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens), + ) + + prompt_cost, completion_cost = generic_cost_per_token( + model=model, + usage=usage, + custom_llm_provider=custom_llm_provider, + ) + + assert prompt_cost == pytest.approx((prompt_tokens - cached_tokens) * input_rate + cached_tokens * cache_read_rate) + assert completion_cost == pytest.approx(completion_tokens * output_rate) + + +def test_vertex_regional_endpoint_uplift_scales_claude_haiku_5_5_over_100k_rates( + _local_model_cost_map: None, +) -> None: + """Vertex regional endpoints bill 1.1x the global rate on all token types + (https://cloud.google.com/vertex-ai/generative-ai/pricing, 2026-10-07: regional + over-100K input is $0.55/MTok), so the uplift scales the over-100k rates too.""" + cached_tokens: Final = 10_000 + prompt_tokens: Final = 100_001 + completion_tokens: Final = 1_000 + usage: Final = Usage( + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + total_tokens=prompt_tokens + completion_tokens, + prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens), + ) + + prompt_cost, completion_cost = generic_cost_per_token( + model="vertex_ai/claude-haiku-5-5", + usage=usage, + custom_llm_provider="vertex_ai", + vertex_location="us-east5", + ) + + assert prompt_cost == pytest.approx((prompt_tokens - cached_tokens) * 5.5e-07 + cached_tokens * 5.5e-08) + assert completion_cost == pytest.approx(completion_tokens * 2.75e-06) + + +# Batch rates read 2026-10-07 from the Batch processing table at +# https://platform.claude.com/docs/en/about-claude/pricing: $0.05 / $0.25 per MTok input and $0.25 / $1.25 output, +# up to and over 100,000 prompt tokens +@pytest.mark.parametrize( + ("prompt_tokens", "input_rate", "output_rate"), + [(100_000, 5e-08, 2.5e-07), (100_001, 2.5e-07, 1.25e-06)], +) +def test_batch_cost_calculator_claude_haiku_5_5_prompt_length_tiers( + _local_model_cost_map: None, + prompt_tokens: int, + input_rate: float, + output_rate: float, +) -> None: + from litellm.cost_calculator import batch_cost_calculator + + completion_tokens: Final = 1_000 + + prompt_cost, completion_cost = batch_cost_calculator( + usage=Usage( + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + total_tokens=prompt_tokens + completion_tokens, + ), + model="claude-haiku-5-5", + custom_llm_provider="anthropic", + ) + + assert prompt_cost == pytest.approx(prompt_tokens * input_rate) + assert completion_cost == pytest.approx(completion_tokens * output_rate) + + +@pytest.mark.parametrize( + ("prompt_tokens", "expected"), + [ + (100_000, (5e-08, 2.5e-07, 5e-09, 6.25e-08)), + (100_001, (2.5e-07, 1.25e-06, 2.5e-08, 3.125e-07)), + ], +) +def test_get_batch_cost_rates_claude_haiku_5_5_prompt_length_tiers( + _local_model_cost_map: None, + prompt_tokens: int, + expected: tuple[float, float, float, float], +) -> None: + """Cache write and cache read batch rates are 50% of the standard rates; Anthropic's batch table omits them.""" + from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates + + rates: Final = get_batch_cost_rates( + litellm.get_model_info(model="claude-haiku-5-5", custom_llm_provider="anthropic"), + Usage(prompt_tokens=prompt_tokens, completion_tokens=1, total_tokens=prompt_tokens + 1), + "anthropic", + ) + + assert (rates.input, rates.output, rates.cache_read, rates.cache_creation) == expected diff --git a/tests/unit/test_claude_haiku_5_5_config.py b/tests/unit/test_claude_haiku_5_5_config.py new file mode 100644 index 00000000000..99c7d167df4 --- /dev/null +++ b/tests/unit/test_claude_haiku_5_5_config.py @@ -0,0 +1,128 @@ +""" +Validate Claude Haiku 5.5 model configuration entries. + +Haiku 5.5 ships with adaptive thinking on by default, but unlike Sonnet 5.5 / +Opus 5.5 thinking can still be turned off (``thinking: disabled`` at high +effort or below) and it accepts a forced ``tool_choice`` (``any`` or a named +tool). Its cost-map rows therefore carry ``thinking_always_on: false`` and +``supports_forced_tool_use: true``. +""" + +import json +import os +from collections.abc import Iterator +from typing import Final, cast + +import pytest + +import litellm +from litellm.litellm_core_utils.get_model_cost_map import GetModelCostMap +from litellm.llms.anthropic.common_utils import AnthropicModelInfo + +REPO_ROOT: Final = os.path.join(os.path.dirname(__file__), "../..") + +GET_WEATHER_TOOL: Final = { + "type": "function", + "function": { + "name": "get_weather", + "description": "Get the weather", + "parameters": { + "type": "object", + "properties": {"city": {"type": "string"}}, + }, + }, +} + + +@pytest.fixture(autouse=True) +def local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> Iterator[None]: + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + litellm.get_model_info.cache_clear() + yield + litellm.get_model_info.cache_clear() + + +def _load_root_cost_map() -> dict[str, dict[str, object]]: + json_path: Final = os.path.join(REPO_ROOT, "model_prices_and_context_window.json") + with open(json_path) as f: + return cast(dict[str, dict[str, object]], json.load(f)) + + +HAIKU_5_5_VARIANTS: Final = ( + "claude-haiku-5-5", + "anthropic.claude-haiku-5-5", + "apac.anthropic.claude-haiku-5-5", + "au.anthropic.claude-haiku-5-5", + "eu.anthropic.claude-haiku-5-5", + "global.anthropic.claude-haiku-5-5", + "jp.anthropic.claude-haiku-5-5", + "us.anthropic.claude-haiku-5-5", + "us-gov.anthropic.claude-haiku-5-5", + "bedrock/us-gov-east-1/anthropic.claude-haiku-5-5", + "bedrock/us-gov-west-1/anthropic.claude-haiku-5-5", + "bedrock_mantle/anthropic.claude-haiku-5-5", + "bedrock_mantle/us-gov-west-1/anthropic.claude-haiku-5-5", + "vertex_ai/claude-haiku-5-5", + "vertex_ai/claude-haiku-5-5@default", +) + + +@pytest.mark.parametrize("model_name", HAIKU_5_5_VARIANTS) +def test_haiku_5_5_rows_allow_disabling_thinking_and_forced_tools( + model_name: str, +) -> None: + root: Final = _load_root_cost_map() + backup: Final = GetModelCostMap.load_local_model_cost_map() + assert model_name in root + row: Final = root[model_name] + # https://platform.claude.com/docs/en/models/haiku-5-5/whats-new-haiku-5-5 (2026-10-07): + # thinking can be disabled, forced tool_choice accepted + assert row["thinking_always_on"] is False + assert row["supports_forced_tool_use"] is True + assert backup[model_name] == row + + +@pytest.mark.parametrize( + ("model", "provider"), + [ + ("claude-haiku-5-5", "anthropic"), + ("anthropic/claude-haiku-5-5", "anthropic"), + ("vertex_ai/claude-haiku-5-5", "vertex_ai"), + ], +) +def test_haiku_5_5_runtime_profile(local_model_cost_map: None, model: str, provider: str) -> None: + assert AnthropicModelInfo.is_adaptive_thinking_model(model, provider) is True + assert AnthropicModelInfo._is_always_on_thinking_model(model, provider) is False + assert AnthropicModelInfo.forced_tool_use_unsupported(model.removeprefix("anthropic/")) is False + + +def test_haiku_5_5_anthropic_tool_choice_required_maps_to_any( + local_model_cost_map: None, +) -> None: + optional_params: Final = litellm.AnthropicConfig().map_openai_params( + non_default_params={ + "tools": [dict(GET_WEATHER_TOOL)], + "tool_choice": "required", + }, + optional_params={}, + model="claude-haiku-5-5", + drop_params=False, + ) + assert optional_params["tool_choice"] == {"type": "any"} + + +def test_haiku_5_5_bedrock_tool_choice_required_maps_to_any( + local_model_cost_map: None, +) -> None: + optional_params: Final = litellm.AmazonConverseConfig().map_openai_params( + non_default_params={ + "tools": [dict(GET_WEATHER_TOOL)], + "tool_choice": "required", + }, + optional_params={}, + model="us.anthropic.claude-haiku-5-5", + drop_params=False, + ) + assert optional_params["tool_choice"] == {"any": {}} diff --git a/tests/unit/test_utils.py b/tests/unit/test_utils.py index 09dc0feed1e..ac5dc6e7c45 100644 --- a/tests/unit/test_utils.py +++ b/tests/unit/test_utils.py @@ -794,6 +794,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_creation_input_token_cost_above_1hr": {"type": "number"}, "cache_creation_input_token_cost_above_32k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_100k_tokens": {"type": "number"}, + "cache_creation_input_token_cost_above_100k_tokens_batches": {"type": "number"}, "cache_creation_input_token_cost_above_128k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_200k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_256k_tokens": {"type": "number"}, @@ -810,6 +811,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_read_input_token_cost": {"type": "number"}, "cache_read_input_token_cost_above_32k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_100k_tokens": {"type": "number"}, + "cache_read_input_token_cost_above_100k_tokens_batches": {"type": "number"}, "cache_read_input_token_cost_above_128k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_200k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_200k_tokens_batches": {"type": "number"}, @@ -839,6 +841,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "input_cost_per_video_token": {"type": "number"}, "input_cost_per_token_above_32k_tokens": {"type": "number"}, "input_cost_per_token_above_100k_tokens": {"type": "number"}, + "input_cost_per_token_above_100k_tokens_batches": {"type": "number"}, "input_cost_per_token_above_200k_tokens": {"type": "number"}, "input_cost_per_token_above_200k_tokens_batches": {"type": "number"}, "input_cost_per_token_above_256k_tokens": {"type": "number"}, @@ -949,6 +952,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "output_cost_per_token": {"type": "number"}, "output_cost_per_token_above_32k_tokens": {"type": "number"}, "output_cost_per_token_above_100k_tokens": {"type": "number"}, + "output_cost_per_token_above_100k_tokens_batches": {"type": "number"}, "output_cost_per_token_above_128k_tokens": {"type": "number"}, "output_cost_per_token_above_200k_tokens": {"type": "number"}, "output_cost_per_token_above_200k_tokens_batches": {"type": "number"}, diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 79b98295d3c..3ada53d1641 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -35560,8 +35560,14 @@ export interface components { cache_creation_input_audio_token_cost?: number | null; /** Cache Creation Input Token Cost */ cache_creation_input_token_cost?: number | null; + /** Cache Creation Input Token Cost Above 100K Tokens */ + cache_creation_input_token_cost_above_100k_tokens?: number | null; + /** Cache Creation Input Token Cost Above 100K Tokens Batches */ + cache_creation_input_token_cost_above_100k_tokens_batches?: number | null; /** Cache Creation Input Token Cost Above 1Hr */ cache_creation_input_token_cost_above_1hr?: number | null; + /** Cache Creation Input Token Cost Above 1Hr Above 100K Tokens */ + cache_creation_input_token_cost_above_1hr_above_100k_tokens?: number | null; /** Cache Creation Input Token Cost Above 200K Tokens */ cache_creation_input_token_cost_above_200k_tokens?: number | null; /** Cache Creation Input Token Cost Above 200K Tokens Batches */ @@ -35590,6 +35596,10 @@ export interface components { cache_read_input_image_token_cost?: number | null; /** Cache Read Input Token Cost */ cache_read_input_token_cost?: number | null; + /** Cache Read Input Token Cost Above 100K Tokens */ + cache_read_input_token_cost_above_100k_tokens?: number | null; + /** Cache Read Input Token Cost Above 100K Tokens Batches */ + cache_read_input_token_cost_above_100k_tokens_batches?: number | null; /** Cache Read Input Token Cost Above 200K Tokens */ cache_read_input_token_cost_above_200k_tokens?: number | null; /** Cache Read Input Token Cost Above 200K Tokens Batches */ @@ -35674,6 +35684,10 @@ export interface components { input_cost_per_second?: number | null; /** Input Cost Per Token */ input_cost_per_token?: number | null; + /** Input Cost Per Token Above 100K Tokens */ + input_cost_per_token_above_100k_tokens?: number | null; + /** Input Cost Per Token Above 100K Tokens Batches */ + input_cost_per_token_above_100k_tokens_batches?: number | null; /** Input Cost Per Token Above 128K Tokens */ input_cost_per_token_above_128k_tokens?: number | null; /** Input Cost Per Token Above 200K Tokens */ @@ -35811,6 +35825,10 @@ export interface components { output_cost_per_second_768p?: number | null; /** Output Cost Per Token */ output_cost_per_token?: number | null; + /** Output Cost Per Token Above 100K Tokens */ + output_cost_per_token_above_100k_tokens?: number | null; + /** Output Cost Per Token Above 100K Tokens Batches */ + output_cost_per_token_above_100k_tokens_batches?: number | null; /** Output Cost Per Token Above 128K Tokens */ output_cost_per_token_above_128k_tokens?: number | null; /** Output Cost Per Token Above 200K Tokens */ @@ -50548,8 +50566,14 @@ export interface components { cache_creation_input_audio_token_cost?: number | null; /** Cache Creation Input Token Cost */ cache_creation_input_token_cost?: number | null; + /** Cache Creation Input Token Cost Above 100K Tokens */ + cache_creation_input_token_cost_above_100k_tokens?: number | null; + /** Cache Creation Input Token Cost Above 100K Tokens Batches */ + cache_creation_input_token_cost_above_100k_tokens_batches?: number | null; /** Cache Creation Input Token Cost Above 1Hr */ cache_creation_input_token_cost_above_1hr?: number | null; + /** Cache Creation Input Token Cost Above 1Hr Above 100K Tokens */ + cache_creation_input_token_cost_above_1hr_above_100k_tokens?: number | null; /** Cache Creation Input Token Cost Above 200K Tokens */ cache_creation_input_token_cost_above_200k_tokens?: number | null; /** Cache Creation Input Token Cost Above 200K Tokens Batches */ @@ -50578,6 +50602,10 @@ export interface components { cache_read_input_image_token_cost?: number | null; /** Cache Read Input Token Cost */ cache_read_input_token_cost?: number | null; + /** Cache Read Input Token Cost Above 100K Tokens */ + cache_read_input_token_cost_above_100k_tokens?: number | null; + /** Cache Read Input Token Cost Above 100K Tokens Batches */ + cache_read_input_token_cost_above_100k_tokens_batches?: number | null; /** Cache Read Input Token Cost Above 200K Tokens */ cache_read_input_token_cost_above_200k_tokens?: number | null; /** Cache Read Input Token Cost Above 200K Tokens Batches */ @@ -50662,6 +50690,10 @@ export interface components { input_cost_per_second?: number | null; /** Input Cost Per Token */ input_cost_per_token?: number | null; + /** Input Cost Per Token Above 100K Tokens */ + input_cost_per_token_above_100k_tokens?: number | null; + /** Input Cost Per Token Above 100K Tokens Batches */ + input_cost_per_token_above_100k_tokens_batches?: number | null; /** Input Cost Per Token Above 128K Tokens */ input_cost_per_token_above_128k_tokens?: number | null; /** Input Cost Per Token Above 200K Tokens */ @@ -50799,6 +50831,10 @@ export interface components { output_cost_per_second_768p?: number | null; /** Output Cost Per Token */ output_cost_per_token?: number | null; + /** Output Cost Per Token Above 100K Tokens */ + output_cost_per_token_above_100k_tokens?: number | null; + /** Output Cost Per Token Above 100K Tokens Batches */ + output_cost_per_token_above_100k_tokens_batches?: number | null; /** Output Cost Per Token Above 128K Tokens */ output_cost_per_token_above_128k_tokens?: number | null; /** Output Cost Per Token Above 200K Tokens */