mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(model_prices): consolidate claude-haiku-5-5 over-100k pricing and capability flags (#45151)
* feat(types): declare above_100k_tokens price fields on ModelInfoBase Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit186f6c81a0) * fix(router): mirror above_100k pricing fields Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit75c9c3ff1e) * chore(ui): regenerate schema.d.ts for above_100k pricing fields Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit53163d2184) * refactor(types): mark above_100k ModelInfoBase fields ReadOnly Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commitfebed3199a) * fix(model_prices): bill claude-haiku-5-5 long prompts on every provider and in batch Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit218c00cf5a) * fix(cost): pass *_above_Nk_tokens_batches rates through get_model_info Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit37b6f2fba0) * fix(model_prices): allow disabling thinking and forced tool use on claude-haiku-5-5 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit6e91e0ed36) * test(model_prices): cite the vendor source for claude-haiku-5-5 capability flags Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit1f97bb27b4) * test(model_prices): type and tidy the claude-haiku-5-5 config tests Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> (cherry picked from commit3d2526035f) * feat(bedrock): add claude haiku 5.5 over 100k token tier Price-Sync: litellm-providers (cherry picked from commitee3822c29c) * feat(model_prices): add openrouter claude-haiku-5.5 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(model_prices): dedupe above_100k pricing keys from text merge Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(model_prices): bill vertex claude-haiku-5-5 prompts over 100k tokens Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(model_prices): add adaptive thinking and cache minimum to openrouter claude-haiku-5.5 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: berriai-litellm-provider-info-sync[bot] <328147090+berriai-litellm-provider-info-sync[bot]@users.noreply.github.com>
This commit is contained in:
parent
fa2c8984ba
commit
46d440ae96
10 changed files with 675 additions and 69 deletions
|
|
@ -29,10 +29,16 @@ pub struct ModelInfo {
|
|||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_creation_input_token_cost_above_32k_tokens: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_creation_input_token_cost_above_100k_tokens: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_creation_input_token_cost_above_100k_tokens_batches: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_creation_input_token_cost_above_128k_tokens: Option<f64>,
|
||||
/// Rate applied once the prompt exceeds the token threshold in the field name.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_creation_input_token_cost_above_1hr: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_creation_input_token_cost_above_1hr_above_100k_tokens: Option<f64>,
|
||||
/// Rate applied once the prompt exceeds the token threshold in the field name.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_creation_input_token_cost_above_1hr_above_200k_tokens: Option<f64>,
|
||||
|
|
@ -82,6 +88,10 @@ pub struct ModelInfo {
|
|||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_read_input_token_cost_above_32k_tokens: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_read_input_token_cost_above_100k_tokens: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_read_input_token_cost_above_100k_tokens_batches: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub cache_read_input_token_cost_above_128k_tokens: Option<f64>,
|
||||
/// Rate applied once the prompt exceeds the token threshold in the field name.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
|
|
@ -200,6 +210,10 @@ pub struct ModelInfo {
|
|||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub input_cost_per_token_above_32k_tokens: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub input_cost_per_token_above_100k_tokens: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub input_cost_per_token_above_100k_tokens_batches: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub input_cost_per_token_above_128k_tokens: Option<f64>,
|
||||
/// Rate applied once the prompt exceeds the token threshold in the field name.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
|
|
@ -355,6 +369,10 @@ pub struct ModelInfo {
|
|||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub output_cost_per_token_above_32k_tokens: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub output_cost_per_token_above_100k_tokens: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub output_cost_per_token_above_100k_tokens_batches: Option<f64>,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub output_cost_per_token_above_128k_tokens: Option<f64>,
|
||||
/// Rate applied once the prompt exceeds the token threshold in the field name.
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
|
|
|
|||
|
|
@ -74,3 +74,20 @@ fn checked_in_catalog_and_backup_match() {
|
|||
"invalid registry aliases"
|
||||
);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::input("input_cost_per_token_above_100k_tokens")]
|
||||
#[case::input_batches("input_cost_per_token_above_100k_tokens_batches")]
|
||||
#[case::output("output_cost_per_token_above_100k_tokens")]
|
||||
#[case::output_batches("output_cost_per_token_above_100k_tokens_batches")]
|
||||
#[case::cache_creation("cache_creation_input_token_cost_above_100k_tokens")]
|
||||
#[case::cache_creation_batches("cache_creation_input_token_cost_above_100k_tokens_batches")]
|
||||
#[case::cache_creation_1hr("cache_creation_input_token_cost_above_1hr_above_100k_tokens")]
|
||||
#[case::cache_read("cache_read_input_token_cost_above_100k_tokens")]
|
||||
#[case::cache_read_batches("cache_read_input_token_cost_above_100k_tokens_batches")]
|
||||
fn registry_validation_keeps_above_100k_tier_rates(#[case] field: &str) {
|
||||
let mut entry = Map::new();
|
||||
entry.insert("litellm_provider".into(), "anthropic".into());
|
||||
entry.insert(field.into(), 5e-7.into());
|
||||
validate_model_entry("test", &Value::Object(entry)).unwrap();
|
||||
}
|
||||
|
|
|
|||
|
|
@ -42380,6 +42380,40 @@
|
|||
"supports_response_schema": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"openrouter/anthropic/claude-haiku-5.5": {
|
||||
"input_cost_per_token": 1e-07,
|
||||
"output_cost_per_token": 5e-07,
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-07,
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_vision": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_web_search": true,
|
||||
"supports_adaptive_thinking": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"supports_sampling_params": false
|
||||
},
|
||||
"openrouter/anthropic/claude-haiku-4.5": {
|
||||
"cache_creation_input_token_cost": 1.25e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-06,
|
||||
|
|
@ -80770,6 +80804,10 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens_batches": 2.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens_batches": 1.25e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens_batches": 3.125e-07,
|
||||
"cache_read_input_token_cost_above_100k_tokens_batches": 2.5e-08,
|
||||
"supports_anthropic_compaction": true,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-07,
|
||||
|
|
@ -80809,8 +80847,8 @@
|
|||
"us": 1.1
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://platform.claude.com/docs/en/models/haiku-5-5/overview",
|
||||
"supports_web_search": true,
|
||||
|
|
@ -80821,6 +80859,11 @@
|
|||
"cache_read_input_token_cost_above_100k_tokens": 5e-08
|
||||
},
|
||||
"bedrock_mantle/anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
|
|
@ -80856,12 +80899,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-haiku-5-5.html"
|
||||
},
|
||||
"bedrock_mantle/us-gov-west-1/anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 6e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 3e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 7.5e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 6e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"cache_creation_input_token_cost": 1.5e-07,
|
||||
|
|
@ -80875,8 +80923,8 @@
|
|||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-07,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
|
|
@ -80898,6 +80946,11 @@
|
|||
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-haiku-5-5.html"
|
||||
},
|
||||
"anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-07,
|
||||
|
|
@ -80933,12 +80986,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"apac.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"supports_tool_search": true,
|
||||
|
|
@ -80964,8 +81022,8 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
|
|
@ -80975,6 +81033,11 @@
|
|||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"au.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
|
|
@ -81010,12 +81073,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"azure_ai/claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"supports_mid_conversation_system": true,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-07,
|
||||
|
|
@ -81052,6 +81120,11 @@
|
|||
"source": "https://platform.claude.com/docs/en/models/haiku-5-5/migration-guide"
|
||||
},
|
||||
"bedrock/us-gov-east-1/anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 6e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 3e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 7.5e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 6e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"cache_creation_input_token_cost": 1.5e-07,
|
||||
|
|
@ -81065,9 +81138,10 @@
|
|||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-07,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -81087,6 +81161,11 @@
|
|||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"bedrock/us-gov-west-1/anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 6e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 3e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 7.5e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 6e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"cache_creation_input_token_cost": 1.5e-07,
|
||||
|
|
@ -81100,9 +81179,10 @@
|
|||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-07,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -81122,6 +81202,11 @@
|
|||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"eu.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
|
|
@ -81157,12 +81242,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"global.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-07,
|
||||
|
|
@ -81198,12 +81288,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"jp.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
|
|
@ -81239,10 +81334,10 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"perplexity/anthropic/claude-haiku-5-5": {
|
||||
"litellm_provider": "perplexity",
|
||||
|
|
@ -81256,6 +81351,11 @@
|
|||
"source": "https://docs.perplexity.ai/docs/agent-api/models"
|
||||
},
|
||||
"us-gov.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 6e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 3e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 7.5e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 6e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"cache_creation_input_token_cost": 1.5e-07,
|
||||
|
|
@ -81269,10 +81369,10 @@
|
|||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-07,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -81292,6 +81392,11 @@
|
|||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"us.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
|
|
@ -81327,12 +81432,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"vertex_ai/claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_mid_conversation_system": true,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
|
|
@ -81370,9 +81480,14 @@
|
|||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://platform.claude.com/docs/en/about-claude/pricing"
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing"
|
||||
},
|
||||
"vertex_ai/claude-haiku-5-5@default": {
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_mid_conversation_system": true,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
|
|
@ -81410,6 +81525,6 @@
|
|||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://platform.claude.com/docs/en/about-claude/pricing"
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing"
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -291,6 +291,8 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
input_cost_per_token_ultrafast: ReadOnly[float | None] # OpenAI ultrafast service tier pricing
|
||||
cache_creation_input_token_cost: float | None
|
||||
cache_creation_input_token_cost_above_200k_tokens: float | None
|
||||
cache_creation_input_token_cost_above_100k_tokens: ReadOnly[float | None]
|
||||
cache_creation_input_token_cost_above_1hr_above_100k_tokens: ReadOnly[float | None]
|
||||
cache_creation_input_token_cost_above_272k_tokens: float | None
|
||||
cache_creation_input_token_cost_above_272k_tokens_priority: float | None
|
||||
cache_creation_input_token_cost_above_272k_tokens_flex: float | None
|
||||
|
|
@ -307,6 +309,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
cache_read_input_token_cost_balanced: ReadOnly[float | None]
|
||||
cache_read_input_token_cost_ultrafast: ReadOnly[float | None] # OpenAI ultrafast service tier pricing
|
||||
cache_read_input_token_cost_above_200k_tokens: float | None
|
||||
cache_read_input_token_cost_above_100k_tokens: ReadOnly[float | None]
|
||||
cache_read_input_token_cost_above_200k_tokens_priority: float | None
|
||||
cache_read_input_token_cost_above_272k_tokens: float | None
|
||||
cache_read_input_token_cost_above_272k_tokens_priority: float | None
|
||||
|
|
@ -315,9 +318,11 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
cache_read_input_token_cost_above_512k_tokens: float | None
|
||||
cache_read_input_token_cost_batches: ReadOnly[float | None]
|
||||
cache_read_input_token_cost_above_200k_tokens_batches: ReadOnly[float | None]
|
||||
cache_read_input_token_cost_above_100k_tokens_batches: ReadOnly[float | None]
|
||||
cache_read_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None]
|
||||
cache_creation_input_token_cost_batches: ReadOnly[float | None]
|
||||
cache_creation_input_token_cost_above_200k_tokens_batches: ReadOnly[float | None]
|
||||
cache_creation_input_token_cost_above_100k_tokens_batches: ReadOnly[float | None]
|
||||
cache_creation_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None]
|
||||
# Smallest prefix this model will actually cache, whatever caching mechanism its provider uses.
|
||||
# Absent means the provider-agnostic default applies; see MINIMUM_PROMPT_CACHE_TOKEN_COUNT.
|
||||
|
|
@ -327,6 +332,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
input_cost_per_audio_token: float | None
|
||||
input_cost_per_token_above_128k_tokens: float | None # only for vertex ai models
|
||||
input_cost_per_token_above_200k_tokens: float | None # only for vertex ai gemini-2.5-pro models
|
||||
input_cost_per_token_above_100k_tokens: ReadOnly[float | None]
|
||||
input_cost_per_token_above_200k_tokens_priority: float | None
|
||||
input_cost_per_token_above_272k_tokens: float | None # GPT-5.4/5.4-pro: prompts >272K priced at 2x input
|
||||
input_cost_per_token_above_272k_tokens_priority: float | None
|
||||
|
|
@ -347,9 +353,11 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
input_cost_per_token_batches: float | None
|
||||
input_cost_per_video_token_batches: ReadOnly[float | None]
|
||||
input_cost_per_token_above_200k_tokens_batches: ReadOnly[float | None]
|
||||
input_cost_per_token_above_100k_tokens_batches: ReadOnly[float | None]
|
||||
input_cost_per_token_above_272k_tokens_batches: ReadOnly[float | None]
|
||||
output_cost_per_token_batches: float | None
|
||||
output_cost_per_token_above_200k_tokens_batches: ReadOnly[float | None]
|
||||
output_cost_per_token_above_100k_tokens_batches: ReadOnly[float | None]
|
||||
output_cost_per_token_above_272k_tokens_batches: ReadOnly[float | None]
|
||||
output_cost_per_token: Required[float | None]
|
||||
output_cost_per_token_flex: float | None # OpenAI flex service tier pricing
|
||||
|
|
@ -369,6 +377,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
output_cost_per_audio_token: float | None
|
||||
output_cost_per_token_above_128k_tokens: float | None # only for vertex ai models
|
||||
output_cost_per_token_above_200k_tokens: float | None # only for vertex ai gemini-2.5-pro models
|
||||
output_cost_per_token_above_100k_tokens: ReadOnly[float | None]
|
||||
output_cost_per_token_above_200k_tokens_priority: float | None
|
||||
output_cost_per_token_above_272k_tokens: float | None # GPT-5.4/5.4-pro: prompts >272K priced at 1.5x output
|
||||
output_cost_per_token_above_272k_tokens_priority: float | None
|
||||
|
|
@ -3816,6 +3825,8 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
input_cost_per_token_balanced: float | None = None
|
||||
input_cost_per_token_ultrafast: float | None = None
|
||||
cache_creation_input_token_cost_above_1hr: float | None = None
|
||||
cache_creation_input_token_cost_above_100k_tokens: float | None = None
|
||||
cache_creation_input_token_cost_above_1hr_above_100k_tokens: float | None = None
|
||||
cache_creation_input_token_cost_above_200k_tokens: float | None = None
|
||||
cache_creation_input_token_cost_above_272k_tokens: float | None = None
|
||||
cache_creation_input_token_cost_above_272k_tokens_priority: float | None = None
|
||||
|
|
@ -3829,15 +3840,18 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
cache_read_input_token_cost_priority: float | None = None
|
||||
cache_read_input_token_cost_balanced: float | None = None
|
||||
cache_read_input_token_cost_ultrafast: float | None = None
|
||||
cache_read_input_token_cost_above_100k_tokens: float | None = None
|
||||
cache_read_input_token_cost_above_200k_tokens: float | None = None
|
||||
cache_read_input_token_cost_above_200k_tokens_priority: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_priority: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_flex: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_ultrafast: float | None = None
|
||||
cache_read_input_token_cost_batches: float | None = None
|
||||
cache_read_input_token_cost_above_100k_tokens_batches: float | None = None
|
||||
cache_read_input_token_cost_above_200k_tokens_batches: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_batches: float | None = None
|
||||
cache_creation_input_token_cost_batches: float | None = None
|
||||
cache_creation_input_token_cost_above_100k_tokens_batches: float | None = None
|
||||
cache_creation_input_token_cost_above_200k_tokens_batches: float | None = None
|
||||
cache_creation_input_token_cost_above_272k_tokens_batches: float | None = None
|
||||
cache_read_input_audio_token_cost: float | None = None
|
||||
|
|
@ -3846,12 +3860,14 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
input_cost_per_audio_token: float | None = None
|
||||
input_cost_per_token_cache_hit: float | None = None
|
||||
input_cost_per_token_above_128k_tokens: float | None = None
|
||||
input_cost_per_token_above_100k_tokens: float | None = None
|
||||
input_cost_per_token_above_200k_tokens: float | None = None
|
||||
input_cost_per_token_above_200k_tokens_priority: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_priority: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_flex: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_ultrafast: float | None = None
|
||||
input_cost_per_token_above_200k_tokens_batches: float | None = None
|
||||
input_cost_per_token_above_100k_tokens_batches: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_batches: float | None = None
|
||||
input_cost_per_query: float | None = None
|
||||
input_cost_per_image: float | None = None
|
||||
|
|
@ -3873,12 +3889,14 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
output_cost_per_token_ultrafast: float | None = None
|
||||
output_cost_per_audio_token: float | None = None
|
||||
output_cost_per_token_above_128k_tokens: float | None = None
|
||||
output_cost_per_token_above_100k_tokens: float | None = None
|
||||
output_cost_per_token_above_200k_tokens: float | None = None
|
||||
output_cost_per_token_above_200k_tokens_priority: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_priority: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_flex: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_ultrafast: float | None = None
|
||||
output_cost_per_token_above_200k_tokens_batches: float | None = None
|
||||
output_cost_per_token_above_100k_tokens_batches: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_batches: float | None = None
|
||||
output_cost_per_character_above_128k_tokens: float | None = None
|
||||
output_cost_per_image: float | None = None
|
||||
|
|
@ -3946,7 +3964,7 @@ def shared_backend_model_info(model_info: dict[str, Any]) -> dict[str, Any]:
|
|||
return {k: v for k, v in model_info.items() if k in SHARED_BACKEND_MODEL_INFO_FIELDS}
|
||||
|
||||
|
||||
ABOVE_THRESHOLD_COST_KEY_PATTERN: Final = re.compile(r"_above_\d+k?_tokens$")
|
||||
ABOVE_THRESHOLD_COST_KEY_PATTERN: Final = re.compile(r"_above_\d+k?_tokens(?:_batches)?$")
|
||||
|
||||
_PRICING_FIELD_EXEMPTIONS: Final[frozenset[str]] = frozenset({"output_vector_size"})
|
||||
|
||||
|
|
|
|||
|
|
@ -42380,6 +42380,40 @@
|
|||
"supports_response_schema": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"openrouter/anthropic/claude-haiku-5.5": {
|
||||
"input_cost_per_token": 1e-07,
|
||||
"output_cost_per_token": 5e-07,
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-07,
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1000000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"source": "https://openrouter.ai/api/v1/models",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_vision": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_web_search": true,
|
||||
"supports_adaptive_thinking": true,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"supports_sampling_params": false
|
||||
},
|
||||
"openrouter/anthropic/claude-haiku-4.5": {
|
||||
"cache_creation_input_token_cost": 1.25e-06,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-06,
|
||||
|
|
@ -80770,6 +80804,10 @@
|
|||
"supports_vision": true
|
||||
},
|
||||
"claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens_batches": 2.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens_batches": 1.25e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens_batches": 3.125e-07,
|
||||
"cache_read_input_token_cost_above_100k_tokens_batches": 2.5e-08,
|
||||
"supports_anthropic_compaction": true,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-07,
|
||||
|
|
@ -80809,8 +80847,8 @@
|
|||
"us": 1.1
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://platform.claude.com/docs/en/models/haiku-5-5/overview",
|
||||
"supports_web_search": true,
|
||||
|
|
@ -80821,6 +80859,11 @@
|
|||
"cache_read_input_token_cost_above_100k_tokens": 5e-08
|
||||
},
|
||||
"bedrock_mantle/anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
|
|
@ -80856,12 +80899,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-haiku-5-5.html"
|
||||
},
|
||||
"bedrock_mantle/us-gov-west-1/anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 6e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 3e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 7.5e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 6e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"cache_creation_input_token_cost": 1.5e-07,
|
||||
|
|
@ -80875,8 +80923,8 @@
|
|||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-07,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
|
|
@ -80898,6 +80946,11 @@
|
|||
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-haiku-5-5.html"
|
||||
},
|
||||
"anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-07,
|
||||
|
|
@ -80933,12 +80986,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"apac.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"litellm_provider": "bedrock_converse",
|
||||
"supports_tool_search": true,
|
||||
|
|
@ -80964,8 +81022,8 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
|
|
@ -80975,6 +81033,11 @@
|
|||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"au.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
|
|
@ -81010,12 +81073,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"azure_ai/claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"supports_mid_conversation_system": true,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-07,
|
||||
|
|
@ -81052,6 +81120,11 @@
|
|||
"source": "https://platform.claude.com/docs/en/models/haiku-5-5/migration-guide"
|
||||
},
|
||||
"bedrock/us-gov-east-1/anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 6e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 3e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 7.5e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 6e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"cache_creation_input_token_cost": 1.5e-07,
|
||||
|
|
@ -81065,9 +81138,10 @@
|
|||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-07,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -81087,6 +81161,11 @@
|
|||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"bedrock/us-gov-west-1/anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 6e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 3e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 7.5e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 6e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"cache_creation_input_token_cost": 1.5e-07,
|
||||
|
|
@ -81100,9 +81179,10 @@
|
|||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-07,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -81122,6 +81202,11 @@
|
|||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"eu.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
|
|
@ -81157,12 +81242,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"global.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2e-07,
|
||||
|
|
@ -81198,12 +81288,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"jp.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
|
|
@ -81239,10 +81334,10 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"perplexity/anthropic/claude-haiku-5-5": {
|
||||
"litellm_provider": "perplexity",
|
||||
|
|
@ -81256,6 +81351,11 @@
|
|||
"source": "https://docs.perplexity.ai/docs/agent-api/models"
|
||||
},
|
||||
"us-gov.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 6e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 3e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 7.5e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.2e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 6e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"cache_creation_input_token_cost": 1.5e-07,
|
||||
|
|
@ -81269,10 +81369,10 @@
|
|||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 6e-07,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json",
|
||||
"supports_adaptive_thinking": true,
|
||||
"supports_assistant_prefill": false,
|
||||
"supports_computer_use": true,
|
||||
|
|
@ -81292,6 +81392,11 @@
|
|||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"us.anthropic.claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5.5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.875e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5.5e-08,
|
||||
"bedrock_converse_supports_strict_tools": false,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
"cache_creation_input_token_cost_above_1hr": 2.2e-07,
|
||||
|
|
@ -81327,12 +81432,17 @@
|
|||
"supports_output_config": true,
|
||||
"bedrock_output_config_effort_ceiling": "xhigh",
|
||||
"supports_parallel_tool_use_config": true,
|
||||
"supports_forced_tool_use": false,
|
||||
"thinking_always_on": true,
|
||||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
"source": "https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json"
|
||||
},
|
||||
"vertex_ai/claude-haiku-5-5": {
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_mid_conversation_system": true,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
|
|
@ -81370,9 +81480,14 @@
|
|||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://platform.claude.com/docs/en/about-claude/pricing"
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing"
|
||||
},
|
||||
"vertex_ai/claude-haiku-5-5@default": {
|
||||
"input_cost_per_token_above_100k_tokens": 5e-07,
|
||||
"output_cost_per_token_above_100k_tokens": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_100k_tokens": 6.25e-07,
|
||||
"cache_creation_input_token_cost_above_1hr_above_100k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_above_100k_tokens": 5e-08,
|
||||
"regional_endpoint_uplift_multiplier": 1.1,
|
||||
"supports_mid_conversation_system": true,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
|
|
@ -81410,6 +81525,6 @@
|
|||
"supports_forced_tool_use": true,
|
||||
"thinking_always_on": false,
|
||||
"prompt_cache_min_tokens": 512,
|
||||
"source": "https://platform.claude.com/docs/en/about-claude/pricing"
|
||||
"source": "https://cloud.google.com/vertex-ai/generative-ai/pricing"
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -88,6 +88,11 @@
|
|||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"cache_creation_input_token_cost_above_100k_tokens_batches": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"cache_creation_input_token_cost_above_128k_tokens": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
@ -189,6 +194,11 @@
|
|||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"cache_read_input_token_cost_above_100k_tokens_batches": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"cache_read_input_token_cost_above_128k_tokens": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
@ -397,6 +407,11 @@
|
|||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"input_cost_per_token_above_100k_tokens_batches": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"input_cost_per_token_above_128k_tokens": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
@ -781,6 +796,11 @@
|
|||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"output_cost_per_token_above_100k_tokens_batches": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"output_cost_per_token_above_128k_tokens": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
|
|||
|
|
@ -3830,3 +3830,138 @@ def test_azure_gpt_5_6_alias_matches_sol_pricing(_local_model_cost_map, region_p
|
|||
assert shared_cost_fields
|
||||
for field in shared_cost_fields:
|
||||
assert alias[field] == sol[field], field
|
||||
|
||||
|
||||
# Per-token rates read 2026-10-07 from https://platform.claude.com/docs/en/about-claude/pricing (direct and
|
||||
# azure_ai, which Microsoft bills at Anthropic's rates per
|
||||
# https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/claude-models-billing) and from the
|
||||
# AmazonBedrockFoundationModels price list at
|
||||
# https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrockFoundationModels/current/index.json (Bedrock)
|
||||
@pytest.mark.parametrize(
|
||||
("model", "custom_llm_provider", "prompt_tokens", "input_rate", "cache_read_rate", "output_rate"),
|
||||
[
|
||||
("claude-haiku-5-5", "anthropic", 100_000, 1e-07, 1e-08, 5e-07),
|
||||
("claude-haiku-5-5", "anthropic", 100_001, 5e-07, 5e-08, 2.5e-06),
|
||||
("azure_ai/claude-haiku-5-5", "azure_ai", 100_000, 1e-07, 1e-08, 5e-07),
|
||||
("azure_ai/claude-haiku-5-5", "azure_ai", 100_001, 5e-07, 5e-08, 2.5e-06),
|
||||
("global.anthropic.claude-haiku-5-5", "bedrock", 100_000, 1e-07, 1e-08, 5e-07),
|
||||
("global.anthropic.claude-haiku-5-5", "bedrock", 100_001, 5e-07, 5e-08, 2.5e-06),
|
||||
("us.anthropic.claude-haiku-5-5", "bedrock", 100_000, 1.1e-07, 1.1e-08, 5.5e-07),
|
||||
("us.anthropic.claude-haiku-5-5", "bedrock", 100_001, 5.5e-07, 5.5e-08, 2.75e-06),
|
||||
("us-gov.anthropic.claude-haiku-5-5", "bedrock", 100_000, 1.2e-07, 1.2e-08, 6e-07),
|
||||
("us-gov.anthropic.claude-haiku-5-5", "bedrock", 100_001, 6e-07, 6e-08, 3e-06),
|
||||
("bedrock_mantle/anthropic.claude-haiku-5-5", "bedrock_mantle", 100_001, 5.5e-07, 5.5e-08, 2.75e-06),
|
||||
("bedrock/us-gov-west-1/anthropic.claude-haiku-5-5", "bedrock", 100_001, 6e-07, 6e-08, 3e-06),
|
||||
("vertex_ai/claude-haiku-5-5", "vertex_ai", 100_000, 1e-07, 1e-08, 5e-07),
|
||||
("vertex_ai/claude-haiku-5-5", "vertex_ai", 100_001, 5e-07, 5e-08, 2.5e-06),
|
||||
],
|
||||
)
|
||||
def test_generic_cost_per_token_claude_haiku_5_5_prompt_length_tiers(
|
||||
_local_model_cost_map: None,
|
||||
model: str,
|
||||
custom_llm_provider: str,
|
||||
prompt_tokens: int,
|
||||
input_rate: float,
|
||||
cache_read_rate: float,
|
||||
output_rate: float,
|
||||
) -> None:
|
||||
"""Claude Haiku 5.5 bills every token at 5x the base rates once the prompt is over 100,000 tokens."""
|
||||
cached_tokens: Final = 10_000
|
||||
completion_tokens: Final = 1_000
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model=model,
|
||||
usage=usage,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx((prompt_tokens - cached_tokens) * input_rate + cached_tokens * cache_read_rate)
|
||||
assert completion_cost == pytest.approx(completion_tokens * output_rate)
|
||||
|
||||
|
||||
def test_vertex_regional_endpoint_uplift_scales_claude_haiku_5_5_over_100k_rates(
|
||||
_local_model_cost_map: None,
|
||||
) -> None:
|
||||
"""Vertex regional endpoints bill 1.1x the global rate on all token types
|
||||
(https://cloud.google.com/vertex-ai/generative-ai/pricing, 2026-10-07: regional
|
||||
over-100K input is $0.55/MTok), so the uplift scales the over-100k rates too."""
|
||||
cached_tokens: Final = 10_000
|
||||
prompt_tokens: Final = 100_001
|
||||
completion_tokens: Final = 1_000
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model="vertex_ai/claude-haiku-5-5",
|
||||
usage=usage,
|
||||
custom_llm_provider="vertex_ai",
|
||||
vertex_location="us-east5",
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx((prompt_tokens - cached_tokens) * 5.5e-07 + cached_tokens * 5.5e-08)
|
||||
assert completion_cost == pytest.approx(completion_tokens * 2.75e-06)
|
||||
|
||||
|
||||
# Batch rates read 2026-10-07 from the Batch processing table at
|
||||
# https://platform.claude.com/docs/en/about-claude/pricing: $0.05 / $0.25 per MTok input and $0.25 / $1.25 output,
|
||||
# up to and over 100,000 prompt tokens
|
||||
@pytest.mark.parametrize(
|
||||
("prompt_tokens", "input_rate", "output_rate"),
|
||||
[(100_000, 5e-08, 2.5e-07), (100_001, 2.5e-07, 1.25e-06)],
|
||||
)
|
||||
def test_batch_cost_calculator_claude_haiku_5_5_prompt_length_tiers(
|
||||
_local_model_cost_map: None,
|
||||
prompt_tokens: int,
|
||||
input_rate: float,
|
||||
output_rate: float,
|
||||
) -> None:
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
completion_tokens: Final = 1_000
|
||||
|
||||
prompt_cost, completion_cost = batch_cost_calculator(
|
||||
usage=Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
),
|
||||
model="claude-haiku-5-5",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(prompt_tokens * input_rate)
|
||||
assert completion_cost == pytest.approx(completion_tokens * output_rate)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("prompt_tokens", "expected"),
|
||||
[
|
||||
(100_000, (5e-08, 2.5e-07, 5e-09, 6.25e-08)),
|
||||
(100_001, (2.5e-07, 1.25e-06, 2.5e-08, 3.125e-07)),
|
||||
],
|
||||
)
|
||||
def test_get_batch_cost_rates_claude_haiku_5_5_prompt_length_tiers(
|
||||
_local_model_cost_map: None,
|
||||
prompt_tokens: int,
|
||||
expected: tuple[float, float, float, float],
|
||||
) -> None:
|
||||
"""Cache write and cache read batch rates are 50% of the standard rates; Anthropic's batch table omits them."""
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
|
||||
|
||||
rates: Final = get_batch_cost_rates(
|
||||
litellm.get_model_info(model="claude-haiku-5-5", custom_llm_provider="anthropic"),
|
||||
Usage(prompt_tokens=prompt_tokens, completion_tokens=1, total_tokens=prompt_tokens + 1),
|
||||
"anthropic",
|
||||
)
|
||||
|
||||
assert (rates.input, rates.output, rates.cache_read, rates.cache_creation) == expected
|
||||
|
|
|
|||
128
tests/unit/test_claude_haiku_5_5_config.py
Normal file
128
tests/unit/test_claude_haiku_5_5_config.py
Normal file
|
|
@ -0,0 +1,128 @@
|
|||
"""
|
||||
Validate Claude Haiku 5.5 model configuration entries.
|
||||
|
||||
Haiku 5.5 ships with adaptive thinking on by default, but unlike Sonnet 5.5 /
|
||||
Opus 5.5 thinking can still be turned off (``thinking: disabled`` at high
|
||||
effort or below) and it accepts a forced ``tool_choice`` (``any`` or a named
|
||||
tool). Its cost-map rows therefore carry ``thinking_always_on: false`` and
|
||||
``supports_forced_tool_use: true``.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
from collections.abc import Iterator
|
||||
from typing import Final, cast
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.get_model_cost_map import GetModelCostMap
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
REPO_ROOT: Final = os.path.join(os.path.dirname(__file__), "../..")
|
||||
|
||||
GET_WEATHER_TOOL: Final = {
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get the weather",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"city": {"type": "string"}},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> Iterator[None]:
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
|
||||
litellm.get_model_info.cache_clear()
|
||||
yield
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
def _load_root_cost_map() -> dict[str, dict[str, object]]:
|
||||
json_path: Final = os.path.join(REPO_ROOT, "model_prices_and_context_window.json")
|
||||
with open(json_path) as f:
|
||||
return cast(dict[str, dict[str, object]], json.load(f))
|
||||
|
||||
|
||||
HAIKU_5_5_VARIANTS: Final = (
|
||||
"claude-haiku-5-5",
|
||||
"anthropic.claude-haiku-5-5",
|
||||
"apac.anthropic.claude-haiku-5-5",
|
||||
"au.anthropic.claude-haiku-5-5",
|
||||
"eu.anthropic.claude-haiku-5-5",
|
||||
"global.anthropic.claude-haiku-5-5",
|
||||
"jp.anthropic.claude-haiku-5-5",
|
||||
"us.anthropic.claude-haiku-5-5",
|
||||
"us-gov.anthropic.claude-haiku-5-5",
|
||||
"bedrock/us-gov-east-1/anthropic.claude-haiku-5-5",
|
||||
"bedrock/us-gov-west-1/anthropic.claude-haiku-5-5",
|
||||
"bedrock_mantle/anthropic.claude-haiku-5-5",
|
||||
"bedrock_mantle/us-gov-west-1/anthropic.claude-haiku-5-5",
|
||||
"vertex_ai/claude-haiku-5-5",
|
||||
"vertex_ai/claude-haiku-5-5@default",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model_name", HAIKU_5_5_VARIANTS)
|
||||
def test_haiku_5_5_rows_allow_disabling_thinking_and_forced_tools(
|
||||
model_name: str,
|
||||
) -> None:
|
||||
root: Final = _load_root_cost_map()
|
||||
backup: Final = GetModelCostMap.load_local_model_cost_map()
|
||||
assert model_name in root
|
||||
row: Final = root[model_name]
|
||||
# https://platform.claude.com/docs/en/models/haiku-5-5/whats-new-haiku-5-5 (2026-10-07):
|
||||
# thinking can be disabled, forced tool_choice accepted
|
||||
assert row["thinking_always_on"] is False
|
||||
assert row["supports_forced_tool_use"] is True
|
||||
assert backup[model_name] == row
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model", "provider"),
|
||||
[
|
||||
("claude-haiku-5-5", "anthropic"),
|
||||
("anthropic/claude-haiku-5-5", "anthropic"),
|
||||
("vertex_ai/claude-haiku-5-5", "vertex_ai"),
|
||||
],
|
||||
)
|
||||
def test_haiku_5_5_runtime_profile(local_model_cost_map: None, model: str, provider: str) -> None:
|
||||
assert AnthropicModelInfo.is_adaptive_thinking_model(model, provider) is True
|
||||
assert AnthropicModelInfo._is_always_on_thinking_model(model, provider) is False
|
||||
assert AnthropicModelInfo.forced_tool_use_unsupported(model.removeprefix("anthropic/")) is False
|
||||
|
||||
|
||||
def test_haiku_5_5_anthropic_tool_choice_required_maps_to_any(
|
||||
local_model_cost_map: None,
|
||||
) -> None:
|
||||
optional_params: Final = litellm.AnthropicConfig().map_openai_params(
|
||||
non_default_params={
|
||||
"tools": [dict(GET_WEATHER_TOOL)],
|
||||
"tool_choice": "required",
|
||||
},
|
||||
optional_params={},
|
||||
model="claude-haiku-5-5",
|
||||
drop_params=False,
|
||||
)
|
||||
assert optional_params["tool_choice"] == {"type": "any"}
|
||||
|
||||
|
||||
def test_haiku_5_5_bedrock_tool_choice_required_maps_to_any(
|
||||
local_model_cost_map: None,
|
||||
) -> None:
|
||||
optional_params: Final = litellm.AmazonConverseConfig().map_openai_params(
|
||||
non_default_params={
|
||||
"tools": [dict(GET_WEATHER_TOOL)],
|
||||
"tool_choice": "required",
|
||||
},
|
||||
optional_params={},
|
||||
model="us.anthropic.claude-haiku-5-5",
|
||||
drop_params=False,
|
||||
)
|
||||
assert optional_params["tool_choice"] == {"any": {}}
|
||||
|
|
@ -794,6 +794,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_creation_input_token_cost_above_1hr": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_32k_tokens": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_100k_tokens": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_100k_tokens_batches": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_128k_tokens": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_200k_tokens": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_256k_tokens": {"type": "number"},
|
||||
|
|
@ -810,6 +811,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_read_input_token_cost": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_32k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_100k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_100k_tokens_batches": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_128k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_200k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_200k_tokens_batches": {"type": "number"},
|
||||
|
|
@ -839,6 +841,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"input_cost_per_video_token": {"type": "number"},
|
||||
"input_cost_per_token_above_32k_tokens": {"type": "number"},
|
||||
"input_cost_per_token_above_100k_tokens": {"type": "number"},
|
||||
"input_cost_per_token_above_100k_tokens_batches": {"type": "number"},
|
||||
"input_cost_per_token_above_200k_tokens": {"type": "number"},
|
||||
"input_cost_per_token_above_200k_tokens_batches": {"type": "number"},
|
||||
"input_cost_per_token_above_256k_tokens": {"type": "number"},
|
||||
|
|
@ -949,6 +952,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"output_cost_per_token": {"type": "number"},
|
||||
"output_cost_per_token_above_32k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_100k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_100k_tokens_batches": {"type": "number"},
|
||||
"output_cost_per_token_above_128k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_200k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_200k_tokens_batches": {"type": "number"},
|
||||
|
|
|
|||
36
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
36
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -35560,8 +35560,14 @@ export interface components {
|
|||
cache_creation_input_audio_token_cost?: number | null;
|
||||
/** Cache Creation Input Token Cost */
|
||||
cache_creation_input_token_cost?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 100K Tokens */
|
||||
cache_creation_input_token_cost_above_100k_tokens?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 100K Tokens Batches */
|
||||
cache_creation_input_token_cost_above_100k_tokens_batches?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 1Hr */
|
||||
cache_creation_input_token_cost_above_1hr?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 1Hr Above 100K Tokens */
|
||||
cache_creation_input_token_cost_above_1hr_above_100k_tokens?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 200K Tokens */
|
||||
cache_creation_input_token_cost_above_200k_tokens?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 200K Tokens Batches */
|
||||
|
|
@ -35590,6 +35596,10 @@ export interface components {
|
|||
cache_read_input_image_token_cost?: number | null;
|
||||
/** Cache Read Input Token Cost */
|
||||
cache_read_input_token_cost?: number | null;
|
||||
/** Cache Read Input Token Cost Above 100K Tokens */
|
||||
cache_read_input_token_cost_above_100k_tokens?: number | null;
|
||||
/** Cache Read Input Token Cost Above 100K Tokens Batches */
|
||||
cache_read_input_token_cost_above_100k_tokens_batches?: number | null;
|
||||
/** Cache Read Input Token Cost Above 200K Tokens */
|
||||
cache_read_input_token_cost_above_200k_tokens?: number | null;
|
||||
/** Cache Read Input Token Cost Above 200K Tokens Batches */
|
||||
|
|
@ -35674,6 +35684,10 @@ export interface components {
|
|||
input_cost_per_second?: number | null;
|
||||
/** Input Cost Per Token */
|
||||
input_cost_per_token?: number | null;
|
||||
/** Input Cost Per Token Above 100K Tokens */
|
||||
input_cost_per_token_above_100k_tokens?: number | null;
|
||||
/** Input Cost Per Token Above 100K Tokens Batches */
|
||||
input_cost_per_token_above_100k_tokens_batches?: number | null;
|
||||
/** Input Cost Per Token Above 128K Tokens */
|
||||
input_cost_per_token_above_128k_tokens?: number | null;
|
||||
/** Input Cost Per Token Above 200K Tokens */
|
||||
|
|
@ -35811,6 +35825,10 @@ export interface components {
|
|||
output_cost_per_second_768p?: number | null;
|
||||
/** Output Cost Per Token */
|
||||
output_cost_per_token?: number | null;
|
||||
/** Output Cost Per Token Above 100K Tokens */
|
||||
output_cost_per_token_above_100k_tokens?: number | null;
|
||||
/** Output Cost Per Token Above 100K Tokens Batches */
|
||||
output_cost_per_token_above_100k_tokens_batches?: number | null;
|
||||
/** Output Cost Per Token Above 128K Tokens */
|
||||
output_cost_per_token_above_128k_tokens?: number | null;
|
||||
/** Output Cost Per Token Above 200K Tokens */
|
||||
|
|
@ -50548,8 +50566,14 @@ export interface components {
|
|||
cache_creation_input_audio_token_cost?: number | null;
|
||||
/** Cache Creation Input Token Cost */
|
||||
cache_creation_input_token_cost?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 100K Tokens */
|
||||
cache_creation_input_token_cost_above_100k_tokens?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 100K Tokens Batches */
|
||||
cache_creation_input_token_cost_above_100k_tokens_batches?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 1Hr */
|
||||
cache_creation_input_token_cost_above_1hr?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 1Hr Above 100K Tokens */
|
||||
cache_creation_input_token_cost_above_1hr_above_100k_tokens?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 200K Tokens */
|
||||
cache_creation_input_token_cost_above_200k_tokens?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 200K Tokens Batches */
|
||||
|
|
@ -50578,6 +50602,10 @@ export interface components {
|
|||
cache_read_input_image_token_cost?: number | null;
|
||||
/** Cache Read Input Token Cost */
|
||||
cache_read_input_token_cost?: number | null;
|
||||
/** Cache Read Input Token Cost Above 100K Tokens */
|
||||
cache_read_input_token_cost_above_100k_tokens?: number | null;
|
||||
/** Cache Read Input Token Cost Above 100K Tokens Batches */
|
||||
cache_read_input_token_cost_above_100k_tokens_batches?: number | null;
|
||||
/** Cache Read Input Token Cost Above 200K Tokens */
|
||||
cache_read_input_token_cost_above_200k_tokens?: number | null;
|
||||
/** Cache Read Input Token Cost Above 200K Tokens Batches */
|
||||
|
|
@ -50662,6 +50690,10 @@ export interface components {
|
|||
input_cost_per_second?: number | null;
|
||||
/** Input Cost Per Token */
|
||||
input_cost_per_token?: number | null;
|
||||
/** Input Cost Per Token Above 100K Tokens */
|
||||
input_cost_per_token_above_100k_tokens?: number | null;
|
||||
/** Input Cost Per Token Above 100K Tokens Batches */
|
||||
input_cost_per_token_above_100k_tokens_batches?: number | null;
|
||||
/** Input Cost Per Token Above 128K Tokens */
|
||||
input_cost_per_token_above_128k_tokens?: number | null;
|
||||
/** Input Cost Per Token Above 200K Tokens */
|
||||
|
|
@ -50799,6 +50831,10 @@ export interface components {
|
|||
output_cost_per_second_768p?: number | null;
|
||||
/** Output Cost Per Token */
|
||||
output_cost_per_token?: number | null;
|
||||
/** Output Cost Per Token Above 100K Tokens */
|
||||
output_cost_per_token_above_100k_tokens?: number | null;
|
||||
/** Output Cost Per Token Above 100K Tokens Batches */
|
||||
output_cost_per_token_above_100k_tokens_batches?: number | null;
|
||||
/** Output Cost Per Token Above 128K Tokens */
|
||||
output_cost_per_token_above_128k_tokens?: number | null;
|
||||
/** Output Cost Per Token Above 200K Tokens */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue