mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Merge pull request #38370 from BerriAI/litellm_azure_gpt_5_6_cache_write_pricing
fix(pricing): add azure gpt-5.6 cache write rates and correct data zone priority
This commit is contained in:
commit
74b6149d18
5 changed files with 267 additions and 48 deletions
|
|
@ -6592,6 +6592,10 @@
|
|||
"supports_web_search": true
|
||||
},
|
||||
"azure/gpt-5.6": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.25e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.25e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.5e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_priority": 1e-06,
|
||||
|
|
@ -6642,6 +6646,10 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/gpt-5.6-sol": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.25e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.25e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.5e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_priority": 1e-06,
|
||||
|
|
@ -6693,6 +6701,10 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/gpt-5.6-terra": {
|
||||
"cache_creation_input_token_cost": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5e-06,
|
||||
"cache_creation_input_token_cost_priority": 5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1e-05,
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4e-07,
|
||||
"cache_read_input_token_cost_priority": 4e-07,
|
||||
|
|
@ -6744,6 +6756,10 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/gpt-5.6-luna": {
|
||||
"cache_creation_input_token_cost": 2.5e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5e-07,
|
||||
"cache_creation_input_token_cost_priority": 5e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1e-06,
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
|
||||
"cache_read_input_token_cost_priority": 4e-08,
|
||||
|
|
@ -6795,12 +6811,18 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/us/gpt-5.6": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.375e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.75e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.375e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_priority": 1.375e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 2.2e-06,
|
||||
"cache_read_input_token_cost_priority": 1.1e-06,
|
||||
"input_cost_per_token": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 1.1e-05,
|
||||
"input_cost_per_token_priority": 1.375e-05,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 2.2e-05,
|
||||
"input_cost_per_token_priority": 1.1e-05,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -6808,7 +6830,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 4.95e-05,
|
||||
"output_cost_per_token_priority": 8.25e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 9.9e-05,
|
||||
"output_cost_per_token_priority": 6.6e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -6842,13 +6865,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/us/gpt-5.6-sol": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.375e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.75e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.375e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_priority": 1.375e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 2.2e-06,
|
||||
"cache_read_input_token_cost_priority": 1.1e-06,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 1.1e-05,
|
||||
"input_cost_per_token_priority": 1.375e-05,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 2.2e-05,
|
||||
"input_cost_per_token_priority": 1.1e-05,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -6856,7 +6885,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 4.95e-05,
|
||||
"output_cost_per_token_priority": 8.25e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 9.9e-05,
|
||||
"output_cost_per_token_priority": 6.6e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -6890,13 +6920,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/us/gpt-5.6-terra": {
|
||||
"cache_creation_input_token_cost": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1.1e-05,
|
||||
"cache_creation_input_token_cost_priority": 5.5e-06,
|
||||
"cache_read_input_token_cost": 2.2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4.4e-07,
|
||||
"cache_read_input_token_cost_priority": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8.8e-07,
|
||||
"cache_read_input_token_cost_priority": 4.4e-07,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 2.2e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-06,
|
||||
"input_cost_per_token_priority": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8.8e-06,
|
||||
"input_cost_per_token_priority": 4.4e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -6904,7 +6940,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.32e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 1.98e-05,
|
||||
"output_cost_per_token_priority": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.96e-05,
|
||||
"output_cost_per_token_priority": 2.64e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -6938,13 +6975,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/us/gpt-5.6-luna": {
|
||||
"cache_creation_input_token_cost": 2.75e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1.1e-06,
|
||||
"cache_creation_input_token_cost_priority": 5.5e-07,
|
||||
"cache_read_input_token_cost": 2.2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4.4e-08,
|
||||
"cache_read_input_token_cost_priority": 5.5e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8.8e-08,
|
||||
"cache_read_input_token_cost_priority": 4.4e-08,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 2.2e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-07,
|
||||
"input_cost_per_token_priority": 5.5e-07,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8.8e-07,
|
||||
"input_cost_per_token_priority": 4.4e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -6952,7 +6995,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.32e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 1.98e-06,
|
||||
"output_cost_per_token_priority": 3.3e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.96e-06,
|
||||
"output_cost_per_token_priority": 2.64e-06,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -6986,12 +7030,18 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/eu/gpt-5.6": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.375e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.75e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.375e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_priority": 1.375e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 2.2e-06,
|
||||
"cache_read_input_token_cost_priority": 1.1e-06,
|
||||
"input_cost_per_token": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 1.1e-05,
|
||||
"input_cost_per_token_priority": 1.375e-05,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 2.2e-05,
|
||||
"input_cost_per_token_priority": 1.1e-05,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -6999,7 +7049,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 4.95e-05,
|
||||
"output_cost_per_token_priority": 8.25e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 9.9e-05,
|
||||
"output_cost_per_token_priority": 6.6e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -7033,13 +7084,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/eu/gpt-5.6-sol": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.375e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.75e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.375e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_priority": 1.375e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 2.2e-06,
|
||||
"cache_read_input_token_cost_priority": 1.1e-06,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 1.1e-05,
|
||||
"input_cost_per_token_priority": 1.375e-05,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 2.2e-05,
|
||||
"input_cost_per_token_priority": 1.1e-05,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -7047,7 +7104,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 4.95e-05,
|
||||
"output_cost_per_token_priority": 8.25e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 9.9e-05,
|
||||
"output_cost_per_token_priority": 6.6e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -7081,13 +7139,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/eu/gpt-5.6-terra": {
|
||||
"cache_creation_input_token_cost": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1.1e-05,
|
||||
"cache_creation_input_token_cost_priority": 5.5e-06,
|
||||
"cache_read_input_token_cost": 2.2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4.4e-07,
|
||||
"cache_read_input_token_cost_priority": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8.8e-07,
|
||||
"cache_read_input_token_cost_priority": 4.4e-07,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 2.2e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-06,
|
||||
"input_cost_per_token_priority": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8.8e-06,
|
||||
"input_cost_per_token_priority": 4.4e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -7095,7 +7159,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.32e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 1.98e-05,
|
||||
"output_cost_per_token_priority": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.96e-05,
|
||||
"output_cost_per_token_priority": 2.64e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -7129,13 +7194,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/eu/gpt-5.6-luna": {
|
||||
"cache_creation_input_token_cost": 2.75e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1.1e-06,
|
||||
"cache_creation_input_token_cost_priority": 5.5e-07,
|
||||
"cache_read_input_token_cost": 2.2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4.4e-08,
|
||||
"cache_read_input_token_cost_priority": 5.5e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8.8e-08,
|
||||
"cache_read_input_token_cost_priority": 4.4e-08,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 2.2e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-07,
|
||||
"input_cost_per_token_priority": 5.5e-07,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8.8e-07,
|
||||
"input_cost_per_token_priority": 4.4e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -7143,7 +7214,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.32e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 1.98e-06,
|
||||
"output_cost_per_token_priority": 3.3e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.96e-06,
|
||||
"output_cost_per_token_priority": 2.64e-06,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
|
|||
|
|
@ -6592,6 +6592,10 @@
|
|||
"supports_web_search": true
|
||||
},
|
||||
"azure/gpt-5.6": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.25e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.25e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.5e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_priority": 1e-06,
|
||||
|
|
@ -6642,6 +6646,10 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/gpt-5.6-sol": {
|
||||
"cache_creation_input_token_cost": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.25e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.25e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.5e-05,
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_priority": 1e-06,
|
||||
|
|
@ -6693,6 +6701,10 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/gpt-5.6-terra": {
|
||||
"cache_creation_input_token_cost": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5e-06,
|
||||
"cache_creation_input_token_cost_priority": 5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1e-05,
|
||||
"cache_read_input_token_cost": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4e-07,
|
||||
"cache_read_input_token_cost_priority": 4e-07,
|
||||
|
|
@ -6744,6 +6756,10 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/gpt-5.6-luna": {
|
||||
"cache_creation_input_token_cost": 2.5e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5e-07,
|
||||
"cache_creation_input_token_cost_priority": 5e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1e-06,
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
|
||||
"cache_read_input_token_cost_priority": 4e-08,
|
||||
|
|
@ -6795,12 +6811,18 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/us/gpt-5.6": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.375e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.75e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.375e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_priority": 1.375e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 2.2e-06,
|
||||
"cache_read_input_token_cost_priority": 1.1e-06,
|
||||
"input_cost_per_token": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 1.1e-05,
|
||||
"input_cost_per_token_priority": 1.375e-05,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 2.2e-05,
|
||||
"input_cost_per_token_priority": 1.1e-05,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -6808,7 +6830,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 4.95e-05,
|
||||
"output_cost_per_token_priority": 8.25e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 9.9e-05,
|
||||
"output_cost_per_token_priority": 6.6e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -6842,13 +6865,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/us/gpt-5.6-sol": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.375e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.75e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.375e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_priority": 1.375e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 2.2e-06,
|
||||
"cache_read_input_token_cost_priority": 1.1e-06,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 1.1e-05,
|
||||
"input_cost_per_token_priority": 1.375e-05,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 2.2e-05,
|
||||
"input_cost_per_token_priority": 1.1e-05,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -6856,7 +6885,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 4.95e-05,
|
||||
"output_cost_per_token_priority": 8.25e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 9.9e-05,
|
||||
"output_cost_per_token_priority": 6.6e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -6890,13 +6920,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/us/gpt-5.6-terra": {
|
||||
"cache_creation_input_token_cost": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1.1e-05,
|
||||
"cache_creation_input_token_cost_priority": 5.5e-06,
|
||||
"cache_read_input_token_cost": 2.2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4.4e-07,
|
||||
"cache_read_input_token_cost_priority": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8.8e-07,
|
||||
"cache_read_input_token_cost_priority": 4.4e-07,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 2.2e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-06,
|
||||
"input_cost_per_token_priority": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8.8e-06,
|
||||
"input_cost_per_token_priority": 4.4e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -6904,7 +6940,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.32e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 1.98e-05,
|
||||
"output_cost_per_token_priority": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.96e-05,
|
||||
"output_cost_per_token_priority": 2.64e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -6938,13 +6975,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/us/gpt-5.6-luna": {
|
||||
"cache_creation_input_token_cost": 2.75e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1.1e-06,
|
||||
"cache_creation_input_token_cost_priority": 5.5e-07,
|
||||
"cache_read_input_token_cost": 2.2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4.4e-08,
|
||||
"cache_read_input_token_cost_priority": 5.5e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8.8e-08,
|
||||
"cache_read_input_token_cost_priority": 4.4e-08,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 2.2e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-07,
|
||||
"input_cost_per_token_priority": 5.5e-07,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8.8e-07,
|
||||
"input_cost_per_token_priority": 4.4e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -6952,7 +6995,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.32e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 1.98e-06,
|
||||
"output_cost_per_token_priority": 3.3e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.96e-06,
|
||||
"output_cost_per_token_priority": 2.64e-06,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -6986,12 +7030,18 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/eu/gpt-5.6": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.375e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.75e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.375e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_priority": 1.375e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 2.2e-06,
|
||||
"cache_read_input_token_cost_priority": 1.1e-06,
|
||||
"input_cost_per_token": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 1.1e-05,
|
||||
"input_cost_per_token_priority": 1.375e-05,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 2.2e-05,
|
||||
"input_cost_per_token_priority": 1.1e-05,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -6999,7 +7049,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 4.95e-05,
|
||||
"output_cost_per_token_priority": 8.25e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 9.9e-05,
|
||||
"output_cost_per_token_priority": 6.6e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -7033,13 +7084,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/eu/gpt-5.6-sol": {
|
||||
"cache_creation_input_token_cost": 6.875e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 1.375e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 2.75e-05,
|
||||
"cache_creation_input_token_cost_priority": 1.375e-05,
|
||||
"cache_read_input_token_cost": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.1e-06,
|
||||
"cache_read_input_token_cost_priority": 1.375e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 2.2e-06,
|
||||
"cache_read_input_token_cost_priority": 1.1e-06,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 1.1e-05,
|
||||
"input_cost_per_token_priority": 1.375e-05,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 2.2e-05,
|
||||
"input_cost_per_token_priority": 1.1e-05,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -7047,7 +7104,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 4.95e-05,
|
||||
"output_cost_per_token_priority": 8.25e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 9.9e-05,
|
||||
"output_cost_per_token_priority": 6.6e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -7081,13 +7139,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/eu/gpt-5.6-terra": {
|
||||
"cache_creation_input_token_cost": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1.1e-05,
|
||||
"cache_creation_input_token_cost_priority": 5.5e-06,
|
||||
"cache_read_input_token_cost": 2.2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4.4e-07,
|
||||
"cache_read_input_token_cost_priority": 5.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8.8e-07,
|
||||
"cache_read_input_token_cost_priority": 4.4e-07,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 2.2e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-06,
|
||||
"input_cost_per_token_priority": 5.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8.8e-06,
|
||||
"input_cost_per_token_priority": 4.4e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -7095,7 +7159,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.32e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 1.98e-05,
|
||||
"output_cost_per_token_priority": 3.3e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.96e-05,
|
||||
"output_cost_per_token_priority": 2.64e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
@ -7129,13 +7194,19 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure/eu/gpt-5.6-luna": {
|
||||
"cache_creation_input_token_cost": 2.75e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1.1e-06,
|
||||
"cache_creation_input_token_cost_priority": 5.5e-07,
|
||||
"cache_read_input_token_cost": 2.2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4.4e-08,
|
||||
"cache_read_input_token_cost_priority": 5.5e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8.8e-08,
|
||||
"cache_read_input_token_cost_priority": 4.4e-08,
|
||||
"deprecation_date": "2028-01-11",
|
||||
"input_cost_per_token": 2.2e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-07,
|
||||
"input_cost_per_token_priority": 5.5e-07,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8.8e-07,
|
||||
"input_cost_per_token_priority": 4.4e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
|
|
@ -7143,7 +7214,8 @@
|
|||
"mode": "chat",
|
||||
"output_cost_per_token": 1.32e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 1.98e-06,
|
||||
"output_cost_per_token_priority": 3.3e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.96e-06,
|
||||
"output_cost_per_token_priority": 2.64e-06,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
|
|
|
|||
|
|
@ -104,6 +104,11 @@
|
|||
"minimum": 0,
|
||||
"description": "Flex service-tier rate for the same-named base field."
|
||||
},
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"description": "Priority service-tier rate for the same-named base field."
|
||||
},
|
||||
"cache_creation_input_token_cost_flex": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
|
|||
|
|
@ -1727,6 +1727,73 @@ def test_azure_ai_cache_cost_calculation(_local_model_cost_map):
|
|||
), f"Output cost mismatch: got {output_cost}, expected {expected_output_cost}"
|
||||
|
||||
|
||||
|
||||
AZURE_GPT_5_6_MAP_KEYS = (
|
||||
"azure/gpt-5.6",
|
||||
"azure/gpt-5.6-sol",
|
||||
"azure/gpt-5.6-terra",
|
||||
"azure/gpt-5.6-luna",
|
||||
"azure/us/gpt-5.6",
|
||||
"azure/us/gpt-5.6-sol",
|
||||
"azure/us/gpt-5.6-terra",
|
||||
"azure/us/gpt-5.6-luna",
|
||||
"azure/eu/gpt-5.6",
|
||||
"azure/eu/gpt-5.6-sol",
|
||||
"azure/eu/gpt-5.6-terra",
|
||||
"azure/eu/gpt-5.6-luna",
|
||||
)
|
||||
|
||||
|
||||
def test_azure_gpt_5_6_cache_write_tokens_are_billed(_local_model_cost_map):
|
||||
"""
|
||||
Azure bills gpt-5.6 prompt cache writes at 1.25x the input rate on every
|
||||
tier, but the azure entries carried no ``cache_creation_input_token_cost``,
|
||||
so cache-write tokens were billed at the plain input rate instead.
|
||||
"""
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
usage = Usage(
|
||||
completion_tokens=100,
|
||||
prompt_tokens=2000,
|
||||
total_tokens=2100,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=0, text_tokens=687),
|
||||
cache_creation_input_tokens=1313,
|
||||
)
|
||||
|
||||
input_cost, output_cost = generic_cost_per_token(
|
||||
model="azure/gpt-5.6-luna", usage=usage, custom_llm_provider="azure"
|
||||
)
|
||||
|
||||
assert input_cost == pytest.approx(687 * 2e-07 + 1313 * 2.5e-07)
|
||||
assert output_cost == pytest.approx(100 * 1.2e-06)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", AZURE_GPT_5_6_MAP_KEYS)
|
||||
def test_azure_gpt_5_6_rates_match_azure_price_page(_local_model_cost_map, model):
|
||||
"""
|
||||
Per the Azure OpenAI price page (rendered 2026-08-26): cache writes cost
|
||||
1.25x input on every gpt-5.6 tier, and Data Zone costs 1.1x Global for
|
||||
standard and priority alike (us/eu priority rates previously sat at 1.25x).
|
||||
"""
|
||||
entry = litellm.model_cost[model]
|
||||
input_keys = [key for key in entry if key.startswith("input_cost_per_token")]
|
||||
assert input_keys
|
||||
for key in input_keys:
|
||||
suffix = key[len("input_cost_per_token") :]
|
||||
assert entry["cache_creation_input_token_cost" + suffix] == pytest.approx(entry[key] * 1.25)
|
||||
|
||||
zone = model.split("/")[1]
|
||||
if zone in ("us", "eu"):
|
||||
global_entry = litellm.model_cost["azure/" + model.split("/", 2)[2]]
|
||||
prefixes = ("input_cost_per_token", "output_cost_per_token", "cache_read", "cache_creation")
|
||||
token_cost_keys = [key for key in entry if key.startswith(prefixes)]
|
||||
global_token_cost_keys = [key for key in global_entry if key.startswith(prefixes)]
|
||||
assert len(token_cost_keys) >= 9
|
||||
assert sorted(token_cost_keys) == sorted(global_token_cost_keys)
|
||||
for key in token_cost_keys:
|
||||
assert entry[key] == pytest.approx(global_entry[key] * 1.1), key
|
||||
|
||||
def test_vertex_regional_deployment_costs_uplift_over_global(monkeypatch):
|
||||
"""
|
||||
Regression for https://github.com/BerriAI/litellm/issues/34393: two Vertex
|
||||
|
|
|
|||
|
|
@ -845,6 +845,9 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_creation_input_token_cost_above_272k_tokens_flex": {
|
||||
"type": "number"
|
||||
},
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": {
|
||||
"type": "number"
|
||||
},
|
||||
"cache_creation_input_token_cost_flex": {"type": "number"},
|
||||
"cache_creation_input_token_cost_priority": {"type": "number"},
|
||||
"cache_read_input_token_cost": {"type": "number"},
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue