diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 7f5e41d45ce..e9df18662b0 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -4557,8 +4557,10 @@ "azure/gpt-4.1": { "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_priority": 8.75e-07, "input_cost_per_token": 2e-06, "input_cost_per_token_batches": 1e-06, + "input_cost_per_token_priority": 3.5e-06, "litellm_provider": "azure", "max_input_tokens": 1047576, "max_output_tokens": 32768, @@ -4566,6 +4568,7 @@ "mode": "chat", "output_cost_per_token": 8e-06, "output_cost_per_token_batches": 4e-06, + "output_cost_per_token_priority": 1.4e-05, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", @@ -4591,8 +4594,10 @@ "azure/gpt-4.1-2025-04-14": { "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_priority": 8.75e-07, "input_cost_per_token": 2e-06, "input_cost_per_token_batches": 1e-06, + "input_cost_per_token_priority": 3.5e-06, "litellm_provider": "azure", "max_input_tokens": 1047576, "max_output_tokens": 32768, @@ -4600,6 +4605,7 @@ "mode": "chat", "output_cost_per_token": 8e-06, "output_cost_per_token_batches": 4e-06, + "output_cost_per_token_priority": 1.4e-05, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", @@ -5803,13 +5809,16 @@ "azure/gpt-5.1": { "deprecation_date": "2027-05-15", "cache_read_input_token_cost": 1.25e-07, + "cache_read_input_token_cost_priority": 2.5e-07, "input_cost_per_token": 1.25e-06, + "input_cost_per_token_priority": 2.5e-06, "litellm_provider": "azure", "max_input_tokens": 272000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 1e-05, + "output_cost_per_token_priority": 2e-05, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", @@ -5832,6 +5841,7 @@ "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, + "supports_service_tier": true, "supports_vision": true, "supports_none_reasoning_effort": true }, @@ -5966,13 +5976,16 @@ "azure/gpt-5.2": { "deprecation_date": "2027-06-08", "cache_read_input_token_cost": 1.75e-07, + "cache_read_input_token_cost_priority": 3.5e-07, "input_cost_per_token": 1.75e-06, + "input_cost_per_token_priority": 3.5e-06, "litellm_provider": "azure", "max_input_tokens": 272000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 1.4e-05, + "output_cost_per_token_priority": 2.8e-05, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", @@ -5995,6 +6008,7 @@ "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, + "supports_service_tier": true, "supports_vision": true }, "azure/gpt-5.2-2025-12-11": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 7f5e41d45ce..e9df18662b0 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -4557,8 +4557,10 @@ "azure/gpt-4.1": { "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_priority": 8.75e-07, "input_cost_per_token": 2e-06, "input_cost_per_token_batches": 1e-06, + "input_cost_per_token_priority": 3.5e-06, "litellm_provider": "azure", "max_input_tokens": 1047576, "max_output_tokens": 32768, @@ -4566,6 +4568,7 @@ "mode": "chat", "output_cost_per_token": 8e-06, "output_cost_per_token_batches": 4e-06, + "output_cost_per_token_priority": 1.4e-05, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", @@ -4591,8 +4594,10 @@ "azure/gpt-4.1-2025-04-14": { "deprecation_date": "2027-04-14", "cache_read_input_token_cost": 5e-07, + "cache_read_input_token_cost_priority": 8.75e-07, "input_cost_per_token": 2e-06, "input_cost_per_token_batches": 1e-06, + "input_cost_per_token_priority": 3.5e-06, "litellm_provider": "azure", "max_input_tokens": 1047576, "max_output_tokens": 32768, @@ -4600,6 +4605,7 @@ "mode": "chat", "output_cost_per_token": 8e-06, "output_cost_per_token_batches": 4e-06, + "output_cost_per_token_priority": 1.4e-05, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", @@ -5803,13 +5809,16 @@ "azure/gpt-5.1": { "deprecation_date": "2027-05-15", "cache_read_input_token_cost": 1.25e-07, + "cache_read_input_token_cost_priority": 2.5e-07, "input_cost_per_token": 1.25e-06, + "input_cost_per_token_priority": 2.5e-06, "litellm_provider": "azure", "max_input_tokens": 272000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 1e-05, + "output_cost_per_token_priority": 2e-05, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", @@ -5832,6 +5841,7 @@ "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, + "supports_service_tier": true, "supports_vision": true, "supports_none_reasoning_effort": true }, @@ -5966,13 +5976,16 @@ "azure/gpt-5.2": { "deprecation_date": "2027-06-08", "cache_read_input_token_cost": 1.75e-07, + "cache_read_input_token_cost_priority": 3.5e-07, "input_cost_per_token": 1.75e-06, + "input_cost_per_token_priority": 3.5e-06, "litellm_provider": "azure", "max_input_tokens": 272000, "max_output_tokens": 128000, "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 1.4e-05, + "output_cost_per_token_priority": 2.8e-05, "supported_endpoints": [ "/v1/chat/completions", "/v1/batch", @@ -5995,6 +6008,7 @@ "supports_response_schema": true, "supports_system_messages": true, "supports_tool_choice": true, + "supports_service_tier": true, "supports_vision": true }, "azure/gpt-5.2-2025-12-11": { diff --git a/tests/test_litellm/test_azure_priority_pricing_metadata.py b/tests/test_litellm/test_azure_priority_pricing_metadata.py new file mode 100644 index 00000000000..8080038ed6c --- /dev/null +++ b/tests/test_litellm/test_azure_priority_pricing_metadata.py @@ -0,0 +1,71 @@ +import json +from pathlib import Path + +import pytest + + +# (model_name, input_priority, output_priority, cache_read_priority) +AZURE_PRIORITY_PRICING = [ + ("azure/gpt-4.1", 3.5e-06, 1.4e-05, 8.75e-07), + ("azure/gpt-4.1-2025-04-14", 3.5e-06, 1.4e-05, 8.75e-07), + ("azure/gpt-5.1", 2.5e-06, 2e-05, 2.5e-07), + ("azure/gpt-5.2", 3.5e-06, 2.8e-05, 3.5e-07), +] + + +@pytest.mark.parametrize( + "model, input_priority, output_priority, cache_read_priority", + AZURE_PRIORITY_PRICING, +) +def test_azure_priority_pricing_keys_present( + model, input_priority, output_priority, cache_read_priority +): + json_path = Path(__file__).parents[2] / "model_prices_and_context_window.json" + with open(json_path) as f: + model_cost = json.load(f) + + info = model_cost.get(model) + assert info is not None, f"{model} not found in model catalog" + assert info["litellm_provider"] == "azure" + + assert info["input_cost_per_token_priority"] == input_priority + assert info["output_cost_per_token_priority"] == output_priority + assert info["cache_read_input_token_cost_priority"] == cache_read_priority + + +# Aliases that should carry supports_service_tier to match their dated counterparts +# (azure/gpt-5.1-2025-11-13, azure/gpt-5.2-2025-12-11). PR #24924 covers the gpt-4.1 aliases. +AZURE_SUPPORTS_SERVICE_TIER_ALIASES = [ + "azure/gpt-5.1", + "azure/gpt-5.2", +] + + +@pytest.mark.parametrize("model", AZURE_SUPPORTS_SERVICE_TIER_ALIASES) +def test_azure_undated_aliases_advertise_service_tier_support(model): + json_path = Path(__file__).parents[2] / "model_prices_and_context_window.json" + with open(json_path) as f: + model_cost = json.load(f) + + info = model_cost.get(model) + assert info is not None, f"{model} not found in model catalog" + assert ( + info.get("supports_service_tier") is True + ), f"{model} should advertise supports_service_tier=true to match its dated variant" + + +def test_azure_priority_pricing_backup_matches_main(): + """Ensure the bundled model cost map stays in sync with the canonical file.""" + repo_root = Path(__file__).parents[2] + main_path = repo_root / "model_prices_and_context_window.json" + backup_path = repo_root / "litellm" / "model_prices_and_context_window_backup.json" + + with open(main_path) as f: + main_cost = json.load(f) + with open(backup_path) as f: + backup_cost = json.load(f) + + for model, *_ in AZURE_PRIORITY_PRICING: + assert backup_cost.get(model) == main_cost.get( + model + ), f"{model} differs between main and backup model cost maps"