This commit is contained in:
michelligabriele 2026-08-27 16:31:38 -04:00 • committed by GitHub
commit 72913c0041
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 99 additions and 0 deletions

View file

@ -4557,8 +4557,10 @@
"azure/gpt-4.1": {
"deprecation_date": "2027-04-14",
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
@ -4566,6 +4568,7 @@
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_batches": 4e-06,
"output_cost_per_token_priority": 1.4e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -4591,8 +4594,10 @@
"azure/gpt-4.1-2025-04-14": {
"deprecation_date": "2027-04-14",
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
@ -4600,6 +4605,7 @@
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_batches": 4e-06,
"output_cost_per_token_priority": 1.4e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -5803,13 +5809,16 @@
"azure/gpt-5.1": {
"deprecation_date": "2027-05-15",
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_priority": 2.5e-07,
"input_cost_per_token": 1.25e-06,
"input_cost_per_token_priority": 2.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 2e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -5832,6 +5841,7 @@
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_service_tier": true,
"supports_vision": true,
"supports_none_reasoning_effort": true
},
@ -5966,13 +5976,16 @@
"azure/gpt-5.2": {
"deprecation_date": "2027-06-08",
"cache_read_input_token_cost": 1.75e-07,
"cache_read_input_token_cost_priority": 3.5e-07,
"input_cost_per_token": 1.75e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"output_cost_per_token_priority": 2.8e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -5995,6 +6008,7 @@
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_service_tier": true,
"supports_vision": true
},
"azure/gpt-5.2-2025-12-11": {

View file

@ -4557,8 +4557,10 @@
"azure/gpt-4.1": {
"deprecation_date": "2027-04-14",
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
@ -4566,6 +4568,7 @@
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_batches": 4e-06,
"output_cost_per_token_priority": 1.4e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -4591,8 +4594,10 @@
"azure/gpt-4.1-2025-04-14": {
"deprecation_date": "2027-04-14",
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
@ -4600,6 +4605,7 @@
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_batches": 4e-06,
"output_cost_per_token_priority": 1.4e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -5803,13 +5809,16 @@
"azure/gpt-5.1": {
"deprecation_date": "2027-05-15",
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_priority": 2.5e-07,
"input_cost_per_token": 1.25e-06,
"input_cost_per_token_priority": 2.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 2e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -5832,6 +5841,7 @@
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_service_tier": true,
"supports_vision": true,
"supports_none_reasoning_effort": true
},
@ -5966,13 +5976,16 @@
"azure/gpt-5.2": {
"deprecation_date": "2027-06-08",
"cache_read_input_token_cost": 1.75e-07,
"cache_read_input_token_cost_priority": 3.5e-07,
"input_cost_per_token": 1.75e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"output_cost_per_token_priority": 2.8e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -5995,6 +6008,7 @@
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_service_tier": true,
"supports_vision": true
},
"azure/gpt-5.2-2025-12-11": {

View file

@ -0,0 +1,71 @@
import json
from pathlib import Path
import pytest
# (model_name, input_priority, output_priority, cache_read_priority)
AZURE_PRIORITY_PRICING = [
("azure/gpt-4.1", 3.5e-06, 1.4e-05, 8.75e-07),
("azure/gpt-4.1-2025-04-14", 3.5e-06, 1.4e-05, 8.75e-07),
("azure/gpt-5.1", 2.5e-06, 2e-05, 2.5e-07),
("azure/gpt-5.2", 3.5e-06, 2.8e-05, 3.5e-07),
]
@pytest.mark.parametrize(
"model, input_priority, output_priority, cache_read_priority",
AZURE_PRIORITY_PRICING,
)
def test_azure_priority_pricing_keys_present(
model, input_priority, output_priority, cache_read_priority
):
json_path = Path(__file__).parents[2] / "model_prices_and_context_window.json"
with open(json_path) as f:
model_cost = json.load(f)
info = model_cost.get(model)
assert info is not None, f"{model} not found in model catalog"
assert info["litellm_provider"] == "azure"
assert info["input_cost_per_token_priority"] == input_priority
assert info["output_cost_per_token_priority"] == output_priority
assert info["cache_read_input_token_cost_priority"] == cache_read_priority
# Aliases that should carry supports_service_tier to match their dated counterparts
# (azure/gpt-5.1-2025-11-13, azure/gpt-5.2-2025-12-11). PR #24924 covers the gpt-4.1 aliases.
AZURE_SUPPORTS_SERVICE_TIER_ALIASES = [
"azure/gpt-5.1",
"azure/gpt-5.2",
]
@pytest.mark.parametrize("model", AZURE_SUPPORTS_SERVICE_TIER_ALIASES)
def test_azure_undated_aliases_advertise_service_tier_support(model):
json_path = Path(__file__).parents[2] / "model_prices_and_context_window.json"
with open(json_path) as f:
model_cost = json.load(f)
info = model_cost.get(model)
assert info is not None, f"{model} not found in model catalog"
assert (
info.get("supports_service_tier") is True
), f"{model} should advertise supports_service_tier=true to match its dated variant"
def test_azure_priority_pricing_backup_matches_main():
"""Ensure the bundled model cost map stays in sync with the canonical file."""
repo_root = Path(__file__).parents[2]
main_path = repo_root / "model_prices_and_context_window.json"
backup_path = repo_root / "litellm" / "model_prices_and_context_window_backup.json"
with open(main_path) as f:
main_cost = json.load(f)
with open(backup_path) as f:
backup_cost = json.load(f)
for model, *_ in AZURE_PRIORITY_PRICING:
assert backup_cost.get(model) == main_cost.get(
model
), f"{model} differs between main and backup model cost maps"