fix(model_catalog): add Azure priority pricing for gpt-4.1/5.1/5.2

This commit is contained in:
michelligabriele 2026-05-21 23:57:52 +02:00
parent 47c92669c2
commit f44b0d3b60
No known key found for this signature in database
3 changed files with 74 additions and 0 deletions

View file

@ -3371,8 +3371,10 @@
},
"azure/gpt-4.1": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
@ -3380,6 +3382,7 @@
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_batches": 4e-06,
"output_cost_per_token_priority": 1.4e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -3405,8 +3408,10 @@
"azure/gpt-4.1-2025-04-14": {
"deprecation_date": "2026-11-04",
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
@ -3414,6 +3419,7 @@
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_batches": 4e-06,
"output_cost_per_token_priority": 1.4e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -4510,13 +4516,16 @@
},
"azure/gpt-5.1": {
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_priority": 2.5e-07,
"input_cost_per_token": 1.25e-06,
"input_cost_per_token_priority": 2.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 2e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -4668,13 +4677,16 @@
},
"azure/gpt-5.2": {
"cache_read_input_token_cost": 1.75e-07,
"cache_read_input_token_cost_priority": 3.5e-07,
"input_cost_per_token": 1.75e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"output_cost_per_token_priority": 2.8e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",

View file

@ -3371,8 +3371,10 @@
},
"azure/gpt-4.1": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
@ -3380,6 +3382,7 @@
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_batches": 4e-06,
"output_cost_per_token_priority": 1.4e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -3405,8 +3408,10 @@
"azure/gpt-4.1-2025-04-14": {
"deprecation_date": "2026-11-04",
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_priority": 8.75e-07,
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 1047576,
"max_output_tokens": 32768,
@ -3414,6 +3419,7 @@
"mode": "chat",
"output_cost_per_token": 8e-06,
"output_cost_per_token_batches": 4e-06,
"output_cost_per_token_priority": 1.4e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -4510,13 +4516,16 @@
},
"azure/gpt-5.1": {
"cache_read_input_token_cost": 1.25e-07,
"cache_read_input_token_cost_priority": 2.5e-07,
"input_cost_per_token": 1.25e-06,
"input_cost_per_token_priority": 2.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1e-05,
"output_cost_per_token_priority": 2e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
@ -4668,13 +4677,16 @@
},
"azure/gpt-5.2": {
"cache_read_input_token_cost": 1.75e-07,
"cache_read_input_token_cost_priority": 3.5e-07,
"input_cost_per_token": 1.75e-06,
"input_cost_per_token_priority": 3.5e-06,
"litellm_provider": "azure",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.4e-05,
"output_cost_per_token_priority": 2.8e-05,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",

View file

@ -0,0 +1,50 @@
import json
from pathlib import Path
import pytest
# (model_name, input_priority, output_priority, cache_read_priority)
AZURE_PRIORITY_PRICING = [
("azure/gpt-4.1", 3.5e-06, 1.4e-05, 8.75e-07),
("azure/gpt-4.1-2025-04-14", 3.5e-06, 1.4e-05, 8.75e-07),
("azure/gpt-5.1", 2.5e-06, 2e-05, 2.5e-07),
("azure/gpt-5.2", 3.5e-06, 2.8e-05, 3.5e-07),
]
@pytest.mark.parametrize(
"model, input_priority, output_priority, cache_read_priority",
AZURE_PRIORITY_PRICING,
)
def test_azure_priority_pricing_keys_present(
model, input_priority, output_priority, cache_read_priority
):
json_path = Path(__file__).parents[2] / "model_prices_and_context_window.json"
with open(json_path) as f:
model_cost = json.load(f)
info = model_cost.get(model)
assert info is not None, f"{model} not found in model catalog"
assert info["litellm_provider"] == "azure"
assert info["input_cost_per_token_priority"] == input_priority
assert info["output_cost_per_token_priority"] == output_priority
assert info["cache_read_input_token_cost_priority"] == cache_read_priority
def test_azure_priority_pricing_backup_matches_main():
"""Ensure the bundled model cost map stays in sync with the canonical file."""
repo_root = Path(__file__).parents[2]
main_path = repo_root / "model_prices_and_context_window.json"
backup_path = repo_root / "litellm" / "model_prices_and_context_window_backup.json"
with open(main_path) as f:
main_cost = json.load(f)
with open(backup_path) as f:
backup_cost = json.load(f)
for model, *_ in AZURE_PRIORITY_PRICING:
assert backup_cost.get(model) == main_cost.get(
model
), f"{model} differs between main and backup model cost maps"