mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(pricing): correct gpt-5.4-mini and gpt-5.4-nano token limits
gpt-5.4-mini and gpt-5.4-nano are 400K-context models (272K input, 128K output), but their cost map entries carried gpt-5.4's 1.05M window. The router's pre-call context window check therefore admitted prompts far past what the models accept, so oversized requests were dispatched to the provider and failed there instead of being caught locally or routed through context_window_fallbacks. The azure_ai entries also inherited gpt-5.4's above-272K tiered pricing. OpenAI applies that surcharge to the 1.05M-window models only, so those keys are removed. Limits per OpenAI's model reference and Azure AI Foundry's model table: gpt-5.4-mini and gpt-5.4-nano are 400,000 context / 272,000 input / 128,000 output
This commit is contained in:
parent
56d51bc32e
commit
e0946ccf0d
3 changed files with 101 additions and 68 deletions
|
|
@ -3454,22 +3454,16 @@
|
|||
},
|
||||
"azure_ai/gpt-5.4-mini": {
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.5e-07,
|
||||
"cache_read_input_token_cost_priority": 1.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 3e-07,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 1.5e-06,
|
||||
"input_cost_per_token_priority": 1.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 3e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 400000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4.5e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 6.75e-06,
|
||||
"output_cost_per_token_priority": 9e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 1.35e-05,
|
||||
"source": "https://ai.azure.com/catalog/models/gpt-5.4-mini",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
|
|
@ -3500,22 +3494,16 @@
|
|||
},
|
||||
"azure_ai/gpt-5.4-mini-2026-03-17": {
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.5e-07,
|
||||
"cache_read_input_token_cost_priority": 1.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 3e-07,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 1.5e-06,
|
||||
"input_cost_per_token_priority": 1.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 3e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 400000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4.5e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 6.75e-06,
|
||||
"output_cost_per_token_priority": 9e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 1.35e-05,
|
||||
"source": "https://ai.azure.com/catalog/models/gpt-5.4-mini",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
|
|
@ -3546,22 +3534,16 @@
|
|||
},
|
||||
"azure_ai/gpt-5.4-nano": {
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
|
||||
"cache_read_input_token_cost_priority": 4e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 4e-07,
|
||||
"input_cost_per_token_priority": 4e-07,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8e-07,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 400000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.25e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 1.875e-06,
|
||||
"output_cost_per_token_priority": 2.5e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.75e-06,
|
||||
"source": "https://ai.azure.com/catalog/models/gpt-5.4-nano",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
|
|
@ -3592,22 +3574,16 @@
|
|||
},
|
||||
"azure_ai/gpt-5.4-nano-2026-03-17": {
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
|
||||
"cache_read_input_token_cost_priority": 4e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 4e-07,
|
||||
"input_cost_per_token_priority": 4e-07,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8e-07,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 400000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.25e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 1.875e-06,
|
||||
"output_cost_per_token_priority": 2.5e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.75e-06,
|
||||
"source": "https://ai.azure.com/catalog/models/gpt-5.4-nano",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
|
|
@ -7201,7 +7177,7 @@
|
|||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
@ -7236,7 +7212,7 @@
|
|||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
@ -7271,7 +7247,7 @@
|
|||
"cache_read_input_token_cost": 2e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
@ -7306,7 +7282,7 @@
|
|||
"cache_read_input_token_cost": 2e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
|
|||
|
|
@ -3454,22 +3454,16 @@
|
|||
},
|
||||
"azure_ai/gpt-5.4-mini": {
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.5e-07,
|
||||
"cache_read_input_token_cost_priority": 1.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 3e-07,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 1.5e-06,
|
||||
"input_cost_per_token_priority": 1.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 3e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 400000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4.5e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 6.75e-06,
|
||||
"output_cost_per_token_priority": 9e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 1.35e-05,
|
||||
"source": "https://ai.azure.com/catalog/models/gpt-5.4-mini",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
|
|
@ -3500,22 +3494,16 @@
|
|||
},
|
||||
"azure_ai/gpt-5.4-mini-2026-03-17": {
|
||||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1.5e-07,
|
||||
"cache_read_input_token_cost_priority": 1.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 3e-07,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 1.5e-06,
|
||||
"input_cost_per_token_priority": 1.5e-06,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 3e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 400000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 4.5e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 6.75e-06,
|
||||
"output_cost_per_token_priority": 9e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 1.35e-05,
|
||||
"source": "https://ai.azure.com/catalog/models/gpt-5.4-mini",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
|
|
@ -3546,22 +3534,16 @@
|
|||
},
|
||||
"azure_ai/gpt-5.4-nano": {
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
|
||||
"cache_read_input_token_cost_priority": 4e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 4e-07,
|
||||
"input_cost_per_token_priority": 4e-07,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8e-07,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 400000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.25e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 1.875e-06,
|
||||
"output_cost_per_token_priority": 2.5e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.75e-06,
|
||||
"source": "https://ai.azure.com/catalog/models/gpt-5.4-nano",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
|
|
@ -3592,22 +3574,16 @@
|
|||
},
|
||||
"azure_ai/gpt-5.4-nano-2026-03-17": {
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
|
||||
"cache_read_input_token_cost_priority": 4e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 4e-07,
|
||||
"input_cost_per_token_priority": 4e-07,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 8e-07,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 400000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.25e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 1.875e-06,
|
||||
"output_cost_per_token_priority": 2.5e-06,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 3.75e-06,
|
||||
"source": "https://ai.azure.com/catalog/models/gpt-5.4-nano",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
|
|
@ -7201,7 +7177,7 @@
|
|||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
@ -7236,7 +7212,7 @@
|
|||
"cache_read_input_token_cost": 7.5e-08,
|
||||
"input_cost_per_token": 7.5e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
@ -7271,7 +7247,7 @@
|
|||
"cache_read_input_token_cost": 2e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
@ -7306,7 +7282,7 @@
|
|||
"cache_read_input_token_cost": 2e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
@ -24364,7 +24340,7 @@
|
|||
"input_cost_per_token_batches": 3.75e-07,
|
||||
"input_cost_per_token_priority": 1.5e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
@ -24410,7 +24386,7 @@
|
|||
"input_cost_per_token_batches": 3.75e-07,
|
||||
"input_cost_per_token_priority": 1.5e-06,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
@ -24454,7 +24430,7 @@
|
|||
"input_cost_per_token_flex": 1e-07,
|
||||
"input_cost_per_token_batches": 1e-07,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
@ -24497,7 +24473,7 @@
|
|||
"input_cost_per_token_flex": 1e-07,
|
||||
"input_cost_per_token_batches": 1e-07,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 1050000,
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
|
|
|
|||
81
tests/test_litellm/test_gpt_5_4_model_metadata.py
Normal file
81
tests/test_litellm/test_gpt_5_4_model_metadata.py
Normal file
|
|
@ -0,0 +1,81 @@
|
|||
import json
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[2]
|
||||
MAIN_PATH = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
BACKUP_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
|
||||
|
||||
DOCUMENTED_MAX_INPUT_TOKENS = 272000
|
||||
DOCUMENTED_MAX_OUTPUT_TOKENS = 128000
|
||||
|
||||
SMALL_MODEL_NAMES = (
|
||||
"gpt-5.4-mini",
|
||||
"gpt-5.4-mini-2026-03-17",
|
||||
"gpt-5.4-nano",
|
||||
"gpt-5.4-nano-2026-03-17",
|
||||
)
|
||||
SMALL_MODELS = tuple(f"{prefix}{name}" for prefix in ("", "azure/", "azure_ai/") for name in SMALL_MODEL_NAMES)
|
||||
|
||||
STANDARD_PRICING = {
|
||||
"gpt-5.4-mini": (7.5e-07, 4.5e-06, 7.5e-08),
|
||||
"gpt-5.4-nano": (2e-07, 1.25e-06, 2e-08),
|
||||
}
|
||||
|
||||
LONG_CONTEXT_MODELS = ("gpt-5.4", "gpt-5.4-pro")
|
||||
|
||||
|
||||
@lru_cache(maxsize=2)
|
||||
def _load(path: Path) -> dict[str, dict[str, object]]:
|
||||
with open(path) as f:
|
||||
return json.load(f)
|
||||
|
||||
|
||||
def _pricing_key(model: str) -> str:
|
||||
return "gpt-5.4-nano" if "nano" in model else "gpt-5.4-mini"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", SMALL_MODELS)
|
||||
def test_gpt_5_4_small_models_use_documented_token_limits(model: str) -> None:
|
||||
"""gpt-5.4-mini/nano are 400K-window models: 272K in, 128K out, not gpt-5.4's 1.05M window."""
|
||||
info = _load(MAIN_PATH).get(model)
|
||||
assert info is not None, f"{model} not found in model_prices_and_context_window.json"
|
||||
|
||||
assert info["max_input_tokens"] == DOCUMENTED_MAX_INPUT_TOKENS
|
||||
assert info["max_output_tokens"] == DOCUMENTED_MAX_OUTPUT_TOKENS
|
||||
assert info["max_tokens"] == DOCUMENTED_MAX_OUTPUT_TOKENS
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", SMALL_MODELS)
|
||||
def test_gpt_5_4_small_models_have_no_long_context_surcharge(model: str) -> None:
|
||||
"""OpenAI prices prompts above 272K at 2x input / 1.5x output for the 1.05M-window models only."""
|
||||
info = _load(MAIN_PATH)[model]
|
||||
assert [key for key in info if "above_272k" in key] == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", SMALL_MODELS)
|
||||
def test_gpt_5_4_small_models_standard_pricing(model: str) -> None:
|
||||
info = _load(MAIN_PATH)[model]
|
||||
input_cost, output_cost, cache_read_cost = STANDARD_PRICING[_pricing_key(model)]
|
||||
|
||||
assert info["input_cost_per_token"] == input_cost
|
||||
assert info["output_cost_per_token"] == output_cost
|
||||
assert info["cache_read_input_token_cost"] == cache_read_cost
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", LONG_CONTEXT_MODELS)
|
||||
def test_gpt_5_4_long_context_models_keep_surcharge(model: str) -> None:
|
||||
"""The mini/nano correction must leave gpt-5.4 and gpt-5.4-pro tiered pricing intact."""
|
||||
info = _load(MAIN_PATH)[model]
|
||||
|
||||
assert info["input_cost_per_token_above_272k_tokens"] == pytest.approx(info["input_cost_per_token"] * 2)
|
||||
assert info["output_cost_per_token_above_272k_tokens"] == pytest.approx(info["output_cost_per_token"] * 1.5)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", SMALL_MODELS)
|
||||
def test_gpt_5_4_small_models_backup_matches_main(model: str) -> None:
|
||||
assert _load(BACKUP_PATH).get(model) == _load(MAIN_PATH).get(model), (
|
||||
f"{model} differs between main and backup model cost maps"
|
||||
)
|
||||
Loading…
Add table
Reference in a new issue