Move Azure Model Router flat cost to model_prices_and_context_window.json

- Added azure_ai/azure-model-router entry in model_prices_and_context_window.json with --.14/M tokens flat cost
- Updated cost_calculator.py to read flat cost from model info instead of hardcoded constant
- Updated tests to read pricing from model info
- Pricing is now centrally managed in the JSON configuration file

Co-authored-by: ishaan <ishaan@berri.ai>
This commit is contained in:
Cursor Agent 2026-01-30 18:19:15 +00:00
parent 9f2f8a4f65
commit d56f97495c
3 changed files with 29 additions and 17 deletions

View file

@ -8,11 +8,7 @@ from typing import Optional, Tuple
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
from litellm.types.utils import Usage
# Azure AI Foundry Model Router pricing
# Source: https://azure.microsoft.com/en-us/pricing/details/ai-services/
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = 0.14 # $0.14 per M input tokens
from litellm.utils import get_model_info
def _is_azure_model_router(model: str) -> bool:
@ -36,7 +32,7 @@ def cost_per_token(
Calculate the cost per token for Azure AI models.
For Azure AI Foundry Model Router:
- Adds a flat cost of $0.14 per million input tokens
- Adds a flat cost of $0.14 per million input tokens (from model_prices_and_context_window.json)
- Plus the cost of the actual model used (handled by generic_cost_per_token)
Args:
@ -55,17 +51,21 @@ def cost_per_token(
)
# Add flat cost for Azure Model Router
# The flat cost is defined in model_prices_and_context_window.json for azure_ai/azure-model-router
if _is_azure_model_router(model):
# Flat cost per million input tokens
flat_cost_per_token = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
router_flat_cost = usage.prompt_tokens * flat_cost_per_token
# Get the model router pricing from model_prices_and_context_window.json
model_info = get_model_info(model="azure-model-router", custom_llm_provider="azure_ai")
router_flat_cost_per_token = model_info.get("input_cost_per_token", 0)
verbose_logger.debug(
f"Azure AI Model Router flat cost: ${router_flat_cost:.6f} "
f"({usage.prompt_tokens} tokens × ${flat_cost_per_token:.9f}/token)"
)
# Add flat cost to prompt cost
prompt_cost += router_flat_cost
if router_flat_cost_per_token > 0:
router_flat_cost = usage.prompt_tokens * router_flat_cost_per_token
verbose_logger.debug(
f"Azure AI Model Router flat cost: ${router_flat_cost:.6f} "
f"({usage.prompt_tokens} tokens × ${router_flat_cost_per_token:.9f}/token)"
)
# Add flat cost to prompt cost
prompt_cost += router_flat_cost
return prompt_cost, completion_cost

View file

@ -1517,6 +1517,14 @@
"supports_response_schema": true,
"supports_tool_choice": true
},
"azure_ai/azure-model-router": {
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 0,
"litellm_provider": "azure_ai",
"mode": "chat",
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-services/",
"comment": "Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure"
},
"azure/eu/gpt-4o-2024-08-06": {
"deprecation_date": "2026-02-27",
"cache_read_input_token_cost": 1.375e-06,

View file

@ -4,11 +4,15 @@ Test Azure AI cost calculator, especially Model Router flat cost.
import pytest
from litellm.llms.azure_ai.cost_calculator import (
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS,
_is_azure_model_router,
cost_per_token,
)
from litellm.types.utils import Usage
from litellm.utils import get_model_info
# Get the flat cost from model_prices_and_context_window.json
_model_info = get_model_info(model="azure-model-router", custom_llm_provider="azure_ai")
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = _model_info.get("input_cost_per_token", 0) * 1_000_000
class TestAzureModelRouterDetection: