Add Azure AI Foundry Model Router flat cost of $0.14 per M input tokens

- Created litellm/llms/azure_ai/cost_calculator.py with flat cost logic
- Integrated Azure AI cost calculator into main cost_calculator.py
- Added comprehensive unit tests in tests/test_litellm/llms/azure_ai/test_cost_calculator.py
- Enhanced integration test in tests/llm_translation/test_azure_ai.py to verify flat cost
- Flat cost of $0.14 per million input tokens is added to all Azure Model Router requests

Co-authored-by: ishaan <ishaan@berri.ai>
This commit is contained in:
Cursor Agent 2026-01-30 03:32:53 +00:00
parent 8fcdf6105f
commit 9f2f8a4f65
4 changed files with 239 additions and 2 deletions

View file

@ -36,6 +36,9 @@ from litellm.llms.anthropic.cost_calculation import (
from litellm.llms.azure.cost_calculation import (
cost_per_token as azure_openai_cost_per_token,
)
from litellm.llms.azure_ai.cost_calculator import (
cost_per_token as azure_ai_cost_per_token,
)
from litellm.llms.base_llm.search.transformation import SearchResponse
from litellm.llms.bedrock.cost_calculation import (
cost_per_token as bedrock_cost_per_token,
@ -427,8 +430,8 @@ def cost_per_token( # noqa: PLR0915
return dashscope_cost_per_token(model=model, usage=usage_block)
elif custom_llm_provider == "azure_ai":
return generic_cost_per_token(
model=model, usage=usage_block, custom_llm_provider=custom_llm_provider
return azure_ai_cost_per_token(
model=model, usage=usage_block, response_time_ms=response_time_ms
)
else:
model_info = _cached_get_model_info_helper(

View file

@ -0,0 +1,71 @@
"""
Azure AI cost calculation helper.
Handles Azure AI Foundry Model Router flat cost and other Azure AI specific pricing.
"""
from typing import Optional, Tuple
from litellm._logging import verbose_logger
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
from litellm.types.utils import Usage
# Azure AI Foundry Model Router pricing
# Source: https://azure.microsoft.com/en-us/pricing/details/ai-services/
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = 0.14 # $0.14 per M input tokens
def _is_azure_model_router(model: str) -> bool:
"""
Check if the model is Azure AI Foundry Model Router.
Args:
model: The model name (e.g., "azure-model-router", "model-router")
Returns:
bool: True if this is a model router model
"""
model_lower = model.lower()
return "model-router" in model_lower or model_lower == "azure-model-router"
def cost_per_token(
model: str, usage: Usage, response_time_ms: Optional[float] = 0.0
) -> Tuple[float, float]:
"""
Calculate the cost per token for Azure AI models.
For Azure AI Foundry Model Router:
- Adds a flat cost of $0.14 per million input tokens
- Plus the cost of the actual model used (handled by generic_cost_per_token)
Args:
model: str, the model name without provider prefix
usage: LiteLLM Usage block
response_time_ms: Optional response time in milliseconds
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
"""
# Calculate base cost using generic cost calculator
prompt_cost, completion_cost = generic_cost_per_token(
model=model,
usage=usage,
custom_llm_provider="azure_ai",
)
# Add flat cost for Azure Model Router
if _is_azure_model_router(model):
# Flat cost per million input tokens
flat_cost_per_token = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
router_flat_cost = usage.prompt_tokens * flat_cost_per_token
verbose_logger.debug(
f"Azure AI Model Router flat cost: ${router_flat_cost:.6f} "
f"({usage.prompt_tokens} tokens × ${flat_cost_per_token:.9f}/token)"
)
# Add flat cost to prompt cost
prompt_cost += router_flat_cost
return prompt_cost, completion_cost

View file

@ -373,6 +373,7 @@ def test_completion_azure_ai_gpt_4o_with_flexible_api_base(api_base):
async def test_azure_ai_model_router():
"""
Test Azure AI model router non-streaming response cost tracking.
Verifies that the flat cost of $0.14 per M input tokens is applied.
"""
litellm._turn_on_debug()
response = await litellm.acompletion(
@ -387,6 +388,18 @@ async def test_azure_ai_model_router():
tracked_cost = response._hidden_params["response_cost"]
assert tracked_cost > 0
print("Tracked cost: ", tracked_cost)
# Verify flat cost is included
# Flat cost = prompt_tokens * $0.14 / 1M = prompt_tokens * 0.00000014
usage = response.usage
if usage and usage.prompt_tokens:
expected_flat_cost = usage.prompt_tokens * 0.14 / 1_000_000
print(f"Prompt tokens: {usage.prompt_tokens}")
print(f"Expected minimum flat cost: ${expected_flat_cost:.9f}")
# Total cost should be at least the flat cost
assert tracked_cost >= expected_flat_cost, (
f"Cost ${tracked_cost:.9f} should be >= flat cost ${expected_flat_cost:.9f}"
)
@pytest.mark.asyncio

View file

@ -0,0 +1,150 @@
"""
Test Azure AI cost calculator, especially Model Router flat cost.
"""
import pytest
from litellm.llms.azure_ai.cost_calculator import (
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS,
_is_azure_model_router,
cost_per_token,
)
from litellm.types.utils import Usage
class TestAzureModelRouterDetection:
"""Test that we correctly identify Azure Model Router models."""
@pytest.mark.parametrize(
"model,expected",
[
("azure-model-router", True),
("AZURE-MODEL-ROUTER", True),
("model-router", True),
("MODEL-ROUTER", True),
("gpt-4o-mini-2024-07-18-model-router", True),
("gpt-4o", False),
("gpt-4o-mini", False),
("claude-sonnet-4-5", False),
],
)
def test_is_azure_model_router(self, model: str, expected: bool):
"""Test Azure Model Router detection."""
assert _is_azure_model_router(model) == expected
class TestAzureModelRouterFlatCost:
"""Test Azure AI Foundry Model Router flat cost calculation."""
def test_model_router_flat_cost_basic(self):
"""Test that flat cost is added for Model Router requests."""
model = "azure-model-router"
usage = Usage(
prompt_tokens=1000,
completion_tokens=500,
total_tokens=1500,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Calculate expected flat cost
expected_flat_cost = (
usage.prompt_tokens * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
)
# Flat cost should be $0.00014 (1000 tokens × $0.14 / 1M tokens)
assert expected_flat_cost == pytest.approx(0.00014, rel=1e-9)
# Prompt cost should include the flat cost
# (plus any base cost from the actual model used, which might be 0 if not in model_cost)
assert prompt_cost >= expected_flat_cost
print(
f"Model Router flat cost for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
)
print(f"Total prompt cost: ${prompt_cost:.6f}")
def test_model_router_flat_cost_large_request(self):
"""Test flat cost calculation for larger requests."""
model = "model-router"
usage = Usage(
prompt_tokens=100_000,
completion_tokens=50_000,
total_tokens=150_000,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Calculate expected flat cost
expected_flat_cost = (
usage.prompt_tokens * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
)
# Flat cost should be $0.014 (100k tokens × $0.14 / 1M tokens)
assert expected_flat_cost == pytest.approx(0.014, rel=1e-9)
assert prompt_cost >= expected_flat_cost
print(
f"Model Router flat cost for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
)
print(f"Total prompt cost: ${prompt_cost:.6f}")
def test_model_router_flat_cost_1m_tokens(self):
"""Test flat cost for exactly 1 million input tokens."""
model = "azure-model-router"
usage = Usage(
prompt_tokens=1_000_000,
completion_tokens=100_000,
total_tokens=1_100_000,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Calculate expected flat cost
expected_flat_cost = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
# Flat cost should be exactly $0.14 for 1M tokens
assert expected_flat_cost == pytest.approx(0.14, rel=1e-9)
assert prompt_cost >= expected_flat_cost
print(f"Model Router flat cost for 1M tokens: ${expected_flat_cost:.6f}")
print(f"Total prompt cost: ${prompt_cost:.6f}")
def test_non_model_router_no_flat_cost(self):
"""Test that non-Model Router models don't get the flat cost."""
model = "gpt-4o"
usage = Usage(
prompt_tokens=1000,
completion_tokens=500,
total_tokens=1500,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# No flat cost should be added for non-Model Router models
# The cost might be 0 or based on the model's pricing
print(f"Non-Model Router prompt cost: ${prompt_cost:.6f}")
# We just ensure it doesn't crash and returns valid values
assert prompt_cost >= 0
assert completion_cost >= 0
def test_model_router_with_cached_tokens(self):
"""Test Model Router flat cost with cached tokens."""
model = "azure-model-router"
usage = Usage(
prompt_tokens=2000,
completion_tokens=800,
total_tokens=2800,
cache_read_input_tokens=500,
cache_creation_input_tokens=200,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Flat cost is based on ALL prompt tokens (including cached)
expected_flat_cost = (
usage.prompt_tokens * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
)
assert expected_flat_cost == pytest.approx(0.00028, rel=1e-9)
assert prompt_cost >= expected_flat_cost
print(
f"Model Router flat cost with caching for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
)
print(f"Total prompt cost: ${prompt_cost:.6f}")