fix(azure_ai): charge the Model Router fee once and correct catalog limits

The router fee was folded into azure_ai.cost_per_token and then added again
by the additional_costs hook, so every routed request paid it twice. The hook
now owns the fee, the entry named by the deployment supplies the price, and a
response priced as the router entry itself is not charged again

model-router, gpt-chat-latest and cohere-command-a carry the limits from the
Foundry models page, and model-router and grok-4-20-* carry their retirement
dates. The router tests now run at the completion_cost level with a Logging
object, which is the path the proxy takes, and fail at the merge base
This commit is contained in:
mateo-berri 2026-09-07 20:46:39 -07:00
parent 95402ccb71
commit 415bdbfd8f
6 changed files with 246 additions and 463 deletions

View file

@ -45,6 +45,9 @@ from litellm.llms.azure.cost_calculation import (
from litellm.llms.azure_ai.cost_calculator import ( from litellm.llms.azure_ai.cost_calculator import (
cost_per_token as azure_ai_cost_per_token, cost_per_token as azure_ai_cost_per_token,
) )
from litellm.llms.azure_ai.cost_calculator import (
is_router_fee_entry as azure_ai_is_router_fee_entry,
)
from litellm.llms.base_llm.search.transformation import SearchResponse from litellm.llms.base_llm.search.transformation import SearchResponse
from litellm.llms.bedrock.cost_calculation import ( from litellm.llms.bedrock.cost_calculation import (
cost_per_token as bedrock_cost_per_token, cost_per_token as bedrock_cost_per_token,
@ -338,8 +341,6 @@ def cost_per_token(
### VERTEX LOCATION ### ### VERTEX LOCATION ###
vertex_location: str | None = None, # for Vertex AI regional-endpoint uplift (e.g. "us-east5", "global") vertex_location: str | None = None, # for Vertex AI regional-endpoint uplift (e.g. "us-east5", "global")
response: Any | None = None, response: Any | None = None,
### REQUEST MODEL ###
request_model: str | None = None, # original request model for router detection
) -> tuple[float, float]: ) -> tuple[float, float]:
""" """
Calculates the cost per token for a given model, prompt tokens, and completion tokens. Calculates the cost per token for a given model, prompt tokens, and completion tokens.
@ -661,7 +662,6 @@ def cost_per_token(
model=model, model=model,
usage=usage_block, usage=usage_block,
response_time_ms=response_time_ms, response_time_ms=response_time_ms,
request_model=request_model,
service_tier=service_tier, service_tier=service_tier,
) )
else: else:
@ -1659,11 +1659,10 @@ def completion_cost(
data_residency=data_residency, data_residency=data_residency,
vertex_location=vertex_location, vertex_location=vertex_location,
response=completion_response, response=completion_response,
request_model=request_model_for_cost,
) )
# Get additional costs from provider (e.g., routing fees, infrastructure costs) # Get additional costs from provider (e.g., routing fees, infrastructure costs)
if custom_llm_provider == "azure_ai": if custom_llm_provider == "azure_ai" and not azure_ai_is_router_fee_entry(model):
model_for_additional_costs = request_model_for_cost model_for_additional_costs = request_model_for_cost
if completion_response is not None: if completion_response is not None:
hidden_params = getattr(completion_response, "_hidden_params", None) or {} hidden_params = getattr(completion_response, "_hidden_params", None) or {}

View file

@ -31,6 +31,18 @@ def _is_azure_model_router(model: str) -> bool:
return "model-router" in model_lower or "model_router" in model_lower or model_lower == "azure-model-router" return "model-router" in model_lower or "model_router" in model_lower or model_lower == "azure-model-router"
ROUTER_FEE_ENTRY_NAMES: Final = frozenset({"model-router", "model_router"})
def is_router_fee_entry(model: str) -> bool:
return model.lower().removeprefix("azure_ai/") in ROUTER_FEE_ENTRY_NAMES
def _router_fee_entry_name(model: str) -> str:
entry_name: Final = model.lower().removeprefix("azure_ai/")
return entry_name if entry_name in ROUTER_FEE_ENTRY_NAMES else "model_router"
def calculate_azure_model_router_flat_cost(model: str, prompt_tokens: int) -> float: def calculate_azure_model_router_flat_cost(model: str, prompt_tokens: int) -> float:
""" """
Calculate the flat cost for Azure AI Foundry Model Router. Calculate the flat cost for Azure AI Foundry Model Router.
@ -44,26 +56,39 @@ def calculate_azure_model_router_flat_cost(model: str, prompt_tokens: int) -> fl
""" """
if not _is_azure_model_router(model): if not _is_azure_model_router(model):
return 0.0 return 0.0
model_info: Final = get_model_info(model=_router_fee_entry_name(model), custom_llm_provider="azure_ai")
# Get the model router pricing from model_prices_and_context_window.json
# Use "model_router" as the key (without actual model name suffix)
model_info: Final = get_model_info(model="model_router", custom_llm_provider="azure_ai")
router_flat_cost_per_token: Final = model_info.get("input_cost_per_token", 0) router_flat_cost_per_token: Final = model_info.get("input_cost_per_token", 0)
if router_flat_cost_per_token and router_flat_cost_per_token > 0: if router_flat_cost_per_token and router_flat_cost_per_token > 0:
return prompt_tokens * router_flat_cost_per_token return prompt_tokens * router_flat_cost_per_token
return 0.0 return 0.0
ROUTER_FEE_ENTRY_NAMES: Final = frozenset({"model-router", "model_router"}) def cost_per_token(
model: str,
usage: Usage,
response_time_ms: float | None = 0.0,
service_tier: str | None = None,
) -> tuple[float, float]:
"""
Price the response model's own tokens for Azure AI.
The Azure AI Foundry Model Router fee is not part of this: completion_cost charges it once through
AzureModelRouterConfig.calculate_additional_costs as the "Azure Model Router Flat Cost" line of the cost
breakdown, and a response priced as the router entry itself already carries it. A router deployment name
that is missing from the cost map prices at zero here so that line item is the whole cost.
def _prices_router_fee_itself(model: str) -> bool: Args:
return model.lower().rsplit("/", 1)[-1] in ROUTER_FEE_ENTRY_NAMES model: str, the model name without provider prefix (from response)
usage: LiteLLM Usage block
response_time_ms: Optional response time in milliseconds
service_tier: Optional service tier the request was priced on
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
def _base_cost_per_token(model: str, usage: Usage, service_tier: str | None) -> tuple[float, float] | None: Raises:
ValueError: If a model that is not a Model Router name is missing from the cost map
"""
try: try:
return generic_cost_per_token( return generic_cost_per_token(
model=model, usage=usage, custom_llm_provider="azure_ai", service_tier=service_tier model=model, usage=usage, custom_llm_provider="azure_ai", service_tier=service_tier
@ -72,44 +97,6 @@ def _base_cost_per_token(model: str, usage: Usage, service_tier: str | None) ->
if not _is_azure_model_router(model): if not _is_azure_model_router(model):
raise raise
verbose_logger.debug( verbose_logger.debug(
"Azure AI Model Router: model '%s' not in cost map, calculating routing flat cost only. Error: %s", model, e "Azure AI Model Router: model '%s' not in cost map, only the routing fee applies. Error: %s", model, e
) )
return None return 0.0, 0.0
def cost_per_token(
model: str,
usage: Usage,
response_time_ms: float | None = 0.0,
request_model: str | None = None,
service_tier: str | None = None,
) -> tuple[float, float]:
"""
Calculate the cost per token for Azure AI models.
For Azure AI Foundry Model Router the routing fee (the azure_ai/model_router entry, $0.14 per
million input tokens) is added on top of the routed model's cost. When the response model is
the router entry itself, generic_cost_per_token has already charged that fee.
Args:
model: str, the model name without provider prefix (from response)
usage: LiteLLM Usage block
response_time_ms: Optional response time in milliseconds
request_model: Optional[str], the original request model name (to detect router usage)
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
Raises:
ValueError: If the model is not found in the cost map and cost cannot be calculated
(except for Model Router models where we return just the routing flat cost)
"""
is_router_request: Final = _is_azure_model_router(model) or (
request_model is not None and _is_azure_model_router(request_model)
)
base_cost: Final = _base_cost_per_token(model=model, usage=usage, service_tier=service_tier)
prompt_cost, completion_cost = base_cost if base_cost is not None else (0.0, 0.0)
if not is_router_request or (base_cost is not None and _prices_router_fee_itself(model)):
return prompt_cost, completion_cost
router_flat_cost: Final = calculate_azure_model_router_flat_cost(request_model or model, usage.prompt_tokens)
return prompt_cost + router_flat_cost, completion_cost

View file

@ -3586,7 +3586,7 @@
"deprecation_date": "2026-12-02", "deprecation_date": "2026-12-02",
"input_cost_per_token": 5e-06, "input_cost_per_token": 5e-06,
"litellm_provider": "azure_ai", "litellm_provider": "azure_ai",
"max_input_tokens": 200000, "max_input_tokens": 272000,
"max_output_tokens": 128000, "max_output_tokens": 128000,
"max_tokens": 128000, "max_tokens": 128000,
"mode": "chat", "mode": "chat",
@ -4068,10 +4068,11 @@
"comment": "Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)" "comment": "Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)"
}, },
"azure_ai/model-router": { "azure_ai/model-router": {
"deprecation_date": "2027-05-20",
"input_cost_per_token": 1.4e-07, "input_cost_per_token": 1.4e-07,
"output_cost_per_token": 0, "output_cost_per_token": 0,
"litellm_provider": "azure_ai", "litellm_provider": "azure_ai",
"max_input_tokens": 1048576, "max_input_tokens": 200000,
"max_output_tokens": 32768, "max_output_tokens": 32768,
"max_tokens": 32768, "max_tokens": 32768,
"mode": "chat", "mode": "chat",
@ -10393,8 +10394,8 @@
"input_cost_per_token": 2.5e-06, "input_cost_per_token": 2.5e-06,
"litellm_provider": "azure_ai", "litellm_provider": "azure_ai",
"max_input_tokens": 131072, "max_input_tokens": 131072,
"max_output_tokens": 4096, "max_output_tokens": 8182,
"max_tokens": 4096, "max_tokens": 8182,
"mode": "chat", "mode": "chat",
"output_cost_per_token": 1e-05, "output_cost_per_token": 1e-05,
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/cohere/", "source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/cohere/",
@ -10753,6 +10754,7 @@
"supports_web_search": true "supports_web_search": true
}, },
"azure_ai/grok-4-20-reasoning": { "azure_ai/grok-4-20-reasoning": {
"deprecation_date": "2027-04-06",
"input_cost_per_token": 1.25e-06, "input_cost_per_token": 1.25e-06,
"litellm_provider": "azure_ai", "litellm_provider": "azure_ai",
"max_input_tokens": 262000, "max_input_tokens": 262000,
@ -10769,6 +10771,7 @@
"supports_reasoning": true "supports_reasoning": true
}, },
"azure_ai/grok-4-20-non-reasoning": { "azure_ai/grok-4-20-non-reasoning": {
"deprecation_date": "2027-04-06",
"input_cost_per_token": 1.25e-06, "input_cost_per_token": 1.25e-06,
"litellm_provider": "azure_ai", "litellm_provider": "azure_ai",
"max_input_tokens": 262000, "max_input_tokens": 262000,

View file

@ -3586,7 +3586,7 @@
"deprecation_date": "2026-12-02", "deprecation_date": "2026-12-02",
"input_cost_per_token": 5e-06, "input_cost_per_token": 5e-06,
"litellm_provider": "azure_ai", "litellm_provider": "azure_ai",
"max_input_tokens": 200000, "max_input_tokens": 272000,
"max_output_tokens": 128000, "max_output_tokens": 128000,
"max_tokens": 128000, "max_tokens": 128000,
"mode": "chat", "mode": "chat",
@ -4068,10 +4068,11 @@
"comment": "Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)" "comment": "Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)"
}, },
"azure_ai/model-router": { "azure_ai/model-router": {
"deprecation_date": "2027-05-20",
"input_cost_per_token": 1.4e-07, "input_cost_per_token": 1.4e-07,
"output_cost_per_token": 0, "output_cost_per_token": 0,
"litellm_provider": "azure_ai", "litellm_provider": "azure_ai",
"max_input_tokens": 1048576, "max_input_tokens": 200000,
"max_output_tokens": 32768, "max_output_tokens": 32768,
"max_tokens": 32768, "max_tokens": 32768,
"mode": "chat", "mode": "chat",
@ -10393,8 +10394,8 @@
"input_cost_per_token": 2.5e-06, "input_cost_per_token": 2.5e-06,
"litellm_provider": "azure_ai", "litellm_provider": "azure_ai",
"max_input_tokens": 131072, "max_input_tokens": 131072,
"max_output_tokens": 4096, "max_output_tokens": 8182,
"max_tokens": 4096, "max_tokens": 8182,
"mode": "chat", "mode": "chat",
"output_cost_per_token": 1e-05, "output_cost_per_token": 1e-05,
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/cohere/", "source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/cohere/",
@ -10753,6 +10754,7 @@
"supports_web_search": true "supports_web_search": true
}, },
"azure_ai/grok-4-20-reasoning": { "azure_ai/grok-4-20-reasoning": {
"deprecation_date": "2027-04-06",
"input_cost_per_token": 1.25e-06, "input_cost_per_token": 1.25e-06,
"litellm_provider": "azure_ai", "litellm_provider": "azure_ai",
"max_input_tokens": 262000, "max_input_tokens": 262000,
@ -10769,6 +10771,7 @@
"supports_reasoning": true "supports_reasoning": true
}, },
"azure_ai/grok-4-20-non-reasoning": { "azure_ai/grok-4-20-non-reasoning": {
"deprecation_date": "2027-04-06",
"input_cost_per_token": 1.25e-06, "input_cost_per_token": 1.25e-06,
"litellm_provider": "azure_ai", "litellm_provider": "azure_ai",
"max_input_tokens": 262000, "max_input_tokens": 262000,

View file

@ -2,20 +2,25 @@
Test Azure AI cost calculator, especially Model Router flat cost. Test Azure AI cost calculator, especially Model Router flat cost.
""" """
from datetime import datetime
from typing import Final
import pytest import pytest
import litellm
from litellm.cost_calculator import completion_cost
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.llms.azure_ai.cost_calculator import ( from litellm.llms.azure_ai.cost_calculator import (
_is_azure_model_router, _is_azure_model_router,
calculate_azure_model_router_flat_cost,
cost_per_token, cost_per_token,
) )
from litellm.types.utils import Usage from litellm.types.utils import Choices, Message, ModelResponse, Usage
from litellm.utils import get_model_info from litellm.utils import get_model_info
# Get the flat cost from model_prices_and_context_window.json # Get the flat cost from model_prices_and_context_window.json
_model_info = get_model_info(model="model_router", custom_llm_provider="azure_ai") _model_info = get_model_info(model="model_router", custom_llm_provider="azure_ai")
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = ( AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = _model_info.get("input_cost_per_token", 0) * 1_000_000
_model_info.get("input_cost_per_token", 0) * 1_000_000
)
class TestAzureModelRouterDetection: class TestAzureModelRouterDetection:
@ -80,377 +85,172 @@ class TestAzureModelRouterPrefix:
assert result == expected assert result == expected
ROUTER_FEE_PER_TOKEN: Final = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
ROUTED_MODEL: Final = "gpt-4.1-nano-2025-04-14"
ROUTED_USAGE: Final = Usage(prompt_tokens=5000, completion_tokens=2000, total_tokens=7000)
ROUTED_FEE: Final = 5000 * ROUTER_FEE_PER_TOKEN
def _router_logging(request_model: str) -> Logging:
return Logging(
model=request_model,
messages=[{"role": "user", "content": "Hello"}],
stream=False,
call_type="completion",
start_time=datetime.now(),
litellm_call_id="test-123",
function_id="test-function",
)
def _azure_ai_response(response_model: str, litellm_model_name: str | None = None) -> ModelResponse:
response: Final = ModelResponse(
id="test-123",
choices=[Choices(finish_reason="stop", index=0, message=Message(role="assistant", content="Hello"))],
created=1234567890,
model=response_model,
object="chat.completion",
usage=ROUTED_USAGE,
)
response._hidden_params = (
{"custom_llm_provider": "azure_ai"}
if litellm_model_name is None
else {"custom_llm_provider": "azure_ai", "litellm_model_name": litellm_model_name}
)
return response
def _routed_model_cost() -> tuple[float, float]:
routed_info: Final = get_model_info(model=ROUTED_MODEL, custom_llm_provider="azure_ai")
return (
ROUTED_USAGE.prompt_tokens * (routed_info["input_cost_per_token"] or 0.0),
ROUTED_USAGE.completion_tokens * (routed_info["output_cost_per_token"] or 0.0),
)
@pytest.mark.usefixtures("local_model_cost_map")
class TestAzureModelRouterFlatCost: class TestAzureModelRouterFlatCost:
"""Test Azure AI Foundry Model Router flat cost calculation.""" """cost_per_token prices the response model only; the router fee is the cost breakdown's own line item."""
def test_model_router_flat_cost_basic(self): def test_unmapped_router_deployment_name_prices_at_zero(self) -> None:
"""Test that flat cost is added for Model Router requests.""" usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
model = "azure-model-router" assert cost_per_token(model="azure-model-router", usage=usage) == (0.0, 0.0)
usage = Usage(
prompt_tokens=1000, @pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
completion_tokens=500, def test_router_entry_prices_its_own_fee(self, router_entry_name: str) -> None:
total_tokens=1500, usage = Usage(prompt_tokens=1_000_000, completion_tokens=0, total_tokens=1_000_000)
prompt_cost, completion_cost_usd = cost_per_token(model=router_entry_name, usage=usage)
assert prompt_cost == pytest.approx(0.14, rel=1e-9)
assert completion_cost_usd == 0.0
def test_routed_model_is_priced_as_itself(self) -> None:
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
prompt_cost, completion_cost_usd = cost_per_token(model=ROUTED_MODEL, usage=ROUTED_USAGE)
assert routed_prompt_cost > 0
assert prompt_cost == pytest.approx(routed_prompt_cost, rel=1e-9)
assert completion_cost_usd == pytest.approx(routed_completion_cost, rel=1e-9)
def test_unmapped_model_that_is_not_a_router_name_raises(self) -> None:
usage = Usage(prompt_tokens=10, completion_tokens=10, total_tokens=20)
with pytest.raises(Exception, match="no-such-azure-ai-model"):
cost_per_token(model="no-such-azure-ai-model", usage=usage)
def test_flat_cost_helper(self) -> None:
assert calculate_azure_model_router_flat_cost(
model="azure-model-router", prompt_tokens=10_000
) == pytest.approx(0.0014, rel=1e-9)
assert calculate_azure_model_router_flat_cost(model="gpt-5-nano", prompt_tokens=10_000) == 0.0
def test_flat_cost_reads_the_fee_from_the_deployment_named_entry(self) -> None:
litellm.register_model(
{"azure_ai/model-router": {"input_cost_per_token": 2e-07, "litellm_provider": "azure_ai", "mode": "chat"}}
) )
litellm.get_model_info.cache_clear()
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage) assert calculate_azure_model_router_flat_cost(model="model-router", prompt_tokens=1_000_000) == pytest.approx(
0.2, rel=1e-9
# Calculate expected flat cost
expected_flat_cost = (
usage.prompt_tokens
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
/ 1_000_000
) )
assert calculate_azure_model_router_flat_cost(
# Flat cost should be $0.00014 (1000 tokens × $0.14 / 1M tokens) model="azure-model-router", prompt_tokens=1_000_000
assert expected_flat_cost == pytest.approx(0.00014, rel=1e-9) ) == pytest.approx(0.14, rel=1e-9)
# Prompt cost should include the flat cost
# (plus any base cost from the actual model used, which might be 0 if not in model_cost)
assert prompt_cost >= expected_flat_cost
print(
f"Model Router flat cost for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
)
print(f"Total prompt cost: ${prompt_cost:.6f}")
def test_model_router_flat_cost_large_request(self):
"""Test flat cost calculation for larger requests."""
model = "model-router"
usage = Usage(
prompt_tokens=100_000,
completion_tokens=50_000,
total_tokens=150_000,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Calculate expected flat cost
expected_flat_cost = (
usage.prompt_tokens
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
/ 1_000_000
)
# Flat cost should be $0.014 (100k tokens × $0.14 / 1M tokens)
assert expected_flat_cost == pytest.approx(0.014, rel=1e-9)
# Use approx for floating-point comparison
assert prompt_cost >= expected_flat_cost or prompt_cost == pytest.approx(
expected_flat_cost, rel=1e-9
)
print(
f"Model Router flat cost for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
)
print(f"Total prompt cost: ${prompt_cost:.6f}")
def test_model_router_flat_cost_1m_tokens(self):
"""Test flat cost for exactly 1 million input tokens."""
model = "azure-model-router"
usage = Usage(
prompt_tokens=1_000_000,
completion_tokens=100_000,
total_tokens=1_100_000,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Calculate expected flat cost
expected_flat_cost = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
# Flat cost should be exactly $0.14 for 1M tokens
assert expected_flat_cost == pytest.approx(0.14, rel=1e-9)
assert prompt_cost >= expected_flat_cost
print(f"Model Router flat cost for 1M tokens: ${expected_flat_cost:.6f}")
print(f"Total prompt cost: ${prompt_cost:.6f}")
def test_non_model_router_no_flat_cost(self):
"""Test that non-Model Router models don't get the flat cost."""
model = "gpt-4o"
usage = Usage(
prompt_tokens=1000,
completion_tokens=500,
total_tokens=1500,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# No flat cost should be added for non-Model Router models
# The cost might be 0 or based on the model's pricing
print(f"Non-Model Router prompt cost: ${prompt_cost:.6f}")
# We just ensure it doesn't crash and returns valid values
assert prompt_cost >= 0
assert completion_cost >= 0
def test_model_router_with_cached_tokens(self):
"""Test Model Router flat cost with cached tokens."""
model = "azure-model-router"
usage = Usage(
prompt_tokens=2000,
completion_tokens=800,
total_tokens=2800,
cache_read_input_tokens=500,
cache_creation_input_tokens=200,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Flat cost is based on ALL prompt tokens (including cached)
expected_flat_cost = (
usage.prompt_tokens
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
/ 1_000_000
)
assert expected_flat_cost == pytest.approx(0.00028, rel=1e-9)
assert prompt_cost >= expected_flat_cost
print(
f"Model Router flat cost with caching for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
)
print(f"Total prompt cost: ${prompt_cost:.6f}")
def test_router_flat_cost_when_response_has_actual_model(self):
"""
Test that router flat cost is added when request was via router but response
contains the actual model (e.g., gpt-5-nano).
This is the key fix: Azure returns the actual model in the response, but we
must still add the router flat cost because the request was made via model router.
"""
usage = Usage(
prompt_tokens=10000,
completion_tokens=5000,
total_tokens=15000,
)
# Response model is the actual model Azure used (not a router name)
response_model = "gpt-5-nano-2025-08-07"
# Request model is the router - user called azure_ai/model_router/model-router
request_model = "azure_ai/model_router/model-router"
prompt_cost, completion_cost = cost_per_token(
model=response_model,
usage=usage,
request_model=request_model,
)
# Expected: model cost (from gpt-5-nano) + router flat cost
expected_flat_cost = (
usage.prompt_tokens
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
/ 1_000_000
)
assert expected_flat_cost == pytest.approx(0.0014, rel=1e-9)
# Total cost should be model cost + flat cost
total_cost = prompt_cost + completion_cost
assert total_cost >= expected_flat_cost
# Prompt cost should include both model prompt cost and router flat cost
assert prompt_cost >= expected_flat_cost
@pytest.mark.usefixtures("local_model_cost_map")
class TestAzureModelRouterCostBreakdown: class TestAzureModelRouterCostBreakdown:
"""Test that Azure Model Router flat cost is tracked in cost breakdown.""" """completion_cost charges the router fee exactly once, as the cost breakdown's additional cost line."""
def test_flat_cost_calculation_helper(self): def test_unmapped_router_deployment_name_costs_only_the_fee(self) -> None:
"""Test that flat cost can be calculated using the helper function."""
from litellm.llms.azure_ai.cost_calculator import (
calculate_azure_model_router_flat_cost,
)
model = "azure-model-router"
prompt_tokens = 10000
# Calculate flat cost using helper function
flat_cost = calculate_azure_model_router_flat_cost(
model=model, prompt_tokens=prompt_tokens
)
# Expected flat cost
expected_flat_cost = (
prompt_tokens * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
)
assert flat_cost > 0
assert flat_cost == pytest.approx(expected_flat_cost, rel=1e-9)
print(f"Flat cost calculated: ${flat_cost:.6f}")
def test_flat_cost_integration_with_completion_cost(self):
"""Test that flat cost is properly integrated into completion_cost calculation."""
import litellm
from litellm.cost_calculator import completion_cost
from litellm.types.utils import Choices, Message, ModelResponse, Usage
# Create a mock response for azure_ai model router
response = ModelResponse(
id="test-123",
choices=[
Choices(
finish_reason="stop",
index=0,
message=Message(
role="assistant",
content="Test response",
),
)
],
created=1234567890,
model="azure-model-router",
object="chat.completion",
usage=Usage(
prompt_tokens=5000,
completion_tokens=2000,
total_tokens=7000,
),
)
# Set hidden params for provider
response._hidden_params = {"custom_llm_provider": "azure_ai"}
# Calculate cost
cost = completion_cost( cost = completion_cost(
completion_response=response, completion_response=_azure_ai_response("azure-model-router"),
model="azure-model-router", model="azure-model-router",
custom_llm_provider="azure_ai", custom_llm_provider="azure_ai",
) )
assert cost == pytest.approx(ROUTED_FEE, rel=1e-9)
# Expected flat cost def test_fee_is_the_breakdown_line_item_for_an_unmapped_router_name(self) -> None:
expected_flat_cost = ( logging_obj = _router_logging("azure-model-router")
5000 * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
)
# Cost should include the flat cost (use approx for floating-point comparison)
assert cost >= expected_flat_cost or cost == pytest.approx(
expected_flat_cost, rel=1e-9
)
print(f"Total cost with flat fee: ${cost:.6f}")
print(f"Expected minimum flat cost: ${expected_flat_cost:.6f}")
def test_additional_costs_in_cost_breakdown(self):
"""Test that Azure Model Router flat cost appears in additional_costs dict."""
from datetime import datetime
from litellm.cost_calculator import completion_cost
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.types.utils import Choices, Message, ModelResponse, Usage
# Create logging object with required parameters
logging_obj = Logging(
model="azure-model-router",
messages=[{"role": "user", "content": "Hello"}],
stream=False,
call_type="completion",
start_time=datetime.now(),
litellm_call_id="test-123",
function_id="test-function",
)
# Create a mock response for azure_ai model router
response = ModelResponse(
id="test-123",
choices=[
Choices(
finish_reason="stop",
index=0,
message=Message(
role="assistant",
content="Test response",
),
)
],
created=1234567890,
model="azure-model-router",
object="chat.completion",
usage=Usage(
prompt_tokens=5000,
completion_tokens=2000,
total_tokens=7000,
),
)
# Set hidden params for provider
response._hidden_params = {"custom_llm_provider": "azure_ai"}
# Calculate cost with logging object
cost = completion_cost( cost = completion_cost(
completion_response=response, completion_response=_azure_ai_response("azure-model-router"),
model="azure-model-router", model="azure-model-router",
custom_llm_provider="azure_ai", custom_llm_provider="azure_ai",
litellm_logging_obj=logging_obj, litellm_logging_obj=logging_obj,
) )
breakdown = logging_obj.cost_breakdown
# Check that cost breakdown contains additional_costs assert breakdown is not None
assert hasattr(logging_obj, "cost_breakdown") assert breakdown["input_cost"] == 0.0
assert logging_obj.cost_breakdown is not None assert breakdown.get("additional_costs") == pytest.approx(
assert "additional_costs" in logging_obj.cost_breakdown {"Azure Model Router Flat Cost": ROUTED_FEE}, rel=1e-9
assert isinstance(logging_obj.cost_breakdown["additional_costs"], dict)
# Check that the Azure Model Router flat cost is in additional_costs
additional_costs = logging_obj.cost_breakdown["additional_costs"]
assert "Azure Model Router Flat Cost" in additional_costs
# Verify the flat cost value
expected_flat_cost = (
5000 * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
) )
actual_flat_cost = additional_costs["Azure Model Router Flat Cost"] assert cost == pytest.approx(ROUTED_FEE, rel=1e-9)
assert actual_flat_cost == pytest.approx(expected_flat_cost, rel=1e-9)
print(f"Additional costs in breakdown: {additional_costs}") def test_router_request_with_routed_response_charges_the_fee_once(self) -> None:
print(f"Azure Model Router Flat Cost: ${actual_flat_cost:.6f}") routed_prompt_cost, routed_completion_cost = _routed_model_cost()
logging_obj = _router_logging("model-router")
def test_additional_costs_when_response_has_actual_model_via_hidden_params(self):
"""additional_costs populated when response has actual model but request was via model router (hidden_params)."""
from datetime import datetime
from litellm.cost_calculator import completion_cost
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.types.utils import Choices, Message, ModelResponse, Usage
logging_obj = Logging(
model="gpt-4.1-nano-2025-04-14",
messages=[{"role": "user", "content": "Hello"}],
stream=False,
call_type="completion",
start_time=datetime.now(),
litellm_call_id="test-123",
function_id="test-function",
)
response = ModelResponse(
id="test-123",
choices=[
Choices(
finish_reason="stop",
index=0,
message=Message(role="assistant", content="Hello"),
)
],
created=1234567890,
model="gpt-4.1-nano-2025-04-14",
object="chat.completion",
usage=Usage(prompt_tokens=5000, completion_tokens=2000, total_tokens=7000),
)
response._hidden_params = {
"custom_llm_provider": "azure_ai",
"litellm_model_name": "azure_ai/model-router",
}
cost = completion_cost( cost = completion_cost(
completion_response=response, completion_response=_azure_ai_response(ROUTED_MODEL),
model="gpt-4.1-nano-2025-04-14", model=ROUTED_MODEL,
custom_llm_provider="azure_ai", custom_llm_provider="azure_ai",
litellm_logging_obj=logging_obj, litellm_logging_obj=logging_obj,
) )
expected_flat_cost = ( breakdown = logging_obj.cost_breakdown
5000 * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000 assert breakdown is not None
assert breakdown["input_cost"] == pytest.approx(routed_prompt_cost, rel=1e-9)
assert breakdown["output_cost"] == pytest.approx(routed_completion_cost, rel=1e-9)
assert breakdown.get("additional_costs") == pytest.approx(
{"Azure Model Router Flat Cost": ROUTED_FEE}, rel=1e-9
) )
assert cost >= expected_flat_cost assert cost == pytest.approx(routed_prompt_cost + routed_completion_cost + ROUTED_FEE, rel=1e-9)
assert logging_obj.cost_breakdown is not None
assert "additional_costs" in logging_obj.cost_breakdown def test_routed_response_named_by_hidden_params_charges_the_fee_once(self) -> None:
assert ( routed_prompt_cost, routed_completion_cost = _routed_model_cost()
"Azure Model Router Flat Cost" logging_obj = _router_logging(ROUTED_MODEL)
in logging_obj.cost_breakdown["additional_costs"] cost = completion_cost(
completion_response=_azure_ai_response(ROUTED_MODEL, litellm_model_name="azure_ai/model-router"),
model=ROUTED_MODEL,
custom_llm_provider="azure_ai",
litellm_logging_obj=logging_obj,
) )
assert logging_obj.cost_breakdown["additional_costs"][ breakdown = logging_obj.cost_breakdown
"Azure Model Router Flat Cost" assert breakdown is not None
] == pytest.approx(expected_flat_cost, rel=1e-9) assert breakdown["input_cost"] == pytest.approx(routed_prompt_cost, rel=1e-9)
assert breakdown.get("additional_costs") == pytest.approx(
{"Azure Model Router Flat Cost": ROUTED_FEE}, rel=1e-9
)
assert cost == pytest.approx(routed_prompt_cost + routed_completion_cost + ROUTED_FEE, rel=1e-9)
@pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
def test_response_priced_as_the_router_entry_charges_the_fee_once(self, router_entry_name: str) -> None:
logging_obj = _router_logging(router_entry_name)
cost = completion_cost(
completion_response=_azure_ai_response(router_entry_name),
model=router_entry_name,
custom_llm_provider="azure_ai",
litellm_logging_obj=logging_obj,
)
breakdown = logging_obj.cost_breakdown
assert breakdown is not None
assert "additional_costs" not in breakdown
assert breakdown["input_cost"] == pytest.approx(ROUTED_FEE, rel=1e-9)
assert cost == pytest.approx(ROUTED_FEE, rel=1e-9)
class TestAzureAIServiceTierCostCalculation: class TestAzureAIServiceTierCostCalculation:
@ -459,26 +259,27 @@ class TestAzureAIServiceTierCostCalculation:
@pytest.fixture(autouse=True) @pytest.fixture(autouse=True)
def register_test_model(self): def register_test_model(self):
import litellm import litellm
litellm.register_model(model_cost={
"test-azure-ai-model": { litellm.register_model(
"input_cost_per_token": 0.001, model_cost={
"output_cost_per_token": 0.002, "test-azure-ai-model": {
"input_cost_per_token_priority": 0.01, "input_cost_per_token": 0.001,
"output_cost_per_token_priority": 0.02, "output_cost_per_token": 0.002,
"input_cost_per_token_flex": 0.0005, "input_cost_per_token_priority": 0.01,
"output_cost_per_token_flex": 0.001, "output_cost_per_token_priority": 0.02,
"litellm_provider": "azure_ai", "input_cost_per_token_flex": 0.0005,
"max_tokens": 8192, "output_cost_per_token_flex": 0.001,
"litellm_provider": "azure_ai",
"max_tokens": 8192,
}
} }
}) )
def test_service_tier_priority_higher_cost(self): def test_service_tier_priority_higher_cost(self):
"""Priority tier should cost more than standard for azure_ai.""" """Priority tier should cost more than standard for azure_ai."""
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500) usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
standard_prompt, standard_completion = cost_per_token( standard_prompt, standard_completion = cost_per_token(model="test-azure-ai-model", usage=usage)
model="test-azure-ai-model", usage=usage
)
priority_prompt, priority_completion = cost_per_token( priority_prompt, priority_completion = cost_per_token(
model="test-azure-ai-model", usage=usage, service_tier="priority" model="test-azure-ai-model", usage=usage, service_tier="priority"
) )
@ -490,12 +291,8 @@ class TestAzureAIServiceTierCostCalculation:
"""Flex tier should cost less than standard for azure_ai.""" """Flex tier should cost less than standard for azure_ai."""
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500) usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
standard_prompt, standard_completion = cost_per_token( standard_prompt, standard_completion = cost_per_token(model="test-azure-ai-model", usage=usage)
model="test-azure-ai-model", usage=usage flex_prompt, flex_completion = cost_per_token(model="test-azure-ai-model", usage=usage, service_tier="flex")
)
flex_prompt, flex_completion = cost_per_token(
model="test-azure-ai-model", usage=usage, service_tier="flex"
)
assert flex_prompt < standard_prompt assert flex_prompt < standard_prompt
assert flex_completion < standard_completion assert flex_completion < standard_completion
@ -528,29 +325,3 @@ def test_mai_thinking_1_model_info_and_cost(local_model_cost_map):
assert model_info["supports_function_calling"] is True assert model_info["supports_function_calling"] is True
assert prompt_cost == pytest.approx(2.0) assert prompt_cost == pytest.approx(2.0)
assert completion_cost == pytest.approx(8.0) assert completion_cost == pytest.approx(8.0)
@pytest.mark.usefixtures("local_model_cost_map")
@pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
def test_router_entry_as_response_model_charges_the_fee_once(router_entry_name: str) -> None:
usage = Usage(prompt_tokens=1_000_000, completion_tokens=0, total_tokens=1_000_000)
prompt_cost, completion_cost = cost_per_token(model=router_entry_name, usage=usage)
assert prompt_cost == pytest.approx(0.14, rel=1e-9)
assert completion_cost == 0.0
@pytest.mark.usefixtures("local_model_cost_map")
def test_unmapped_router_deployment_name_still_charges_the_fee() -> None:
usage = Usage(prompt_tokens=1_000_000, completion_tokens=0, total_tokens=1_000_000)
prompt_cost, completion_cost = cost_per_token(model="azure-model-router", usage=usage)
assert prompt_cost == pytest.approx(0.14, rel=1e-9)
assert completion_cost == 0.0
@pytest.mark.usefixtures("local_model_cost_map")
def test_routed_model_response_adds_the_fee_on_top() -> None:
usage = Usage(prompt_tokens=1_000_000, completion_tokens=0, total_tokens=1_000_000)
routed_prompt_cost, _ = cost_per_token(model="gpt-5-nano", usage=usage)
prompt_cost, _ = cost_per_token(model="gpt-5-nano", usage=usage, request_model="azure_ai/model-router")
assert routed_prompt_cost > 0
assert prompt_cost == pytest.approx(routed_prompt_cost + 0.14, rel=1e-9)

View file

@ -26,6 +26,7 @@ class TokenPricedCatalogModel:
max_input_tokens: int max_input_tokens: int
max_output_tokens: int max_output_tokens: int
cache_read_input_token_cost: float | None cache_read_input_token_cost: float | None
deprecation_date: str | None
supported_flags: tuple[str, ...] supported_flags: tuple[str, ...]
@ -36,9 +37,10 @@ TOKEN_PRICED_MODELS: Final = (
source=AZURE_OPENAI_PRICING, source=AZURE_OPENAI_PRICING,
input_cost_per_token=5e-06, input_cost_per_token=5e-06,
output_cost_per_token=3e-05, output_cost_per_token=3e-05,
max_input_tokens=200000, max_input_tokens=272000,
max_output_tokens=128000, max_output_tokens=128000,
cache_read_input_token_cost=5e-07, cache_read_input_token_cost=5e-07,
deprecation_date="2026-12-02",
supported_flags=( supported_flags=(
"supports_function_calling", "supports_function_calling",
"supports_prompt_caching", "supports_prompt_caching",
@ -58,7 +60,13 @@ TOKEN_PRICED_MODELS: Final = (
max_input_tokens=200000, max_input_tokens=200000,
max_output_tokens=100000, max_output_tokens=100000,
cache_read_input_token_cost=3.75e-07, cache_read_input_token_cost=3.75e-07,
supported_flags=("supports_function_calling", "supports_prompt_caching", "supports_reasoning", "supports_vision"), deprecation_date="2026-11-15",
supported_flags=(
"supports_function_calling",
"supports_prompt_caching",
"supports_reasoning",
"supports_vision",
),
), ),
TokenPricedCatalogModel( TokenPricedCatalogModel(
catalog_name="model-router", catalog_name="model-router",
@ -66,9 +74,10 @@ TOKEN_PRICED_MODELS: Final = (
source=FOUNDRY_AOAI_PRICING, source=FOUNDRY_AOAI_PRICING,
input_cost_per_token=1.4e-07, input_cost_per_token=1.4e-07,
output_cost_per_token=0.0, output_cost_per_token=0.0,
max_input_tokens=1048576, max_input_tokens=200000,
max_output_tokens=32768, max_output_tokens=32768,
cache_read_input_token_cost=None, cache_read_input_token_cost=None,
deprecation_date="2027-05-20",
supported_flags=(), supported_flags=(),
), ),
TokenPricedCatalogModel( TokenPricedCatalogModel(
@ -78,8 +87,9 @@ TOKEN_PRICED_MODELS: Final = (
input_cost_per_token=2.5e-06, input_cost_per_token=2.5e-06,
output_cost_per_token=1e-05, output_cost_per_token=1e-05,
max_input_tokens=131072, max_input_tokens=131072,
max_output_tokens=4096, max_output_tokens=8182,
cache_read_input_token_cost=None, cache_read_input_token_cost=None,
deprecation_date=None,
supported_flags=("supports_function_calling", "supports_tool_choice"), supported_flags=("supports_function_calling", "supports_tool_choice"),
), ),
TokenPricedCatalogModel( TokenPricedCatalogModel(
@ -91,6 +101,7 @@ TOKEN_PRICED_MODELS: Final = (
max_input_tokens=262000, max_input_tokens=262000,
max_output_tokens=8192, max_output_tokens=8192,
cache_read_input_token_cost=None, cache_read_input_token_cost=None,
deprecation_date="2027-04-06",
supported_flags=( supported_flags=(
"supports_function_calling", "supports_function_calling",
"supports_reasoning", "supports_reasoning",
@ -109,6 +120,7 @@ TOKEN_PRICED_MODELS: Final = (
max_input_tokens=262000, max_input_tokens=262000,
max_output_tokens=8192, max_output_tokens=8192,
cache_read_input_token_cost=None, cache_read_input_token_cost=None,
deprecation_date="2027-04-06",
supported_flags=( supported_flags=(
"supports_function_calling", "supports_function_calling",
"supports_response_schema", "supports_response_schema",
@ -146,7 +158,9 @@ def test_azure_ai_catalog_name_is_priced_and_routed(spec: TokenPricedCatalogMode
@pytest.mark.usefixtures("local_model_cost_map") @pytest.mark.usefixtures("local_model_cost_map")
@pytest.mark.parametrize( @pytest.mark.parametrize(
"spec", [spec for spec in TOKEN_PRICED_MODELS if spec.catalog_name != "model-router"], ids=lambda spec: spec.catalog_name "spec",
[spec for spec in TOKEN_PRICED_MODELS if spec.catalog_name != "model-router"],
ids=lambda spec: spec.catalog_name,
) )
def test_azure_ai_catalog_name_costs_a_million_tokens_at_list_price(spec: TokenPricedCatalogModel) -> None: def test_azure_ai_catalog_name_costs_a_million_tokens_at_list_price(spec: TokenPricedCatalogModel) -> None:
prompt_cost, completion_cost = cost_per_token( prompt_cost, completion_cost = cost_per_token(
@ -174,3 +188,9 @@ def test_azure_ai_catalog_entry_source_and_backup_match(catalog_name: str) -> No
assert str(main_entry["source"]).startswith("https://azure.microsoft.com/en-us/pricing/details/") assert str(main_entry["source"]).startswith("https://azure.microsoft.com/en-us/pricing/details/")
assert backup_entry == main_entry assert backup_entry == main_entry
@pytest.mark.parametrize("spec", TOKEN_PRICED_MODELS, ids=lambda spec: spec.catalog_name)
def test_azure_ai_catalog_entry_carries_its_retirement_date(spec: TokenPricedCatalogModel) -> None:
entry = _cost_map_entry(REPO_ROOT / "model_prices_and_context_window.json", spec.catalog_name)
assert entry.get("deprecation_date") == spec.deprecation_date