Merge pull request #40189 from BerriAI/litellm_lit_3157_azure_ai_catalog_models

fix(azure_ai): price seven Foundry catalog names and charge the model router fee once
This commit is contained in:
Mateo Wang 2026-09-08 20:08:40 -07:00 committed by GitHub
commit ee7c7e14f3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
9 changed files with 717 additions and 420 deletions

View file

@ -45,6 +45,9 @@ from litellm.llms.azure.cost_calculation import (
from litellm.llms.azure_ai.cost_calculator import (
cost_per_token as azure_ai_cost_per_token,
)
from litellm.llms.azure_ai.cost_calculator import (
is_azure_model_router as azure_ai_is_model_router_name,
)
from litellm.llms.base_llm.search.transformation import SearchResponse
from litellm.llms.bedrock.cost_calculation import (
cost_per_token as bedrock_cost_per_token,
@ -1659,11 +1662,10 @@ def completion_cost(
data_residency=data_residency,
vertex_location=vertex_location,
response=completion_response,
request_model=request_model_for_cost,
)
# Get additional costs from provider (e.g., routing fees, infrastructure costs)
if custom_llm_provider == "azure_ai":
if custom_llm_provider == "azure_ai" and not azure_ai_is_model_router_name(model):
model_for_additional_costs = request_model_for_cost
if completion_response is not None:
hidden_params = getattr(completion_response, "_hidden_params", None) or {}

View file

@ -37,6 +37,7 @@ class AzureAudioTranscription(AzureChatCompletion):
azure_ad_token: str | None = None,
atranscription: bool = False,
litellm_params: dict | None = None,
custom_llm_provider: str = "azure",
) -> TranscriptionResponse | Coroutine[Any, Any, TranscriptionResponse]:
data: Final = {"model": model, "file": audio_file, **optional_params}
@ -53,6 +54,7 @@ class AzureAudioTranscription(AzureChatCompletion):
logging_obj=logging_obj,
model=model,
litellm_params=litellm_params,
custom_llm_provider=custom_llm_provider,
)
azure_client: Final = self.get_azure_openai_client(
@ -99,7 +101,7 @@ class AzureAudioTranscription(AzureChatCompletion):
additional_args={"complete_input_dict": data},
original_response=stringified_response,
)
hidden_params: Final = {"model": model, "custom_llm_provider": "azure"}
hidden_params: Final = {"model": model, "custom_llm_provider": custom_llm_provider}
final_response: Final[TranscriptionResponse] = convert_to_model_response_object(
response_object=stringified_response,
model_response_object=model_response,
@ -122,6 +124,7 @@ class AzureAudioTranscription(AzureChatCompletion):
client=None,
max_retries=None,
litellm_params: dict | None = None,
custom_llm_provider: str = "azure",
) -> TranscriptionResponse:
response = None
try:
@ -178,7 +181,7 @@ class AzureAudioTranscription(AzureChatCompletion):
},
original_response=stringified_response,
)
hidden_params: Final = {"model": model, "custom_llm_provider": "azure"}
hidden_params: Final = {"model": model, "custom_llm_provider": custom_llm_provider}
response = convert_to_model_response_object(
_response_headers=headers,
response_object=stringified_response,

View file

@ -11,7 +11,7 @@ from litellm.types.utils import Usage
from litellm.utils import get_model_info
def _is_azure_model_router(model: str) -> bool:
def is_azure_model_router(model: str) -> bool:
"""
Check if the model is Azure AI Foundry Model Router.
@ -31,6 +31,18 @@ def _is_azure_model_router(model: str) -> bool:
return "model-router" in model_lower or "model_router" in model_lower or model_lower == "azure-model-router"
ROUTER_FEE_ENTRY_NAMES: Final = frozenset({"model-router", "model_router"})
def is_router_fee_entry(model: str) -> bool:
return model.lower().removeprefix("azure_ai/") in ROUTER_FEE_ENTRY_NAMES
def _router_fee_entry_name(model: str) -> str:
entry_name: Final = model.lower().removeprefix("azure_ai/")
return entry_name if entry_name in ROUTER_FEE_ENTRY_NAMES else "model_router"
def calculate_azure_model_router_flat_cost(model: str, prompt_tokens: int) -> float:
"""
Calculate the flat cost for Azure AI Foundry Model Router.
@ -42,20 +54,39 @@ def calculate_azure_model_router_flat_cost(model: str, prompt_tokens: int) -> fl
Returns:
float: The flat cost in USD, or 0.0 if not applicable
"""
if not _is_azure_model_router(model):
if not is_azure_model_router(model):
return 0.0
# Get the model router pricing from model_prices_and_context_window.json
# Use "model_router" as the key (without actual model name suffix)
model_info: Final = get_model_info(model="model_router", custom_llm_provider="azure_ai")
model_info: Final = get_model_info(model=_router_fee_entry_name(model), custom_llm_provider="azure_ai")
router_flat_cost_per_token: Final = model_info.get("input_cost_per_token", 0)
if router_flat_cost_per_token and router_flat_cost_per_token > 0:
return prompt_tokens * router_flat_cost_per_token
return 0.0
def _response_model_cost(model: str, usage: Usage, service_tier: str | None) -> tuple[float, float]:
try:
return generic_cost_per_token(
model=model, usage=usage, custom_llm_provider="azure_ai", service_tier=service_tier
)
except Exception as e:
if not is_azure_model_router(model):
raise
verbose_logger.debug(
"Azure AI Model Router: model '%s' not in cost map, only the routing fee applies. Error: %s", model, e
)
return 0.0, 0.0
def _router_fee_name(model: str, request_model: str | None) -> str | None:
if is_router_fee_entry(model):
return None
if is_azure_model_router(model):
return model
if request_model is not None and is_azure_model_router(request_model):
return request_model
return None
def cost_per_token(
model: str,
usage: Usage,
@ -64,68 +95,31 @@ def cost_per_token(
service_tier: str | None = None,
) -> tuple[float, float]:
"""
Calculate the cost per token for Azure AI models.
Price the response model's own tokens for Azure AI, plus the Model Router fee exactly once when either the
priced name or request_model is a Model Router name.
For Azure AI Foundry Model Router:
- Adds a flat cost of $0.14 per million input tokens (from model_prices_and_context_window.json)
- Plus the cost of the actual model used (handled by generic_cost_per_token)
A response priced as the router entry itself already carries the fee, so nothing is added on top of it. A
router deployment name that is missing from the cost map prices at the fee alone.
completion_cost passes only the priced name: when that name is a routed model it adds the fee itself through
AzureModelRouterConfig.calculate_additional_costs as the "Azure Model Router Flat Cost" line of the cost
breakdown, and when the name is router-shaped the fee is already in the prompt cost returned here.
Args:
model: str, the model name without provider prefix (from response)
usage: LiteLLM Usage block
response_time_ms: Optional response time in milliseconds
request_model: Optional[str], the original request model name (to detect router usage)
request_model: Optional[str], the original request model name; a Model Router name adds the routing fee
service_tier: Optional service tier the request was priced on
Returns:
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
Raises:
ValueError: If the model is not found in the cost map and cost cannot be calculated
(except for Model Router models where we return just the routing flat cost)
ValueError: If a model that is not a Model Router name is missing from the cost map
"""
prompt_cost = 0.0
completion_cost = 0.0
# Determine if this was a model router request
# Check both the response model and the request model
is_router_request: Final = _is_azure_model_router(model) or (
request_model is not None and _is_azure_model_router(request_model)
)
# Calculate base cost using generic cost calculator
# This may raise an exception if the model is not in the cost map
try:
prompt_cost, completion_cost = generic_cost_per_token(
model=model,
usage=usage,
custom_llm_provider="azure_ai",
service_tier=service_tier,
)
except Exception as e:
# For Model Router, the model name (e.g., "azure-model-router") may not be in the cost map
# because it's a routing service, not an actual model. In this case, we continue
# to calculate just the routing flat cost.
if not _is_azure_model_router(model):
# Re-raise for non-router models - they should have pricing defined
raise
verbose_logger.debug(
"Azure AI Model Router: model '%s' not in cost map, calculating routing flat cost only. Error: %s", model, e
)
# Add flat cost for Azure Model Router
# The flat cost is defined in model_prices_and_context_window.json for azure_ai/model_router
if is_router_request:
# Use the request model for flat cost calculation if available, otherwise use response model
router_model_for_calc: Final = request_model if request_model else model
router_flat_cost: Final = calculate_azure_model_router_flat_cost(router_model_for_calc, usage.prompt_tokens)
if router_flat_cost > 0:
verbose_logger.debug(
f"Azure AI Model Router flat cost: ${router_flat_cost:.6f} "
f"({usage.prompt_tokens} tokens × ${router_flat_cost / usage.prompt_tokens:.9f}/token)"
)
# Add flat cost to prompt cost
prompt_cost += router_flat_cost
return prompt_cost, completion_cost
prompt_cost, completion_cost = _response_model_cost(model=model, usage=usage, service_tier=service_tier)
fee_name: Final = _router_fee_name(model=model, request_model=request_model)
if fee_name is None:
return prompt_cost, completion_cost
return prompt_cost + calculate_azure_model_router_flat_cost(fee_name, usage.prompt_tokens), completion_cost

View file

@ -7805,6 +7805,7 @@ def transcription(
azure_ad_token=azure_ad_token,
max_retries=max_retries,
litellm_params=litellm_params_dict,
custom_llm_provider=custom_llm_provider,
)
elif custom_llm_provider == "openai" or (custom_llm_provider in litellm.openai_compatible_providers):
api_base = (

View file

@ -3626,6 +3626,79 @@
"supports_xhigh_reasoning_effort": true,
"supports_minimal_reasoning_effort": false
},
"azure_ai/gpt-chat-latest": {
"cache_read_input_token_cost": 5e-07,
"deprecation_date": "2026-12-02",
"input_cost_per_token": 5e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3e-05,
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true
},
"azure_ai/codex-mini": {
"cache_read_input_token_cost": 3.75e-07,
"deprecation_date": "2026-11-15",
"input_cost_per_token": 1.5e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 200000,
"max_output_tokens": 100000,
"max_tokens": 100000,
"mode": "responses",
"output_cost_per_token": 6e-06,
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
"supported_endpoints": [
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true
},
"azure_ai/whisper": {
"deprecation_date": "2026-12-15",
"input_cost_per_second": 0.0001,
"litellm_provider": "azure_ai",
"mode": "audio_transcription",
"output_cost_per_second": 0.0001,
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/"
},
"azure_ai/gpt-5.5-2026-04-23": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
@ -4029,13 +4102,29 @@
"supports_minimal_reasoning_effort": false
},
"azure_ai/model_router": {
"deprecation_date": "2027-05-20",
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 0,
"litellm_provider": "azure_ai",
"max_input_tokens": 200000,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/aoai/",
"comment": "Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)"
},
"azure_ai/model-router": {
"deprecation_date": "2027-05-20",
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 0,
"litellm_provider": "azure_ai",
"max_input_tokens": 200000,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/aoai/",
"comment": "Catalog-name twin of azure_ai/model_router: the flat $0.14 per M input tokens is the router's own fee, the routed model is priced on top of it"
},
"azure/eu/gpt-4o-2024-08-06": {
"deprecation_date": "2027-04-14",
"cache_read_input_token_cost": 1.375e-06,
@ -10347,6 +10436,18 @@
"/v1/ocr"
]
},
"azure_ai/cohere-command-a": {
"input_cost_per_token": 2.5e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 131072,
"max_output_tokens": 8182,
"max_tokens": 8182,
"mode": "chat",
"output_cost_per_token": 1e-05,
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/cohere/",
"supports_function_calling": true,
"supports_tool_choice": true
},
"azure_ai/doc-intelligence/prebuilt-read": {
"litellm_provider": "azure_ai",
"ocr_cost_per_page": 0.0015,
@ -10698,6 +10799,41 @@
"supports_vision": true,
"supports_web_search": true
},
"azure_ai/grok-4-20-reasoning": {
"cache_read_input_token_cost": 1.25e-06,
"deprecation_date": "2027-04-06",
"input_cost_per_token": 1.25e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 262000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 2.5e-06,
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/",
"supports_function_calling": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true,
"supports_reasoning": true
},
"azure_ai/grok-4-20-non-reasoning": {
"cache_read_input_token_cost": 1.25e-06,
"deprecation_date": "2027-04-06",
"input_cost_per_token": 1.25e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 262000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 2.5e-06,
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/",
"supports_function_calling": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true
},
"azure_ai/grok-4-fast-non-reasoning": {
"deprecation_date": "2026-05-01",
"input_cost_per_token": 2e-07,

View file

@ -3626,6 +3626,79 @@
"supports_xhigh_reasoning_effort": true,
"supports_minimal_reasoning_effort": false
},
"azure_ai/gpt-chat-latest": {
"cache_read_input_token_cost": 5e-07,
"deprecation_date": "2026-12-02",
"input_cost_per_token": 5e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 272000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3e-05,
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_function_calling": true,
"supports_native_streaming": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true
},
"azure_ai/codex-mini": {
"cache_read_input_token_cost": 3.75e-07,
"deprecation_date": "2026-11-15",
"input_cost_per_token": 1.5e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 200000,
"max_output_tokens": 100000,
"max_tokens": 100000,
"mode": "responses",
"output_cost_per_token": 6e-06,
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
"supported_endpoints": [
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true
},
"azure_ai/whisper": {
"deprecation_date": "2026-12-15",
"input_cost_per_second": 0.0001,
"litellm_provider": "azure_ai",
"mode": "audio_transcription",
"output_cost_per_second": 0.0001,
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/"
},
"azure_ai/gpt-5.5-2026-04-23": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
@ -4029,13 +4102,29 @@
"supports_minimal_reasoning_effort": false
},
"azure_ai/model_router": {
"deprecation_date": "2027-05-20",
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 0,
"litellm_provider": "azure_ai",
"max_input_tokens": 200000,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/aoai/",
"comment": "Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)"
},
"azure_ai/model-router": {
"deprecation_date": "2027-05-20",
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 0,
"litellm_provider": "azure_ai",
"max_input_tokens": 200000,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/aoai/",
"comment": "Catalog-name twin of azure_ai/model_router: the flat $0.14 per M input tokens is the router's own fee, the routed model is priced on top of it"
},
"azure/eu/gpt-4o-2024-08-06": {
"deprecation_date": "2027-04-14",
"cache_read_input_token_cost": 1.375e-06,
@ -10347,6 +10436,18 @@
"/v1/ocr"
]
},
"azure_ai/cohere-command-a": {
"input_cost_per_token": 2.5e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 131072,
"max_output_tokens": 8182,
"max_tokens": 8182,
"mode": "chat",
"output_cost_per_token": 1e-05,
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/cohere/",
"supports_function_calling": true,
"supports_tool_choice": true
},
"azure_ai/doc-intelligence/prebuilt-read": {
"litellm_provider": "azure_ai",
"ocr_cost_per_page": 0.0015,
@ -10698,6 +10799,41 @@
"supports_vision": true,
"supports_web_search": true
},
"azure_ai/grok-4-20-reasoning": {
"cache_read_input_token_cost": 1.25e-06,
"deprecation_date": "2027-04-06",
"input_cost_per_token": 1.25e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 262000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 2.5e-06,
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/",
"supports_function_calling": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true,
"supports_reasoning": true
},
"azure_ai/grok-4-20-non-reasoning": {
"cache_read_input_token_cost": 1.25e-06,
"deprecation_date": "2027-04-06",
"input_cost_per_token": 1.25e-06,
"litellm_provider": "azure_ai",
"max_input_tokens": 262000,
"max_output_tokens": 8192,
"max_tokens": 8192,
"mode": "chat",
"output_cost_per_token": 2.5e-06,
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/",
"supports_function_calling": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true
},
"azure_ai/grok-4-fast-non-reasoning": {
"deprecation_date": "2026-05-01",
"input_cost_per_token": 2e-07,

View file

@ -0,0 +1,61 @@
import json
from pathlib import Path
from typing import Final
import httpx
import pytest
from openai import AzureOpenAI
import litellm
from litellm.cost_calculator import completion_cost
from litellm.litellm_core_utils.audio_utils.utils import calculate_request_duration
AUDIO_FILE: Final = Path(__file__).parents[3] / "gettysburg.wav"
WHISPER_COST_PER_SECOND: Final = 0.0001
def _transcription_client() -> AzureOpenAI:
def handler(request: httpx.Request) -> httpx.Response:
return httpx.Response(200, json={"text": "Four score and seven years ago"})
return AzureOpenAI(
api_key="test-key",
api_version="2024-06-01",
azure_endpoint="https://example.cognitiveservices.azure.com",
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
)
def test_azure_ai_transcription_is_priced_at_the_azure_ai_entry():
with AUDIO_FILE.open("rb") as audio:
response = litellm.transcription(
model="azure_ai/whisper",
file=audio,
api_base="https://example.cognitiveservices.azure.com",
api_key="test-key",
api_version="2024-06-01",
client=_transcription_client(),
)
with AUDIO_FILE.open("rb") as audio:
duration = calculate_request_duration(audio)
assert duration is not None and duration > 0
assert response._hidden_params["custom_llm_provider"] == "azure_ai"
assert completion_cost(completion_response=response, call_type="transcription") == pytest.approx(
WHISPER_COST_PER_SECOND * duration
)
def test_azure_transcription_keeps_the_azure_provider():
with AUDIO_FILE.open("rb") as audio:
response = litellm.transcription(
model="azure/whisper-1",
file=audio,
api_base="https://example.openai.azure.com",
api_key="test-key",
api_version="2024-06-01",
client=_transcription_client(),
)
assert response._hidden_params["custom_llm_provider"] == "azure"
assert json.loads(response.model_dump_json())["text"] == "Four score and seven years ago"

View file

@ -2,20 +2,25 @@
Test Azure AI cost calculator, especially Model Router flat cost.
"""
from datetime import datetime
from typing import Final
import pytest
import litellm
from litellm.cost_calculator import completion_cost
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.llms.azure_ai.cost_calculator import (
_is_azure_model_router,
calculate_azure_model_router_flat_cost,
cost_per_token,
is_azure_model_router,
)
from litellm.types.utils import Usage
from litellm.types.utils import Choices, Message, ModelResponse, Usage
from litellm.utils import get_model_info
# Get the flat cost from model_prices_and_context_window.json
_model_info = get_model_info(model="model_router", custom_llm_provider="azure_ai")
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = (
_model_info.get("input_cost_per_token", 0) * 1_000_000
)
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = _model_info.get("input_cost_per_token", 0) * 1_000_000
class TestAzureModelRouterDetection:
@ -49,7 +54,7 @@ class TestAzureModelRouterDetection:
)
def test_is_azure_model_router(self, model: str, expected: bool):
"""Test Azure Model Router detection."""
assert _is_azure_model_router(model) == expected
assert is_azure_model_router(model) == expected
class TestAzureModelRouterPrefix:
@ -80,108 +85,60 @@ class TestAzureModelRouterPrefix:
assert result == expected
ROUTER_FEE_PER_TOKEN: Final = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
ROUTED_MODEL: Final = "gpt-4.1-nano-2025-04-14"
ROUTED_USAGE: Final = Usage(prompt_tokens=5000, completion_tokens=2000, total_tokens=7000)
ROUTED_FEE: Final = 5000 * ROUTER_FEE_PER_TOKEN
def _router_logging(request_model: str) -> Logging:
return Logging(
model=request_model,
messages=[{"role": "user", "content": "Hello"}],
stream=False,
call_type="completion",
start_time=datetime.now(),
litellm_call_id="test-123",
function_id="test-function",
)
def _azure_ai_response(response_model: str, litellm_model_name: str | None = None) -> ModelResponse:
response: Final = ModelResponse(
id="test-123",
choices=[Choices(finish_reason="stop", index=0, message=Message(role="assistant", content="Hello"))],
created=1234567890,
model=response_model,
object="chat.completion",
usage=ROUTED_USAGE,
)
response._hidden_params = (
{"custom_llm_provider": "azure_ai"}
if litellm_model_name is None
else {"custom_llm_provider": "azure_ai", "litellm_model_name": litellm_model_name}
)
return response
def _routed_model_cost() -> tuple[float, float]:
routed_info: Final = get_model_info(model=ROUTED_MODEL, custom_llm_provider="azure_ai")
return (
ROUTED_USAGE.prompt_tokens * (routed_info["input_cost_per_token"] or 0.0),
ROUTED_USAGE.completion_tokens * (routed_info["output_cost_per_token"] or 0.0),
)
@pytest.mark.usefixtures("local_model_cost_map")
class TestAzureModelRouterFlatCost:
"""Test Azure AI Foundry Model Router flat cost calculation."""
"""cost_per_token charges the router fee once, for whichever router name the caller gives it."""
def test_model_router_flat_cost_basic(self):
"""Test that flat cost is added for Model Router requests."""
model = "azure-model-router"
usage = Usage(
prompt_tokens=1000,
completion_tokens=500,
total_tokens=1500,
)
def test_unmapped_router_deployment_name_prices_the_fee(self) -> None:
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
prompt_cost, completion_cost_usd = cost_per_token(model="azure-model-router", usage=usage)
assert prompt_cost == pytest.approx(1000 * ROUTER_FEE_PER_TOKEN, rel=1e-9)
assert completion_cost_usd == 0.0
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Calculate expected flat cost
expected_flat_cost = (
usage.prompt_tokens
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
/ 1_000_000
)
# Flat cost should be $0.00014 (1000 tokens × $0.14 / 1M tokens)
assert expected_flat_cost == pytest.approx(0.00014, rel=1e-9)
# Prompt cost should include the flat cost
# (plus any base cost from the actual model used, which might be 0 if not in model_cost)
assert prompt_cost >= expected_flat_cost
print(
f"Model Router flat cost for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
)
print(f"Total prompt cost: ${prompt_cost:.6f}")
def test_model_router_flat_cost_large_request(self):
"""Test flat cost calculation for larger requests."""
model = "model-router"
usage = Usage(
prompt_tokens=100_000,
completion_tokens=50_000,
total_tokens=150_000,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Calculate expected flat cost
expected_flat_cost = (
usage.prompt_tokens
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
/ 1_000_000
)
# Flat cost should be $0.014 (100k tokens × $0.14 / 1M tokens)
assert expected_flat_cost == pytest.approx(0.014, rel=1e-9)
# Use approx for floating-point comparison
assert prompt_cost >= expected_flat_cost or prompt_cost == pytest.approx(
expected_flat_cost, rel=1e-9
)
print(
f"Model Router flat cost for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
)
print(f"Total prompt cost: ${prompt_cost:.6f}")
def test_model_router_flat_cost_1m_tokens(self):
"""Test flat cost for exactly 1 million input tokens."""
model = "azure-model-router"
usage = Usage(
prompt_tokens=1_000_000,
completion_tokens=100_000,
total_tokens=1_100_000,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Calculate expected flat cost
expected_flat_cost = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
# Flat cost should be exactly $0.14 for 1M tokens
assert expected_flat_cost == pytest.approx(0.14, rel=1e-9)
assert prompt_cost >= expected_flat_cost
print(f"Model Router flat cost for 1M tokens: ${expected_flat_cost:.6f}")
print(f"Total prompt cost: ${prompt_cost:.6f}")
def test_non_model_router_no_flat_cost(self):
"""Test that non-Model Router models don't get the flat cost."""
model = "gpt-4o"
usage = Usage(
prompt_tokens=1000,
completion_tokens=500,
total_tokens=1500,
)
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# No flat cost should be added for non-Model Router models
# The cost might be 0 or based on the model's pricing
print(f"Non-Model Router prompt cost: ${prompt_cost:.6f}")
# We just ensure it doesn't crash and returns valid values
assert prompt_cost >= 0
assert completion_cost >= 0
def test_model_router_with_cached_tokens(self):
"""Test Model Router flat cost with cached tokens."""
model = "azure-model-router"
def test_unmapped_router_deployment_name_charges_the_fee_over_cached_prompt_tokens_too(self) -> None:
usage = Usage(
prompt_tokens=2000,
completion_tokens=800,
@ -189,268 +146,165 @@ class TestAzureModelRouterFlatCost:
cache_read_input_tokens=500,
cache_creation_input_tokens=200,
)
prompt_cost, completion_cost_usd = cost_per_token(model="azure-model-router", usage=usage)
assert prompt_cost == pytest.approx(2000 * ROUTER_FEE_PER_TOKEN, rel=1e-9)
assert completion_cost_usd == 0.0
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
# Flat cost is based on ALL prompt tokens (including cached)
expected_flat_cost = (
usage.prompt_tokens
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
/ 1_000_000
def test_router_deployment_name_as_both_names_charges_the_fee_once(self) -> None:
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
prompt_cost, completion_cost_usd = cost_per_token(
model="model_router/my-deployment", usage=usage, request_model="azure_ai/model_router/my-deployment"
)
assert prompt_cost == pytest.approx(1000 * ROUTER_FEE_PER_TOKEN, rel=1e-9)
assert completion_cost_usd == 0.0
assert expected_flat_cost == pytest.approx(0.00028, rel=1e-9)
assert prompt_cost >= expected_flat_cost
print(
f"Model Router flat cost with caching for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
@pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
def test_router_entry_prices_its_own_fee(self, router_entry_name: str) -> None:
usage = Usage(prompt_tokens=1_000_000, completion_tokens=0, total_tokens=1_000_000)
prompt_cost, completion_cost_usd = cost_per_token(model=router_entry_name, usage=usage)
assert prompt_cost == pytest.approx(0.14, rel=1e-9)
assert completion_cost_usd == 0.0
def test_routed_model_is_priced_as_itself(self) -> None:
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
prompt_cost, completion_cost_usd = cost_per_token(model=ROUTED_MODEL, usage=ROUTED_USAGE)
assert routed_prompt_cost > 0
assert prompt_cost == pytest.approx(routed_prompt_cost, rel=1e-9)
assert completion_cost_usd == pytest.approx(routed_completion_cost, rel=1e-9)
def test_unmapped_model_that_is_not_a_router_name_raises(self) -> None:
usage = Usage(prompt_tokens=10, completion_tokens=10, total_tokens=20)
with pytest.raises(Exception, match="no-such-azure-ai-model"):
cost_per_token(model="no-such-azure-ai-model", usage=usage)
def test_request_model_through_the_router_adds_the_fee_once(self) -> None:
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
prompt_cost, completion_cost_usd = cost_per_token(
model=ROUTED_MODEL, usage=ROUTED_USAGE, request_model="azure_ai/model-router"
)
print(f"Total prompt cost: ${prompt_cost:.6f}")
assert prompt_cost == pytest.approx(routed_prompt_cost + ROUTED_FEE, rel=1e-9)
assert completion_cost_usd == pytest.approx(routed_completion_cost, rel=1e-9)
def test_router_flat_cost_when_response_has_actual_model(self):
"""
Test that router flat cost is added when request was via router but response
contains the actual model (e.g., gpt-5-nano).
def test_request_model_that_is_not_the_router_adds_nothing(self) -> None:
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
assert cost_per_token(
model=ROUTED_MODEL, usage=ROUTED_USAGE, request_model=f"azure_ai/{ROUTED_MODEL}"
) == pytest.approx((routed_prompt_cost, routed_completion_cost), rel=1e-9)
This is the key fix: Azure returns the actual model in the response, but we
must still add the router flat cost because the request was made via model router.
"""
usage = Usage(
prompt_tokens=10000,
completion_tokens=5000,
total_tokens=15000,
@pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
def test_request_model_does_not_double_the_router_entry(self, router_entry_name: str) -> None:
prompt_cost, completion_cost_usd = cost_per_token(
model=router_entry_name, usage=ROUTED_USAGE, request_model=f"azure_ai/{router_entry_name}"
)
assert prompt_cost == pytest.approx(ROUTED_FEE, rel=1e-9)
assert completion_cost_usd == 0.0
# Response model is the actual model Azure used (not a router name)
response_model = "gpt-5-nano-2025-08-07"
# Request model is the router - user called azure_ai/model_router/model-router
request_model = "azure_ai/model_router/model-router"
prompt_cost, completion_cost = cost_per_token(
model=response_model,
usage=usage,
request_model=request_model,
def test_public_cost_per_token_keeps_the_request_model_keyword(self) -> None:
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
prompt_cost, completion_cost_usd = litellm.cost_per_token(
model=ROUTED_MODEL,
custom_llm_provider="azure_ai",
usage_object=ROUTED_USAGE,
request_model="azure_ai/model-router",
)
assert prompt_cost == pytest.approx(routed_prompt_cost + ROUTED_FEE, rel=1e-9)
assert completion_cost_usd == pytest.approx(routed_completion_cost, rel=1e-9)
# Expected: model cost (from gpt-5-nano) + router flat cost
expected_flat_cost = (
usage.prompt_tokens
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
/ 1_000_000
def test_flat_cost_helper(self) -> None:
assert calculate_azure_model_router_flat_cost(
model="azure-model-router", prompt_tokens=10_000
) == pytest.approx(0.0014, rel=1e-9)
assert calculate_azure_model_router_flat_cost(model="gpt-5-nano", prompt_tokens=10_000) == 0.0
def test_flat_cost_reads_the_fee_from_the_deployment_named_entry(self) -> None:
litellm.register_model(
{"azure_ai/model-router": {"input_cost_per_token": 2e-07, "litellm_provider": "azure_ai", "mode": "chat"}}
)
assert expected_flat_cost == pytest.approx(0.0014, rel=1e-9)
# Total cost should be model cost + flat cost
total_cost = prompt_cost + completion_cost
assert total_cost >= expected_flat_cost
# Prompt cost should include both model prompt cost and router flat cost
assert prompt_cost >= expected_flat_cost
litellm.get_model_info.cache_clear()
assert calculate_azure_model_router_flat_cost(model="model-router", prompt_tokens=1_000_000) == pytest.approx(
0.2, rel=1e-9
)
assert calculate_azure_model_router_flat_cost(
model="azure-model-router", prompt_tokens=1_000_000
) == pytest.approx(0.14, rel=1e-9)
@pytest.mark.usefixtures("local_model_cost_map")
class TestAzureModelRouterCostBreakdown:
"""Test that Azure Model Router flat cost is tracked in cost breakdown."""
"""completion_cost charges the router fee exactly once: as the breakdown's additional cost line when a routed
model is priced as itself, inside the input cost when the priced name is the router."""
def test_flat_cost_calculation_helper(self):
"""Test that flat cost can be calculated using the helper function."""
from litellm.llms.azure_ai.cost_calculator import (
calculate_azure_model_router_flat_cost,
)
model = "azure-model-router"
prompt_tokens = 10000
# Calculate flat cost using helper function
flat_cost = calculate_azure_model_router_flat_cost(
model=model, prompt_tokens=prompt_tokens
)
# Expected flat cost
expected_flat_cost = (
prompt_tokens * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
)
assert flat_cost > 0
assert flat_cost == pytest.approx(expected_flat_cost, rel=1e-9)
print(f"Flat cost calculated: ${flat_cost:.6f}")
def test_flat_cost_integration_with_completion_cost(self):
"""Test that flat cost is properly integrated into completion_cost calculation."""
import litellm
from litellm.cost_calculator import completion_cost
from litellm.types.utils import Choices, Message, ModelResponse, Usage
# Create a mock response for azure_ai model router
response = ModelResponse(
id="test-123",
choices=[
Choices(
finish_reason="stop",
index=0,
message=Message(
role="assistant",
content="Test response",
),
)
],
created=1234567890,
model="azure-model-router",
object="chat.completion",
usage=Usage(
prompt_tokens=5000,
completion_tokens=2000,
total_tokens=7000,
),
)
# Set hidden params for provider
response._hidden_params = {"custom_llm_provider": "azure_ai"}
# Calculate cost
def test_unmapped_router_deployment_name_costs_only_the_fee(self) -> None:
cost = completion_cost(
completion_response=response,
completion_response=_azure_ai_response("azure-model-router"),
model="azure-model-router",
custom_llm_provider="azure_ai",
)
assert cost == pytest.approx(ROUTED_FEE, rel=1e-9)
# Expected flat cost
expected_flat_cost = (
5000 * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
)
# Cost should include the flat cost (use approx for floating-point comparison)
assert cost >= expected_flat_cost or cost == pytest.approx(
expected_flat_cost, rel=1e-9
)
print(f"Total cost with flat fee: ${cost:.6f}")
print(f"Expected minimum flat cost: ${expected_flat_cost:.6f}")
def test_additional_costs_in_cost_breakdown(self):
"""Test that Azure Model Router flat cost appears in additional_costs dict."""
from datetime import datetime
from litellm.cost_calculator import completion_cost
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.types.utils import Choices, Message, ModelResponse, Usage
# Create logging object with required parameters
logging_obj = Logging(
model="azure-model-router",
messages=[{"role": "user", "content": "Hello"}],
stream=False,
call_type="completion",
start_time=datetime.now(),
litellm_call_id="test-123",
function_id="test-function",
)
# Create a mock response for azure_ai model router
response = ModelResponse(
id="test-123",
choices=[
Choices(
finish_reason="stop",
index=0,
message=Message(
role="assistant",
content="Test response",
),
)
],
created=1234567890,
model="azure-model-router",
object="chat.completion",
usage=Usage(
prompt_tokens=5000,
completion_tokens=2000,
total_tokens=7000,
),
)
# Set hidden params for provider
response._hidden_params = {"custom_llm_provider": "azure_ai"}
# Calculate cost with logging object
def test_unmapped_router_name_carries_the_fee_as_its_input_cost(self) -> None:
logging_obj = _router_logging("azure-model-router")
cost = completion_cost(
completion_response=response,
completion_response=_azure_ai_response("azure-model-router"),
model="azure-model-router",
custom_llm_provider="azure_ai",
litellm_logging_obj=logging_obj,
)
breakdown = logging_obj.cost_breakdown
assert breakdown is not None
assert breakdown["input_cost"] == pytest.approx(ROUTED_FEE, rel=1e-9)
assert "additional_costs" not in breakdown
assert cost == pytest.approx(ROUTED_FEE, rel=1e-9)
# Check that cost breakdown contains additional_costs
assert hasattr(logging_obj, "cost_breakdown")
assert logging_obj.cost_breakdown is not None
assert "additional_costs" in logging_obj.cost_breakdown
assert isinstance(logging_obj.cost_breakdown["additional_costs"], dict)
# Check that the Azure Model Router flat cost is in additional_costs
additional_costs = logging_obj.cost_breakdown["additional_costs"]
assert "Azure Model Router Flat Cost" in additional_costs
# Verify the flat cost value
expected_flat_cost = (
5000 * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
)
actual_flat_cost = additional_costs["Azure Model Router Flat Cost"]
assert actual_flat_cost == pytest.approx(expected_flat_cost, rel=1e-9)
print(f"Additional costs in breakdown: {additional_costs}")
print(f"Azure Model Router Flat Cost: ${actual_flat_cost:.6f}")
def test_additional_costs_when_response_has_actual_model_via_hidden_params(self):
"""additional_costs populated when response has actual model but request was via model router (hidden_params)."""
from datetime import datetime
from litellm.cost_calculator import completion_cost
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.types.utils import Choices, Message, ModelResponse, Usage
logging_obj = Logging(
model="gpt-4.1-nano-2025-04-14",
messages=[{"role": "user", "content": "Hello"}],
stream=False,
call_type="completion",
start_time=datetime.now(),
litellm_call_id="test-123",
function_id="test-function",
)
response = ModelResponse(
id="test-123",
choices=[
Choices(
finish_reason="stop",
index=0,
message=Message(role="assistant", content="Hello"),
)
],
created=1234567890,
model="gpt-4.1-nano-2025-04-14",
object="chat.completion",
usage=Usage(prompt_tokens=5000, completion_tokens=2000, total_tokens=7000),
)
response._hidden_params = {
"custom_llm_provider": "azure_ai",
"litellm_model_name": "azure_ai/model-router",
}
def test_router_request_with_routed_response_charges_the_fee_once(self) -> None:
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
logging_obj = _router_logging("model-router")
cost = completion_cost(
completion_response=response,
model="gpt-4.1-nano-2025-04-14",
completion_response=_azure_ai_response(ROUTED_MODEL),
model=ROUTED_MODEL,
custom_llm_provider="azure_ai",
litellm_logging_obj=logging_obj,
)
expected_flat_cost = (
5000 * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
breakdown = logging_obj.cost_breakdown
assert breakdown is not None
assert breakdown["input_cost"] == pytest.approx(routed_prompt_cost, rel=1e-9)
assert breakdown["output_cost"] == pytest.approx(routed_completion_cost, rel=1e-9)
assert breakdown.get("additional_costs") == pytest.approx(
{"Azure Model Router Flat Cost": ROUTED_FEE}, rel=1e-9
)
assert cost >= expected_flat_cost
assert logging_obj.cost_breakdown is not None
assert "additional_costs" in logging_obj.cost_breakdown
assert (
"Azure Model Router Flat Cost"
in logging_obj.cost_breakdown["additional_costs"]
assert cost == pytest.approx(routed_prompt_cost + routed_completion_cost + ROUTED_FEE, rel=1e-9)
def test_routed_response_named_by_hidden_params_charges_the_fee_once(self) -> None:
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
logging_obj = _router_logging(ROUTED_MODEL)
cost = completion_cost(
completion_response=_azure_ai_response(ROUTED_MODEL, litellm_model_name="azure_ai/model-router"),
model=ROUTED_MODEL,
custom_llm_provider="azure_ai",
litellm_logging_obj=logging_obj,
)
assert logging_obj.cost_breakdown["additional_costs"][
"Azure Model Router Flat Cost"
] == pytest.approx(expected_flat_cost, rel=1e-9)
breakdown = logging_obj.cost_breakdown
assert breakdown is not None
assert breakdown["input_cost"] == pytest.approx(routed_prompt_cost, rel=1e-9)
assert breakdown.get("additional_costs") == pytest.approx(
{"Azure Model Router Flat Cost": ROUTED_FEE}, rel=1e-9
)
assert cost == pytest.approx(routed_prompt_cost + routed_completion_cost + ROUTED_FEE, rel=1e-9)
@pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
def test_response_priced_as_the_router_entry_charges_the_fee_once(self, router_entry_name: str) -> None:
logging_obj = _router_logging(router_entry_name)
cost = completion_cost(
completion_response=_azure_ai_response(router_entry_name),
model=router_entry_name,
custom_llm_provider="azure_ai",
litellm_logging_obj=logging_obj,
)
breakdown = logging_obj.cost_breakdown
assert breakdown is not None
assert "additional_costs" not in breakdown
assert breakdown["input_cost"] == pytest.approx(ROUTED_FEE, rel=1e-9)
assert cost == pytest.approx(ROUTED_FEE, rel=1e-9)
class TestAzureAIServiceTierCostCalculation:
@ -459,26 +313,27 @@ class TestAzureAIServiceTierCostCalculation:
@pytest.fixture(autouse=True)
def register_test_model(self):
import litellm
litellm.register_model(model_cost={
"test-azure-ai-model": {
"input_cost_per_token": 0.001,
"output_cost_per_token": 0.002,
"input_cost_per_token_priority": 0.01,
"output_cost_per_token_priority": 0.02,
"input_cost_per_token_flex": 0.0005,
"output_cost_per_token_flex": 0.001,
"litellm_provider": "azure_ai",
"max_tokens": 8192,
litellm.register_model(
model_cost={
"test-azure-ai-model": {
"input_cost_per_token": 0.001,
"output_cost_per_token": 0.002,
"input_cost_per_token_priority": 0.01,
"output_cost_per_token_priority": 0.02,
"input_cost_per_token_flex": 0.0005,
"output_cost_per_token_flex": 0.001,
"litellm_provider": "azure_ai",
"max_tokens": 8192,
}
}
})
)
def test_service_tier_priority_higher_cost(self):
"""Priority tier should cost more than standard for azure_ai."""
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
standard_prompt, standard_completion = cost_per_token(
model="test-azure-ai-model", usage=usage
)
standard_prompt, standard_completion = cost_per_token(model="test-azure-ai-model", usage=usage)
priority_prompt, priority_completion = cost_per_token(
model="test-azure-ai-model", usage=usage, service_tier="priority"
)
@ -490,12 +345,8 @@ class TestAzureAIServiceTierCostCalculation:
"""Flex tier should cost less than standard for azure_ai."""
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
standard_prompt, standard_completion = cost_per_token(
model="test-azure-ai-model", usage=usage
)
flex_prompt, flex_completion = cost_per_token(
model="test-azure-ai-model", usage=usage, service_tier="flex"
)
standard_prompt, standard_completion = cost_per_token(model="test-azure-ai-model", usage=usage)
flex_prompt, flex_completion = cost_per_token(model="test-azure-ai-model", usage=usage, service_tier="flex")
assert flex_prompt < standard_prompt
assert flex_completion < standard_completion

View file

@ -0,0 +1,113 @@
from pathlib import Path
from typing import Final
import pytest
from pydantic import TypeAdapter
from litellm import completion_cost, cost_per_token, get_model_info
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
from litellm.types.utils import TranscriptionResponse
REPO_ROOT: Final = Path(__file__).parents[4]
MAIN_COST_MAP: Final = REPO_ROOT / "model_prices_and_context_window.json"
BACKUP_COST_MAP: Final = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
COST_MAP_ADAPTER: Final = TypeAdapter(dict[str, dict[str, object]])
AZURE_PRICING_PREFIX: Final = "https://azure.microsoft.com/en-us/pricing/details/"
A_MILLION: Final = 1_000_000
AN_HOUR_IN_SECONDS: Final = 3600
TOKEN_PRICED_NAMES: Final = (
"gpt-chat-latest",
"codex-mini",
"model-router",
"cohere-command-a",
"grok-4-20-reasoning",
"grok-4-20-non-reasoning",
)
GROK_4_20_NAMES: Final = ("grok-4-20-reasoning", "grok-4-20-non-reasoning")
CATALOG_NAMES: Final = TOKEN_PRICED_NAMES + ("whisper",)
def _cost_map_entry(path: Path, catalog_name: str) -> dict[str, object]:
return COST_MAP_ADAPTER.validate_json(path.read_bytes())[f"azure_ai/{catalog_name}"]
def _whisper_transcription_cost(duration_seconds: int) -> float:
transcription: Final = TranscriptionResponse(text="hello")
transcription._hidden_params = { # pyright: ignore[reportPrivateUsage] # TranscriptionResponse exposes no public hidden-params setter
"custom_llm_provider": "azure_ai",
"model": "azure_ai/whisper",
"audio_transcription_duration": duration_seconds,
}
return completion_cost(
completion_response=transcription,
model="azure_ai/whisper",
custom_llm_provider="azure_ai",
call_type="atranscription",
)
@pytest.mark.parametrize("catalog_name", CATALOG_NAMES)
def test_azure_ai_catalog_name_routes_to_azure_ai(catalog_name: str) -> None:
routed_model, provider, _, _ = get_llm_provider(model=f"azure_ai/{catalog_name}")
assert (routed_model, provider) == (catalog_name, "azure_ai")
@pytest.mark.usefixtures("local_model_cost_map")
@pytest.mark.parametrize("catalog_name", TOKEN_PRICED_NAMES)
def test_azure_ai_catalog_name_charges_its_own_entry_per_token(catalog_name: str) -> None:
entry: Final = get_model_info(f"azure_ai/{catalog_name}")
prompt_cost, completion_cost_usd = cost_per_token(
model=f"azure_ai/{catalog_name}", prompt_tokens=A_MILLION, completion_tokens=A_MILLION
)
assert prompt_cost > 0
assert prompt_cost == pytest.approx(A_MILLION * entry["input_cost_per_token"])
assert completion_cost_usd == pytest.approx(A_MILLION * entry["output_cost_per_token"])
@pytest.mark.usefixtures("local_model_cost_map")
@pytest.mark.parametrize("catalog_name", TOKEN_PRICED_NAMES)
def test_azure_ai_catalog_name_prices_the_same_in_any_casing(catalog_name: str) -> None:
lowercase_cost = cost_per_token(model=f"azure_ai/{catalog_name}", prompt_tokens=A_MILLION, completion_tokens=0)
upper_cost = cost_per_token(model=f"azure_ai/{catalog_name.upper()}", prompt_tokens=A_MILLION, completion_tokens=0)
assert upper_cost == lowercase_cost
@pytest.mark.usefixtures("local_model_cost_map")
@pytest.mark.parametrize("catalog_name", GROK_4_20_NAMES)
def test_azure_ai_grok_4_20_bills_cached_prompt_tokens_at_the_input_price(catalog_name: str) -> None:
uncached_prompt_cost, _ = cost_per_token(model=f"azure_ai/{catalog_name}", prompt_tokens=A_MILLION, completion_tokens=0)
cached_prompt_cost, _ = cost_per_token(
model=f"azure_ai/{catalog_name}",
prompt_tokens=A_MILLION,
completion_tokens=0,
cache_read_input_tokens=A_MILLION,
)
assert uncached_prompt_cost > 0
assert cached_prompt_cost == pytest.approx(uncached_prompt_cost)
@pytest.mark.usefixtures("local_model_cost_map")
def test_azure_ai_whisper_catalog_name_is_priced_per_second() -> None:
one_second_cost: Final = _whisper_transcription_cost(1)
one_hour_cost: Final = _whisper_transcription_cost(AN_HOUR_IN_SECONDS)
assert one_second_cost > 0
assert one_hour_cost == pytest.approx(AN_HOUR_IN_SECONDS * one_second_cost)
@pytest.mark.parametrize("catalog_name", CATALOG_NAMES)
def test_azure_ai_catalog_entry_source_and_backup_match(catalog_name: str) -> None:
main_entry = _cost_map_entry(MAIN_COST_MAP, catalog_name)
backup_entry = _cost_map_entry(BACKUP_COST_MAP, catalog_name)
assert str(main_entry["source"]).startswith(AZURE_PRICING_PREFIX)
assert backup_entry == main_entry
def test_azure_ai_model_router_spellings_share_one_entry() -> None:
underscore_entry = _cost_map_entry(MAIN_COST_MAP, "model_router")
hyphen_entry = _cost_map_entry(MAIN_COST_MAP, "model-router")
assert {k: v for k, v in underscore_entry.items() if k != "comment"} == {
k: v for k, v in hyphen_entry.items() if k != "comment"
}