mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-09 22:31:41 +00:00
Merge pull request #40189 from BerriAI/litellm_lit_3157_azure_ai_catalog_models
fix(azure_ai): price seven Foundry catalog names and charge the model router fee once
This commit is contained in:
commit
ee7c7e14f3
9 changed files with 717 additions and 420 deletions
|
|
@ -45,6 +45,9 @@ from litellm.llms.azure.cost_calculation import (
|
|||
from litellm.llms.azure_ai.cost_calculator import (
|
||||
cost_per_token as azure_ai_cost_per_token,
|
||||
)
|
||||
from litellm.llms.azure_ai.cost_calculator import (
|
||||
is_azure_model_router as azure_ai_is_model_router_name,
|
||||
)
|
||||
from litellm.llms.base_llm.search.transformation import SearchResponse
|
||||
from litellm.llms.bedrock.cost_calculation import (
|
||||
cost_per_token as bedrock_cost_per_token,
|
||||
|
|
@ -1659,11 +1662,10 @@ def completion_cost(
|
|||
data_residency=data_residency,
|
||||
vertex_location=vertex_location,
|
||||
response=completion_response,
|
||||
request_model=request_model_for_cost,
|
||||
)
|
||||
|
||||
# Get additional costs from provider (e.g., routing fees, infrastructure costs)
|
||||
if custom_llm_provider == "azure_ai":
|
||||
if custom_llm_provider == "azure_ai" and not azure_ai_is_model_router_name(model):
|
||||
model_for_additional_costs = request_model_for_cost
|
||||
if completion_response is not None:
|
||||
hidden_params = getattr(completion_response, "_hidden_params", None) or {}
|
||||
|
|
|
|||
|
|
@ -37,6 +37,7 @@ class AzureAudioTranscription(AzureChatCompletion):
|
|||
azure_ad_token: str | None = None,
|
||||
atranscription: bool = False,
|
||||
litellm_params: dict | None = None,
|
||||
custom_llm_provider: str = "azure",
|
||||
) -> TranscriptionResponse | Coroutine[Any, Any, TranscriptionResponse]:
|
||||
data: Final = {"model": model, "file": audio_file, **optional_params}
|
||||
|
||||
|
|
@ -53,6 +54,7 @@ class AzureAudioTranscription(AzureChatCompletion):
|
|||
logging_obj=logging_obj,
|
||||
model=model,
|
||||
litellm_params=litellm_params,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
azure_client: Final = self.get_azure_openai_client(
|
||||
|
|
@ -99,7 +101,7 @@ class AzureAudioTranscription(AzureChatCompletion):
|
|||
additional_args={"complete_input_dict": data},
|
||||
original_response=stringified_response,
|
||||
)
|
||||
hidden_params: Final = {"model": model, "custom_llm_provider": "azure"}
|
||||
hidden_params: Final = {"model": model, "custom_llm_provider": custom_llm_provider}
|
||||
final_response: Final[TranscriptionResponse] = convert_to_model_response_object(
|
||||
response_object=stringified_response,
|
||||
model_response_object=model_response,
|
||||
|
|
@ -122,6 +124,7 @@ class AzureAudioTranscription(AzureChatCompletion):
|
|||
client=None,
|
||||
max_retries=None,
|
||||
litellm_params: dict | None = None,
|
||||
custom_llm_provider: str = "azure",
|
||||
) -> TranscriptionResponse:
|
||||
response = None
|
||||
try:
|
||||
|
|
@ -178,7 +181,7 @@ class AzureAudioTranscription(AzureChatCompletion):
|
|||
},
|
||||
original_response=stringified_response,
|
||||
)
|
||||
hidden_params: Final = {"model": model, "custom_llm_provider": "azure"}
|
||||
hidden_params: Final = {"model": model, "custom_llm_provider": custom_llm_provider}
|
||||
response = convert_to_model_response_object(
|
||||
_response_headers=headers,
|
||||
response_object=stringified_response,
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ from litellm.types.utils import Usage
|
|||
from litellm.utils import get_model_info
|
||||
|
||||
|
||||
def _is_azure_model_router(model: str) -> bool:
|
||||
def is_azure_model_router(model: str) -> bool:
|
||||
"""
|
||||
Check if the model is Azure AI Foundry Model Router.
|
||||
|
||||
|
|
@ -31,6 +31,18 @@ def _is_azure_model_router(model: str) -> bool:
|
|||
return "model-router" in model_lower or "model_router" in model_lower or model_lower == "azure-model-router"
|
||||
|
||||
|
||||
ROUTER_FEE_ENTRY_NAMES: Final = frozenset({"model-router", "model_router"})
|
||||
|
||||
|
||||
def is_router_fee_entry(model: str) -> bool:
|
||||
return model.lower().removeprefix("azure_ai/") in ROUTER_FEE_ENTRY_NAMES
|
||||
|
||||
|
||||
def _router_fee_entry_name(model: str) -> str:
|
||||
entry_name: Final = model.lower().removeprefix("azure_ai/")
|
||||
return entry_name if entry_name in ROUTER_FEE_ENTRY_NAMES else "model_router"
|
||||
|
||||
|
||||
def calculate_azure_model_router_flat_cost(model: str, prompt_tokens: int) -> float:
|
||||
"""
|
||||
Calculate the flat cost for Azure AI Foundry Model Router.
|
||||
|
|
@ -42,20 +54,39 @@ def calculate_azure_model_router_flat_cost(model: str, prompt_tokens: int) -> fl
|
|||
Returns:
|
||||
float: The flat cost in USD, or 0.0 if not applicable
|
||||
"""
|
||||
if not _is_azure_model_router(model):
|
||||
if not is_azure_model_router(model):
|
||||
return 0.0
|
||||
|
||||
# Get the model router pricing from model_prices_and_context_window.json
|
||||
# Use "model_router" as the key (without actual model name suffix)
|
||||
model_info: Final = get_model_info(model="model_router", custom_llm_provider="azure_ai")
|
||||
model_info: Final = get_model_info(model=_router_fee_entry_name(model), custom_llm_provider="azure_ai")
|
||||
router_flat_cost_per_token: Final = model_info.get("input_cost_per_token", 0)
|
||||
|
||||
if router_flat_cost_per_token and router_flat_cost_per_token > 0:
|
||||
return prompt_tokens * router_flat_cost_per_token
|
||||
|
||||
return 0.0
|
||||
|
||||
|
||||
def _response_model_cost(model: str, usage: Usage, service_tier: str | None) -> tuple[float, float]:
|
||||
try:
|
||||
return generic_cost_per_token(
|
||||
model=model, usage=usage, custom_llm_provider="azure_ai", service_tier=service_tier
|
||||
)
|
||||
except Exception as e:
|
||||
if not is_azure_model_router(model):
|
||||
raise
|
||||
verbose_logger.debug(
|
||||
"Azure AI Model Router: model '%s' not in cost map, only the routing fee applies. Error: %s", model, e
|
||||
)
|
||||
return 0.0, 0.0
|
||||
|
||||
|
||||
def _router_fee_name(model: str, request_model: str | None) -> str | None:
|
||||
if is_router_fee_entry(model):
|
||||
return None
|
||||
if is_azure_model_router(model):
|
||||
return model
|
||||
if request_model is not None and is_azure_model_router(request_model):
|
||||
return request_model
|
||||
return None
|
||||
|
||||
|
||||
def cost_per_token(
|
||||
model: str,
|
||||
usage: Usage,
|
||||
|
|
@ -64,68 +95,31 @@ def cost_per_token(
|
|||
service_tier: str | None = None,
|
||||
) -> tuple[float, float]:
|
||||
"""
|
||||
Calculate the cost per token for Azure AI models.
|
||||
Price the response model's own tokens for Azure AI, plus the Model Router fee exactly once when either the
|
||||
priced name or request_model is a Model Router name.
|
||||
|
||||
For Azure AI Foundry Model Router:
|
||||
- Adds a flat cost of $0.14 per million input tokens (from model_prices_and_context_window.json)
|
||||
- Plus the cost of the actual model used (handled by generic_cost_per_token)
|
||||
A response priced as the router entry itself already carries the fee, so nothing is added on top of it. A
|
||||
router deployment name that is missing from the cost map prices at the fee alone.
|
||||
|
||||
completion_cost passes only the priced name: when that name is a routed model it adds the fee itself through
|
||||
AzureModelRouterConfig.calculate_additional_costs as the "Azure Model Router Flat Cost" line of the cost
|
||||
breakdown, and when the name is router-shaped the fee is already in the prompt cost returned here.
|
||||
|
||||
Args:
|
||||
model: str, the model name without provider prefix (from response)
|
||||
usage: LiteLLM Usage block
|
||||
response_time_ms: Optional response time in milliseconds
|
||||
request_model: Optional[str], the original request model name (to detect router usage)
|
||||
request_model: Optional[str], the original request model name; a Model Router name adds the routing fee
|
||||
service_tier: Optional service tier the request was priced on
|
||||
|
||||
Returns:
|
||||
Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd
|
||||
|
||||
Raises:
|
||||
ValueError: If the model is not found in the cost map and cost cannot be calculated
|
||||
(except for Model Router models where we return just the routing flat cost)
|
||||
ValueError: If a model that is not a Model Router name is missing from the cost map
|
||||
"""
|
||||
prompt_cost = 0.0
|
||||
completion_cost = 0.0
|
||||
|
||||
# Determine if this was a model router request
|
||||
# Check both the response model and the request model
|
||||
is_router_request: Final = _is_azure_model_router(model) or (
|
||||
request_model is not None and _is_azure_model_router(request_model)
|
||||
)
|
||||
|
||||
# Calculate base cost using generic cost calculator
|
||||
# This may raise an exception if the model is not in the cost map
|
||||
try:
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model=model,
|
||||
usage=usage,
|
||||
custom_llm_provider="azure_ai",
|
||||
service_tier=service_tier,
|
||||
)
|
||||
except Exception as e:
|
||||
# For Model Router, the model name (e.g., "azure-model-router") may not be in the cost map
|
||||
# because it's a routing service, not an actual model. In this case, we continue
|
||||
# to calculate just the routing flat cost.
|
||||
if not _is_azure_model_router(model):
|
||||
# Re-raise for non-router models - they should have pricing defined
|
||||
raise
|
||||
verbose_logger.debug(
|
||||
"Azure AI Model Router: model '%s' not in cost map, calculating routing flat cost only. Error: %s", model, e
|
||||
)
|
||||
|
||||
# Add flat cost for Azure Model Router
|
||||
# The flat cost is defined in model_prices_and_context_window.json for azure_ai/model_router
|
||||
if is_router_request:
|
||||
# Use the request model for flat cost calculation if available, otherwise use response model
|
||||
router_model_for_calc: Final = request_model if request_model else model
|
||||
router_flat_cost: Final = calculate_azure_model_router_flat_cost(router_model_for_calc, usage.prompt_tokens)
|
||||
|
||||
if router_flat_cost > 0:
|
||||
verbose_logger.debug(
|
||||
f"Azure AI Model Router flat cost: ${router_flat_cost:.6f} "
|
||||
f"({usage.prompt_tokens} tokens × ${router_flat_cost / usage.prompt_tokens:.9f}/token)"
|
||||
)
|
||||
|
||||
# Add flat cost to prompt cost
|
||||
prompt_cost += router_flat_cost
|
||||
|
||||
return prompt_cost, completion_cost
|
||||
prompt_cost, completion_cost = _response_model_cost(model=model, usage=usage, service_tier=service_tier)
|
||||
fee_name: Final = _router_fee_name(model=model, request_model=request_model)
|
||||
if fee_name is None:
|
||||
return prompt_cost, completion_cost
|
||||
return prompt_cost + calculate_azure_model_router_flat_cost(fee_name, usage.prompt_tokens), completion_cost
|
||||
|
|
|
|||
|
|
@ -7805,6 +7805,7 @@ def transcription(
|
|||
azure_ad_token=azure_ad_token,
|
||||
max_retries=max_retries,
|
||||
litellm_params=litellm_params_dict,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
elif custom_llm_provider == "openai" or (custom_llm_provider in litellm.openai_compatible_providers):
|
||||
api_base = (
|
||||
|
|
|
|||
|
|
@ -3626,6 +3626,79 @@
|
|||
"supports_xhigh_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure_ai/gpt-chat-latest": {
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"deprecation_date": "2026-12-02",
|
||||
"input_cost_per_token": 5e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-05,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"azure_ai/codex-mini": {
|
||||
"cache_read_input_token_cost": 3.75e-07,
|
||||
"deprecation_date": "2026-11-15",
|
||||
"input_cost_per_token": 1.5e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 100000,
|
||||
"max_tokens": 100000,
|
||||
"mode": "responses",
|
||||
"output_cost_per_token": 6e-06,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
|
||||
"supported_endpoints": [
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"azure_ai/whisper": {
|
||||
"deprecation_date": "2026-12-15",
|
||||
"input_cost_per_second": 0.0001,
|
||||
"litellm_provider": "azure_ai",
|
||||
"mode": "audio_transcription",
|
||||
"output_cost_per_second": 0.0001,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/"
|
||||
},
|
||||
"azure_ai/gpt-5.5-2026-04-23": {
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
|
|
@ -4029,13 +4102,29 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure_ai/model_router": {
|
||||
"deprecation_date": "2027-05-20",
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 0,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "chat",
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/aoai/",
|
||||
"comment": "Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)"
|
||||
},
|
||||
"azure_ai/model-router": {
|
||||
"deprecation_date": "2027-05-20",
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 0,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "chat",
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/aoai/",
|
||||
"comment": "Catalog-name twin of azure_ai/model_router: the flat $0.14 per M input tokens is the router's own fee, the routed model is priced on top of it"
|
||||
},
|
||||
"azure/eu/gpt-4o-2024-08-06": {
|
||||
"deprecation_date": "2027-04-14",
|
||||
"cache_read_input_token_cost": 1.375e-06,
|
||||
|
|
@ -10347,6 +10436,18 @@
|
|||
"/v1/ocr"
|
||||
]
|
||||
},
|
||||
"azure_ai/cohere-command-a": {
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 8182,
|
||||
"max_tokens": 8182,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1e-05,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/cohere/",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"azure_ai/doc-intelligence/prebuilt-read": {
|
||||
"litellm_provider": "azure_ai",
|
||||
"ocr_cost_per_page": 0.0015,
|
||||
|
|
@ -10698,6 +10799,41 @@
|
|||
"supports_vision": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"azure_ai/grok-4-20-reasoning": {
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"deprecation_date": "2027-04-06",
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 262000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.5e-06,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/",
|
||||
"supports_function_calling": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"azure_ai/grok-4-20-non-reasoning": {
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"deprecation_date": "2027-04-06",
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 262000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.5e-06,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/",
|
||||
"supports_function_calling": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"azure_ai/grok-4-fast-non-reasoning": {
|
||||
"deprecation_date": "2026-05-01",
|
||||
"input_cost_per_token": 2e-07,
|
||||
|
|
|
|||
|
|
@ -3626,6 +3626,79 @@
|
|||
"supports_xhigh_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure_ai/gpt-chat-latest": {
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"deprecation_date": "2026-12-02",
|
||||
"input_cost_per_token": 5e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 272000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 3e-05,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"azure_ai/codex-mini": {
|
||||
"cache_read_input_token_cost": 3.75e-07,
|
||||
"deprecation_date": "2026-11-15",
|
||||
"input_cost_per_token": 1.5e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 100000,
|
||||
"max_tokens": 100000,
|
||||
"mode": "responses",
|
||||
"output_cost_per_token": 6e-06,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/",
|
||||
"supported_endpoints": [
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
"azure_ai/whisper": {
|
||||
"deprecation_date": "2026-12-15",
|
||||
"input_cost_per_second": 0.0001,
|
||||
"litellm_provider": "azure_ai",
|
||||
"mode": "audio_transcription",
|
||||
"output_cost_per_second": 0.0001,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/"
|
||||
},
|
||||
"azure_ai/gpt-5.5-2026-04-23": {
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
|
|
@ -4029,13 +4102,29 @@
|
|||
"supports_minimal_reasoning_effort": false
|
||||
},
|
||||
"azure_ai/model_router": {
|
||||
"deprecation_date": "2027-05-20",
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 0,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "chat",
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/aoai/",
|
||||
"comment": "Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)"
|
||||
},
|
||||
"azure_ai/model-router": {
|
||||
"deprecation_date": "2027-05-20",
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 0,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "chat",
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/aoai/",
|
||||
"comment": "Catalog-name twin of azure_ai/model_router: the flat $0.14 per M input tokens is the router's own fee, the routed model is priced on top of it"
|
||||
},
|
||||
"azure/eu/gpt-4o-2024-08-06": {
|
||||
"deprecation_date": "2027-04-14",
|
||||
"cache_read_input_token_cost": 1.375e-06,
|
||||
|
|
@ -10347,6 +10436,18 @@
|
|||
"/v1/ocr"
|
||||
]
|
||||
},
|
||||
"azure_ai/cohere-command-a": {
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 8182,
|
||||
"max_tokens": 8182,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1e-05,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/cohere/",
|
||||
"supports_function_calling": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"azure_ai/doc-intelligence/prebuilt-read": {
|
||||
"litellm_provider": "azure_ai",
|
||||
"ocr_cost_per_page": 0.0015,
|
||||
|
|
@ -10698,6 +10799,41 @@
|
|||
"supports_vision": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"azure_ai/grok-4-20-reasoning": {
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"deprecation_date": "2027-04-06",
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 262000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.5e-06,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/",
|
||||
"supports_function_calling": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"supports_reasoning": true
|
||||
},
|
||||
"azure_ai/grok-4-20-non-reasoning": {
|
||||
"cache_read_input_token_cost": 1.25e-06,
|
||||
"deprecation_date": "2027-04-06",
|
||||
"input_cost_per_token": 1.25e-06,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_input_tokens": 262000,
|
||||
"max_output_tokens": 8192,
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2.5e-06,
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/",
|
||||
"supports_function_calling": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true
|
||||
},
|
||||
"azure_ai/grok-4-fast-non-reasoning": {
|
||||
"deprecation_date": "2026-05-01",
|
||||
"input_cost_per_token": 2e-07,
|
||||
|
|
|
|||
61
tests/test_litellm/llms/azure/test_audio_transcriptions.py
Normal file
61
tests/test_litellm/llms/azure/test_audio_transcriptions.py
Normal file
|
|
@ -0,0 +1,61 @@
|
|||
import json
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
from openai import AzureOpenAI
|
||||
|
||||
import litellm
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.litellm_core_utils.audio_utils.utils import calculate_request_duration
|
||||
|
||||
AUDIO_FILE: Final = Path(__file__).parents[3] / "gettysburg.wav"
|
||||
WHISPER_COST_PER_SECOND: Final = 0.0001
|
||||
|
||||
|
||||
def _transcription_client() -> AzureOpenAI:
|
||||
def handler(request: httpx.Request) -> httpx.Response:
|
||||
return httpx.Response(200, json={"text": "Four score and seven years ago"})
|
||||
|
||||
return AzureOpenAI(
|
||||
api_key="test-key",
|
||||
api_version="2024-06-01",
|
||||
azure_endpoint="https://example.cognitiveservices.azure.com",
|
||||
http_client=httpx.Client(transport=httpx.MockTransport(handler)),
|
||||
)
|
||||
|
||||
|
||||
def test_azure_ai_transcription_is_priced_at_the_azure_ai_entry():
|
||||
with AUDIO_FILE.open("rb") as audio:
|
||||
response = litellm.transcription(
|
||||
model="azure_ai/whisper",
|
||||
file=audio,
|
||||
api_base="https://example.cognitiveservices.azure.com",
|
||||
api_key="test-key",
|
||||
api_version="2024-06-01",
|
||||
client=_transcription_client(),
|
||||
)
|
||||
with AUDIO_FILE.open("rb") as audio:
|
||||
duration = calculate_request_duration(audio)
|
||||
|
||||
assert duration is not None and duration > 0
|
||||
assert response._hidden_params["custom_llm_provider"] == "azure_ai"
|
||||
assert completion_cost(completion_response=response, call_type="transcription") == pytest.approx(
|
||||
WHISPER_COST_PER_SECOND * duration
|
||||
)
|
||||
|
||||
|
||||
def test_azure_transcription_keeps_the_azure_provider():
|
||||
with AUDIO_FILE.open("rb") as audio:
|
||||
response = litellm.transcription(
|
||||
model="azure/whisper-1",
|
||||
file=audio,
|
||||
api_base="https://example.openai.azure.com",
|
||||
api_key="test-key",
|
||||
api_version="2024-06-01",
|
||||
client=_transcription_client(),
|
||||
)
|
||||
|
||||
assert response._hidden_params["custom_llm_provider"] == "azure"
|
||||
assert json.loads(response.model_dump_json())["text"] == "Four score and seven years ago"
|
||||
|
|
@ -2,20 +2,25 @@
|
|||
Test Azure AI cost calculator, especially Model Router flat cost.
|
||||
"""
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging
|
||||
from litellm.llms.azure_ai.cost_calculator import (
|
||||
_is_azure_model_router,
|
||||
calculate_azure_model_router_flat_cost,
|
||||
cost_per_token,
|
||||
is_azure_model_router,
|
||||
)
|
||||
from litellm.types.utils import Usage
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
||||
from litellm.utils import get_model_info
|
||||
|
||||
# Get the flat cost from model_prices_and_context_window.json
|
||||
_model_info = get_model_info(model="model_router", custom_llm_provider="azure_ai")
|
||||
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = (
|
||||
_model_info.get("input_cost_per_token", 0) * 1_000_000
|
||||
)
|
||||
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = _model_info.get("input_cost_per_token", 0) * 1_000_000
|
||||
|
||||
|
||||
class TestAzureModelRouterDetection:
|
||||
|
|
@ -49,7 +54,7 @@ class TestAzureModelRouterDetection:
|
|||
)
|
||||
def test_is_azure_model_router(self, model: str, expected: bool):
|
||||
"""Test Azure Model Router detection."""
|
||||
assert _is_azure_model_router(model) == expected
|
||||
assert is_azure_model_router(model) == expected
|
||||
|
||||
|
||||
class TestAzureModelRouterPrefix:
|
||||
|
|
@ -80,108 +85,60 @@ class TestAzureModelRouterPrefix:
|
|||
assert result == expected
|
||||
|
||||
|
||||
ROUTER_FEE_PER_TOKEN: Final = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
|
||||
ROUTED_MODEL: Final = "gpt-4.1-nano-2025-04-14"
|
||||
ROUTED_USAGE: Final = Usage(prompt_tokens=5000, completion_tokens=2000, total_tokens=7000)
|
||||
ROUTED_FEE: Final = 5000 * ROUTER_FEE_PER_TOKEN
|
||||
|
||||
|
||||
def _router_logging(request_model: str) -> Logging:
|
||||
return Logging(
|
||||
model=request_model,
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
stream=False,
|
||||
call_type="completion",
|
||||
start_time=datetime.now(),
|
||||
litellm_call_id="test-123",
|
||||
function_id="test-function",
|
||||
)
|
||||
|
||||
|
||||
def _azure_ai_response(response_model: str, litellm_model_name: str | None = None) -> ModelResponse:
|
||||
response: Final = ModelResponse(
|
||||
id="test-123",
|
||||
choices=[Choices(finish_reason="stop", index=0, message=Message(role="assistant", content="Hello"))],
|
||||
created=1234567890,
|
||||
model=response_model,
|
||||
object="chat.completion",
|
||||
usage=ROUTED_USAGE,
|
||||
)
|
||||
response._hidden_params = (
|
||||
{"custom_llm_provider": "azure_ai"}
|
||||
if litellm_model_name is None
|
||||
else {"custom_llm_provider": "azure_ai", "litellm_model_name": litellm_model_name}
|
||||
)
|
||||
return response
|
||||
|
||||
|
||||
def _routed_model_cost() -> tuple[float, float]:
|
||||
routed_info: Final = get_model_info(model=ROUTED_MODEL, custom_llm_provider="azure_ai")
|
||||
return (
|
||||
ROUTED_USAGE.prompt_tokens * (routed_info["input_cost_per_token"] or 0.0),
|
||||
ROUTED_USAGE.completion_tokens * (routed_info["output_cost_per_token"] or 0.0),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
class TestAzureModelRouterFlatCost:
|
||||
"""Test Azure AI Foundry Model Router flat cost calculation."""
|
||||
"""cost_per_token charges the router fee once, for whichever router name the caller gives it."""
|
||||
|
||||
def test_model_router_flat_cost_basic(self):
|
||||
"""Test that flat cost is added for Model Router requests."""
|
||||
model = "azure-model-router"
|
||||
usage = Usage(
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=500,
|
||||
total_tokens=1500,
|
||||
)
|
||||
def test_unmapped_router_deployment_name_prices_the_fee(self) -> None:
|
||||
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
|
||||
prompt_cost, completion_cost_usd = cost_per_token(model="azure-model-router", usage=usage)
|
||||
assert prompt_cost == pytest.approx(1000 * ROUTER_FEE_PER_TOKEN, rel=1e-9)
|
||||
assert completion_cost_usd == 0.0
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
|
||||
|
||||
# Calculate expected flat cost
|
||||
expected_flat_cost = (
|
||||
usage.prompt_tokens
|
||||
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
|
||||
/ 1_000_000
|
||||
)
|
||||
|
||||
# Flat cost should be $0.00014 (1000 tokens × $0.14 / 1M tokens)
|
||||
assert expected_flat_cost == pytest.approx(0.00014, rel=1e-9)
|
||||
|
||||
# Prompt cost should include the flat cost
|
||||
# (plus any base cost from the actual model used, which might be 0 if not in model_cost)
|
||||
assert prompt_cost >= expected_flat_cost
|
||||
print(
|
||||
f"Model Router flat cost for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
|
||||
)
|
||||
print(f"Total prompt cost: ${prompt_cost:.6f}")
|
||||
|
||||
def test_model_router_flat_cost_large_request(self):
|
||||
"""Test flat cost calculation for larger requests."""
|
||||
model = "model-router"
|
||||
usage = Usage(
|
||||
prompt_tokens=100_000,
|
||||
completion_tokens=50_000,
|
||||
total_tokens=150_000,
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
|
||||
|
||||
# Calculate expected flat cost
|
||||
expected_flat_cost = (
|
||||
usage.prompt_tokens
|
||||
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
|
||||
/ 1_000_000
|
||||
)
|
||||
|
||||
# Flat cost should be $0.014 (100k tokens × $0.14 / 1M tokens)
|
||||
assert expected_flat_cost == pytest.approx(0.014, rel=1e-9)
|
||||
# Use approx for floating-point comparison
|
||||
assert prompt_cost >= expected_flat_cost or prompt_cost == pytest.approx(
|
||||
expected_flat_cost, rel=1e-9
|
||||
)
|
||||
print(
|
||||
f"Model Router flat cost for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
|
||||
)
|
||||
print(f"Total prompt cost: ${prompt_cost:.6f}")
|
||||
|
||||
def test_model_router_flat_cost_1m_tokens(self):
|
||||
"""Test flat cost for exactly 1 million input tokens."""
|
||||
model = "azure-model-router"
|
||||
usage = Usage(
|
||||
prompt_tokens=1_000_000,
|
||||
completion_tokens=100_000,
|
||||
total_tokens=1_100_000,
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
|
||||
|
||||
# Calculate expected flat cost
|
||||
expected_flat_cost = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
|
||||
|
||||
# Flat cost should be exactly $0.14 for 1M tokens
|
||||
assert expected_flat_cost == pytest.approx(0.14, rel=1e-9)
|
||||
assert prompt_cost >= expected_flat_cost
|
||||
print(f"Model Router flat cost for 1M tokens: ${expected_flat_cost:.6f}")
|
||||
print(f"Total prompt cost: ${prompt_cost:.6f}")
|
||||
|
||||
def test_non_model_router_no_flat_cost(self):
|
||||
"""Test that non-Model Router models don't get the flat cost."""
|
||||
model = "gpt-4o"
|
||||
usage = Usage(
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=500,
|
||||
total_tokens=1500,
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
|
||||
|
||||
# No flat cost should be added for non-Model Router models
|
||||
# The cost might be 0 or based on the model's pricing
|
||||
print(f"Non-Model Router prompt cost: ${prompt_cost:.6f}")
|
||||
# We just ensure it doesn't crash and returns valid values
|
||||
assert prompt_cost >= 0
|
||||
assert completion_cost >= 0
|
||||
|
||||
def test_model_router_with_cached_tokens(self):
|
||||
"""Test Model Router flat cost with cached tokens."""
|
||||
model = "azure-model-router"
|
||||
def test_unmapped_router_deployment_name_charges_the_fee_over_cached_prompt_tokens_too(self) -> None:
|
||||
usage = Usage(
|
||||
prompt_tokens=2000,
|
||||
completion_tokens=800,
|
||||
|
|
@ -189,268 +146,165 @@ class TestAzureModelRouterFlatCost:
|
|||
cache_read_input_tokens=500,
|
||||
cache_creation_input_tokens=200,
|
||||
)
|
||||
prompt_cost, completion_cost_usd = cost_per_token(model="azure-model-router", usage=usage)
|
||||
assert prompt_cost == pytest.approx(2000 * ROUTER_FEE_PER_TOKEN, rel=1e-9)
|
||||
assert completion_cost_usd == 0.0
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=model, usage=usage)
|
||||
|
||||
# Flat cost is based on ALL prompt tokens (including cached)
|
||||
expected_flat_cost = (
|
||||
usage.prompt_tokens
|
||||
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
|
||||
/ 1_000_000
|
||||
def test_router_deployment_name_as_both_names_charges_the_fee_once(self) -> None:
|
||||
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
|
||||
prompt_cost, completion_cost_usd = cost_per_token(
|
||||
model="model_router/my-deployment", usage=usage, request_model="azure_ai/model_router/my-deployment"
|
||||
)
|
||||
assert prompt_cost == pytest.approx(1000 * ROUTER_FEE_PER_TOKEN, rel=1e-9)
|
||||
assert completion_cost_usd == 0.0
|
||||
|
||||
assert expected_flat_cost == pytest.approx(0.00028, rel=1e-9)
|
||||
assert prompt_cost >= expected_flat_cost
|
||||
print(
|
||||
f"Model Router flat cost with caching for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}"
|
||||
@pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
|
||||
def test_router_entry_prices_its_own_fee(self, router_entry_name: str) -> None:
|
||||
usage = Usage(prompt_tokens=1_000_000, completion_tokens=0, total_tokens=1_000_000)
|
||||
prompt_cost, completion_cost_usd = cost_per_token(model=router_entry_name, usage=usage)
|
||||
assert prompt_cost == pytest.approx(0.14, rel=1e-9)
|
||||
assert completion_cost_usd == 0.0
|
||||
|
||||
def test_routed_model_is_priced_as_itself(self) -> None:
|
||||
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
|
||||
prompt_cost, completion_cost_usd = cost_per_token(model=ROUTED_MODEL, usage=ROUTED_USAGE)
|
||||
assert routed_prompt_cost > 0
|
||||
assert prompt_cost == pytest.approx(routed_prompt_cost, rel=1e-9)
|
||||
assert completion_cost_usd == pytest.approx(routed_completion_cost, rel=1e-9)
|
||||
|
||||
def test_unmapped_model_that_is_not_a_router_name_raises(self) -> None:
|
||||
usage = Usage(prompt_tokens=10, completion_tokens=10, total_tokens=20)
|
||||
with pytest.raises(Exception, match="no-such-azure-ai-model"):
|
||||
cost_per_token(model="no-such-azure-ai-model", usage=usage)
|
||||
|
||||
def test_request_model_through_the_router_adds_the_fee_once(self) -> None:
|
||||
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
|
||||
prompt_cost, completion_cost_usd = cost_per_token(
|
||||
model=ROUTED_MODEL, usage=ROUTED_USAGE, request_model="azure_ai/model-router"
|
||||
)
|
||||
print(f"Total prompt cost: ${prompt_cost:.6f}")
|
||||
assert prompt_cost == pytest.approx(routed_prompt_cost + ROUTED_FEE, rel=1e-9)
|
||||
assert completion_cost_usd == pytest.approx(routed_completion_cost, rel=1e-9)
|
||||
|
||||
def test_router_flat_cost_when_response_has_actual_model(self):
|
||||
"""
|
||||
Test that router flat cost is added when request was via router but response
|
||||
contains the actual model (e.g., gpt-5-nano).
|
||||
def test_request_model_that_is_not_the_router_adds_nothing(self) -> None:
|
||||
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
|
||||
assert cost_per_token(
|
||||
model=ROUTED_MODEL, usage=ROUTED_USAGE, request_model=f"azure_ai/{ROUTED_MODEL}"
|
||||
) == pytest.approx((routed_prompt_cost, routed_completion_cost), rel=1e-9)
|
||||
|
||||
This is the key fix: Azure returns the actual model in the response, but we
|
||||
must still add the router flat cost because the request was made via model router.
|
||||
"""
|
||||
usage = Usage(
|
||||
prompt_tokens=10000,
|
||||
completion_tokens=5000,
|
||||
total_tokens=15000,
|
||||
@pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
|
||||
def test_request_model_does_not_double_the_router_entry(self, router_entry_name: str) -> None:
|
||||
prompt_cost, completion_cost_usd = cost_per_token(
|
||||
model=router_entry_name, usage=ROUTED_USAGE, request_model=f"azure_ai/{router_entry_name}"
|
||||
)
|
||||
assert prompt_cost == pytest.approx(ROUTED_FEE, rel=1e-9)
|
||||
assert completion_cost_usd == 0.0
|
||||
|
||||
# Response model is the actual model Azure used (not a router name)
|
||||
response_model = "gpt-5-nano-2025-08-07"
|
||||
# Request model is the router - user called azure_ai/model_router/model-router
|
||||
request_model = "azure_ai/model_router/model-router"
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model=response_model,
|
||||
usage=usage,
|
||||
request_model=request_model,
|
||||
def test_public_cost_per_token_keeps_the_request_model_keyword(self) -> None:
|
||||
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
|
||||
prompt_cost, completion_cost_usd = litellm.cost_per_token(
|
||||
model=ROUTED_MODEL,
|
||||
custom_llm_provider="azure_ai",
|
||||
usage_object=ROUTED_USAGE,
|
||||
request_model="azure_ai/model-router",
|
||||
)
|
||||
assert prompt_cost == pytest.approx(routed_prompt_cost + ROUTED_FEE, rel=1e-9)
|
||||
assert completion_cost_usd == pytest.approx(routed_completion_cost, rel=1e-9)
|
||||
|
||||
# Expected: model cost (from gpt-5-nano) + router flat cost
|
||||
expected_flat_cost = (
|
||||
usage.prompt_tokens
|
||||
* AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS
|
||||
/ 1_000_000
|
||||
def test_flat_cost_helper(self) -> None:
|
||||
assert calculate_azure_model_router_flat_cost(
|
||||
model="azure-model-router", prompt_tokens=10_000
|
||||
) == pytest.approx(0.0014, rel=1e-9)
|
||||
assert calculate_azure_model_router_flat_cost(model="gpt-5-nano", prompt_tokens=10_000) == 0.0
|
||||
|
||||
def test_flat_cost_reads_the_fee_from_the_deployment_named_entry(self) -> None:
|
||||
litellm.register_model(
|
||||
{"azure_ai/model-router": {"input_cost_per_token": 2e-07, "litellm_provider": "azure_ai", "mode": "chat"}}
|
||||
)
|
||||
assert expected_flat_cost == pytest.approx(0.0014, rel=1e-9)
|
||||
|
||||
# Total cost should be model cost + flat cost
|
||||
total_cost = prompt_cost + completion_cost
|
||||
assert total_cost >= expected_flat_cost
|
||||
|
||||
# Prompt cost should include both model prompt cost and router flat cost
|
||||
assert prompt_cost >= expected_flat_cost
|
||||
litellm.get_model_info.cache_clear()
|
||||
assert calculate_azure_model_router_flat_cost(model="model-router", prompt_tokens=1_000_000) == pytest.approx(
|
||||
0.2, rel=1e-9
|
||||
)
|
||||
assert calculate_azure_model_router_flat_cost(
|
||||
model="azure-model-router", prompt_tokens=1_000_000
|
||||
) == pytest.approx(0.14, rel=1e-9)
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
class TestAzureModelRouterCostBreakdown:
|
||||
"""Test that Azure Model Router flat cost is tracked in cost breakdown."""
|
||||
"""completion_cost charges the router fee exactly once: as the breakdown's additional cost line when a routed
|
||||
model is priced as itself, inside the input cost when the priced name is the router."""
|
||||
|
||||
def test_flat_cost_calculation_helper(self):
|
||||
"""Test that flat cost can be calculated using the helper function."""
|
||||
from litellm.llms.azure_ai.cost_calculator import (
|
||||
calculate_azure_model_router_flat_cost,
|
||||
)
|
||||
|
||||
model = "azure-model-router"
|
||||
prompt_tokens = 10000
|
||||
|
||||
# Calculate flat cost using helper function
|
||||
flat_cost = calculate_azure_model_router_flat_cost(
|
||||
model=model, prompt_tokens=prompt_tokens
|
||||
)
|
||||
|
||||
# Expected flat cost
|
||||
expected_flat_cost = (
|
||||
prompt_tokens * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
|
||||
)
|
||||
|
||||
assert flat_cost > 0
|
||||
assert flat_cost == pytest.approx(expected_flat_cost, rel=1e-9)
|
||||
print(f"Flat cost calculated: ${flat_cost:.6f}")
|
||||
|
||||
def test_flat_cost_integration_with_completion_cost(self):
|
||||
"""Test that flat cost is properly integrated into completion_cost calculation."""
|
||||
import litellm
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
||||
|
||||
# Create a mock response for azure_ai model router
|
||||
response = ModelResponse(
|
||||
id="test-123",
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
message=Message(
|
||||
role="assistant",
|
||||
content="Test response",
|
||||
),
|
||||
)
|
||||
],
|
||||
created=1234567890,
|
||||
model="azure-model-router",
|
||||
object="chat.completion",
|
||||
usage=Usage(
|
||||
prompt_tokens=5000,
|
||||
completion_tokens=2000,
|
||||
total_tokens=7000,
|
||||
),
|
||||
)
|
||||
|
||||
# Set hidden params for provider
|
||||
response._hidden_params = {"custom_llm_provider": "azure_ai"}
|
||||
|
||||
# Calculate cost
|
||||
def test_unmapped_router_deployment_name_costs_only_the_fee(self) -> None:
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
completion_response=_azure_ai_response("azure-model-router"),
|
||||
model="azure-model-router",
|
||||
custom_llm_provider="azure_ai",
|
||||
)
|
||||
assert cost == pytest.approx(ROUTED_FEE, rel=1e-9)
|
||||
|
||||
# Expected flat cost
|
||||
expected_flat_cost = (
|
||||
5000 * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
|
||||
)
|
||||
|
||||
# Cost should include the flat cost (use approx for floating-point comparison)
|
||||
assert cost >= expected_flat_cost or cost == pytest.approx(
|
||||
expected_flat_cost, rel=1e-9
|
||||
)
|
||||
print(f"Total cost with flat fee: ${cost:.6f}")
|
||||
print(f"Expected minimum flat cost: ${expected_flat_cost:.6f}")
|
||||
|
||||
def test_additional_costs_in_cost_breakdown(self):
|
||||
"""Test that Azure Model Router flat cost appears in additional_costs dict."""
|
||||
from datetime import datetime
|
||||
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
||||
|
||||
# Create logging object with required parameters
|
||||
logging_obj = Logging(
|
||||
model="azure-model-router",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
stream=False,
|
||||
call_type="completion",
|
||||
start_time=datetime.now(),
|
||||
litellm_call_id="test-123",
|
||||
function_id="test-function",
|
||||
)
|
||||
|
||||
# Create a mock response for azure_ai model router
|
||||
response = ModelResponse(
|
||||
id="test-123",
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
message=Message(
|
||||
role="assistant",
|
||||
content="Test response",
|
||||
),
|
||||
)
|
||||
],
|
||||
created=1234567890,
|
||||
model="azure-model-router",
|
||||
object="chat.completion",
|
||||
usage=Usage(
|
||||
prompt_tokens=5000,
|
||||
completion_tokens=2000,
|
||||
total_tokens=7000,
|
||||
),
|
||||
)
|
||||
|
||||
# Set hidden params for provider
|
||||
response._hidden_params = {"custom_llm_provider": "azure_ai"}
|
||||
|
||||
# Calculate cost with logging object
|
||||
def test_unmapped_router_name_carries_the_fee_as_its_input_cost(self) -> None:
|
||||
logging_obj = _router_logging("azure-model-router")
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
completion_response=_azure_ai_response("azure-model-router"),
|
||||
model="azure-model-router",
|
||||
custom_llm_provider="azure_ai",
|
||||
litellm_logging_obj=logging_obj,
|
||||
)
|
||||
breakdown = logging_obj.cost_breakdown
|
||||
assert breakdown is not None
|
||||
assert breakdown["input_cost"] == pytest.approx(ROUTED_FEE, rel=1e-9)
|
||||
assert "additional_costs" not in breakdown
|
||||
assert cost == pytest.approx(ROUTED_FEE, rel=1e-9)
|
||||
|
||||
# Check that cost breakdown contains additional_costs
|
||||
assert hasattr(logging_obj, "cost_breakdown")
|
||||
assert logging_obj.cost_breakdown is not None
|
||||
assert "additional_costs" in logging_obj.cost_breakdown
|
||||
assert isinstance(logging_obj.cost_breakdown["additional_costs"], dict)
|
||||
|
||||
# Check that the Azure Model Router flat cost is in additional_costs
|
||||
additional_costs = logging_obj.cost_breakdown["additional_costs"]
|
||||
assert "Azure Model Router Flat Cost" in additional_costs
|
||||
|
||||
# Verify the flat cost value
|
||||
expected_flat_cost = (
|
||||
5000 * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
|
||||
)
|
||||
actual_flat_cost = additional_costs["Azure Model Router Flat Cost"]
|
||||
assert actual_flat_cost == pytest.approx(expected_flat_cost, rel=1e-9)
|
||||
|
||||
print(f"Additional costs in breakdown: {additional_costs}")
|
||||
print(f"Azure Model Router Flat Cost: ${actual_flat_cost:.6f}")
|
||||
|
||||
def test_additional_costs_when_response_has_actual_model_via_hidden_params(self):
|
||||
"""additional_costs populated when response has actual model but request was via model router (hidden_params)."""
|
||||
from datetime import datetime
|
||||
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
||||
|
||||
logging_obj = Logging(
|
||||
model="gpt-4.1-nano-2025-04-14",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
stream=False,
|
||||
call_type="completion",
|
||||
start_time=datetime.now(),
|
||||
litellm_call_id="test-123",
|
||||
function_id="test-function",
|
||||
)
|
||||
response = ModelResponse(
|
||||
id="test-123",
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
message=Message(role="assistant", content="Hello"),
|
||||
)
|
||||
],
|
||||
created=1234567890,
|
||||
model="gpt-4.1-nano-2025-04-14",
|
||||
object="chat.completion",
|
||||
usage=Usage(prompt_tokens=5000, completion_tokens=2000, total_tokens=7000),
|
||||
)
|
||||
response._hidden_params = {
|
||||
"custom_llm_provider": "azure_ai",
|
||||
"litellm_model_name": "azure_ai/model-router",
|
||||
}
|
||||
def test_router_request_with_routed_response_charges_the_fee_once(self) -> None:
|
||||
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
|
||||
logging_obj = _router_logging("model-router")
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model="gpt-4.1-nano-2025-04-14",
|
||||
completion_response=_azure_ai_response(ROUTED_MODEL),
|
||||
model=ROUTED_MODEL,
|
||||
custom_llm_provider="azure_ai",
|
||||
litellm_logging_obj=logging_obj,
|
||||
)
|
||||
expected_flat_cost = (
|
||||
5000 * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
|
||||
breakdown = logging_obj.cost_breakdown
|
||||
assert breakdown is not None
|
||||
assert breakdown["input_cost"] == pytest.approx(routed_prompt_cost, rel=1e-9)
|
||||
assert breakdown["output_cost"] == pytest.approx(routed_completion_cost, rel=1e-9)
|
||||
assert breakdown.get("additional_costs") == pytest.approx(
|
||||
{"Azure Model Router Flat Cost": ROUTED_FEE}, rel=1e-9
|
||||
)
|
||||
assert cost >= expected_flat_cost
|
||||
assert logging_obj.cost_breakdown is not None
|
||||
assert "additional_costs" in logging_obj.cost_breakdown
|
||||
assert (
|
||||
"Azure Model Router Flat Cost"
|
||||
in logging_obj.cost_breakdown["additional_costs"]
|
||||
assert cost == pytest.approx(routed_prompt_cost + routed_completion_cost + ROUTED_FEE, rel=1e-9)
|
||||
|
||||
def test_routed_response_named_by_hidden_params_charges_the_fee_once(self) -> None:
|
||||
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
|
||||
logging_obj = _router_logging(ROUTED_MODEL)
|
||||
cost = completion_cost(
|
||||
completion_response=_azure_ai_response(ROUTED_MODEL, litellm_model_name="azure_ai/model-router"),
|
||||
model=ROUTED_MODEL,
|
||||
custom_llm_provider="azure_ai",
|
||||
litellm_logging_obj=logging_obj,
|
||||
)
|
||||
assert logging_obj.cost_breakdown["additional_costs"][
|
||||
"Azure Model Router Flat Cost"
|
||||
] == pytest.approx(expected_flat_cost, rel=1e-9)
|
||||
breakdown = logging_obj.cost_breakdown
|
||||
assert breakdown is not None
|
||||
assert breakdown["input_cost"] == pytest.approx(routed_prompt_cost, rel=1e-9)
|
||||
assert breakdown.get("additional_costs") == pytest.approx(
|
||||
{"Azure Model Router Flat Cost": ROUTED_FEE}, rel=1e-9
|
||||
)
|
||||
assert cost == pytest.approx(routed_prompt_cost + routed_completion_cost + ROUTED_FEE, rel=1e-9)
|
||||
|
||||
@pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
|
||||
def test_response_priced_as_the_router_entry_charges_the_fee_once(self, router_entry_name: str) -> None:
|
||||
logging_obj = _router_logging(router_entry_name)
|
||||
cost = completion_cost(
|
||||
completion_response=_azure_ai_response(router_entry_name),
|
||||
model=router_entry_name,
|
||||
custom_llm_provider="azure_ai",
|
||||
litellm_logging_obj=logging_obj,
|
||||
)
|
||||
breakdown = logging_obj.cost_breakdown
|
||||
assert breakdown is not None
|
||||
assert "additional_costs" not in breakdown
|
||||
assert breakdown["input_cost"] == pytest.approx(ROUTED_FEE, rel=1e-9)
|
||||
assert cost == pytest.approx(ROUTED_FEE, rel=1e-9)
|
||||
|
||||
|
||||
class TestAzureAIServiceTierCostCalculation:
|
||||
|
|
@ -459,26 +313,27 @@ class TestAzureAIServiceTierCostCalculation:
|
|||
@pytest.fixture(autouse=True)
|
||||
def register_test_model(self):
|
||||
import litellm
|
||||
litellm.register_model(model_cost={
|
||||
"test-azure-ai-model": {
|
||||
"input_cost_per_token": 0.001,
|
||||
"output_cost_per_token": 0.002,
|
||||
"input_cost_per_token_priority": 0.01,
|
||||
"output_cost_per_token_priority": 0.02,
|
||||
"input_cost_per_token_flex": 0.0005,
|
||||
"output_cost_per_token_flex": 0.001,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_tokens": 8192,
|
||||
|
||||
litellm.register_model(
|
||||
model_cost={
|
||||
"test-azure-ai-model": {
|
||||
"input_cost_per_token": 0.001,
|
||||
"output_cost_per_token": 0.002,
|
||||
"input_cost_per_token_priority": 0.01,
|
||||
"output_cost_per_token_priority": 0.02,
|
||||
"input_cost_per_token_flex": 0.0005,
|
||||
"output_cost_per_token_flex": 0.001,
|
||||
"litellm_provider": "azure_ai",
|
||||
"max_tokens": 8192,
|
||||
}
|
||||
}
|
||||
})
|
||||
)
|
||||
|
||||
def test_service_tier_priority_higher_cost(self):
|
||||
"""Priority tier should cost more than standard for azure_ai."""
|
||||
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
|
||||
|
||||
standard_prompt, standard_completion = cost_per_token(
|
||||
model="test-azure-ai-model", usage=usage
|
||||
)
|
||||
standard_prompt, standard_completion = cost_per_token(model="test-azure-ai-model", usage=usage)
|
||||
priority_prompt, priority_completion = cost_per_token(
|
||||
model="test-azure-ai-model", usage=usage, service_tier="priority"
|
||||
)
|
||||
|
|
@ -490,12 +345,8 @@ class TestAzureAIServiceTierCostCalculation:
|
|||
"""Flex tier should cost less than standard for azure_ai."""
|
||||
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
|
||||
|
||||
standard_prompt, standard_completion = cost_per_token(
|
||||
model="test-azure-ai-model", usage=usage
|
||||
)
|
||||
flex_prompt, flex_completion = cost_per_token(
|
||||
model="test-azure-ai-model", usage=usage, service_tier="flex"
|
||||
)
|
||||
standard_prompt, standard_completion = cost_per_token(model="test-azure-ai-model", usage=usage)
|
||||
flex_prompt, flex_completion = cost_per_token(model="test-azure-ai-model", usage=usage, service_tier="flex")
|
||||
|
||||
assert flex_prompt < standard_prompt
|
||||
assert flex_completion < standard_completion
|
||||
|
|
|
|||
|
|
@ -0,0 +1,113 @@
|
|||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
from litellm import completion_cost, cost_per_token, get_model_info
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
||||
from litellm.types.utils import TranscriptionResponse
|
||||
|
||||
REPO_ROOT: Final = Path(__file__).parents[4]
|
||||
MAIN_COST_MAP: Final = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
BACKUP_COST_MAP: Final = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
|
||||
COST_MAP_ADAPTER: Final = TypeAdapter(dict[str, dict[str, object]])
|
||||
AZURE_PRICING_PREFIX: Final = "https://azure.microsoft.com/en-us/pricing/details/"
|
||||
A_MILLION: Final = 1_000_000
|
||||
AN_HOUR_IN_SECONDS: Final = 3600
|
||||
|
||||
TOKEN_PRICED_NAMES: Final = (
|
||||
"gpt-chat-latest",
|
||||
"codex-mini",
|
||||
"model-router",
|
||||
"cohere-command-a",
|
||||
"grok-4-20-reasoning",
|
||||
"grok-4-20-non-reasoning",
|
||||
)
|
||||
GROK_4_20_NAMES: Final = ("grok-4-20-reasoning", "grok-4-20-non-reasoning")
|
||||
CATALOG_NAMES: Final = TOKEN_PRICED_NAMES + ("whisper",)
|
||||
|
||||
|
||||
def _cost_map_entry(path: Path, catalog_name: str) -> dict[str, object]:
|
||||
return COST_MAP_ADAPTER.validate_json(path.read_bytes())[f"azure_ai/{catalog_name}"]
|
||||
|
||||
|
||||
def _whisper_transcription_cost(duration_seconds: int) -> float:
|
||||
transcription: Final = TranscriptionResponse(text="hello")
|
||||
transcription._hidden_params = { # pyright: ignore[reportPrivateUsage] # TranscriptionResponse exposes no public hidden-params setter
|
||||
"custom_llm_provider": "azure_ai",
|
||||
"model": "azure_ai/whisper",
|
||||
"audio_transcription_duration": duration_seconds,
|
||||
}
|
||||
return completion_cost(
|
||||
completion_response=transcription,
|
||||
model="azure_ai/whisper",
|
||||
custom_llm_provider="azure_ai",
|
||||
call_type="atranscription",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("catalog_name", CATALOG_NAMES)
|
||||
def test_azure_ai_catalog_name_routes_to_azure_ai(catalog_name: str) -> None:
|
||||
routed_model, provider, _, _ = get_llm_provider(model=f"azure_ai/{catalog_name}")
|
||||
assert (routed_model, provider) == (catalog_name, "azure_ai")
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
@pytest.mark.parametrize("catalog_name", TOKEN_PRICED_NAMES)
|
||||
def test_azure_ai_catalog_name_charges_its_own_entry_per_token(catalog_name: str) -> None:
|
||||
entry: Final = get_model_info(f"azure_ai/{catalog_name}")
|
||||
prompt_cost, completion_cost_usd = cost_per_token(
|
||||
model=f"azure_ai/{catalog_name}", prompt_tokens=A_MILLION, completion_tokens=A_MILLION
|
||||
)
|
||||
assert prompt_cost > 0
|
||||
assert prompt_cost == pytest.approx(A_MILLION * entry["input_cost_per_token"])
|
||||
assert completion_cost_usd == pytest.approx(A_MILLION * entry["output_cost_per_token"])
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
@pytest.mark.parametrize("catalog_name", TOKEN_PRICED_NAMES)
|
||||
def test_azure_ai_catalog_name_prices_the_same_in_any_casing(catalog_name: str) -> None:
|
||||
lowercase_cost = cost_per_token(model=f"azure_ai/{catalog_name}", prompt_tokens=A_MILLION, completion_tokens=0)
|
||||
upper_cost = cost_per_token(model=f"azure_ai/{catalog_name.upper()}", prompt_tokens=A_MILLION, completion_tokens=0)
|
||||
assert upper_cost == lowercase_cost
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
@pytest.mark.parametrize("catalog_name", GROK_4_20_NAMES)
|
||||
def test_azure_ai_grok_4_20_bills_cached_prompt_tokens_at_the_input_price(catalog_name: str) -> None:
|
||||
uncached_prompt_cost, _ = cost_per_token(model=f"azure_ai/{catalog_name}", prompt_tokens=A_MILLION, completion_tokens=0)
|
||||
cached_prompt_cost, _ = cost_per_token(
|
||||
model=f"azure_ai/{catalog_name}",
|
||||
prompt_tokens=A_MILLION,
|
||||
completion_tokens=0,
|
||||
cache_read_input_tokens=A_MILLION,
|
||||
)
|
||||
assert uncached_prompt_cost > 0
|
||||
assert cached_prompt_cost == pytest.approx(uncached_prompt_cost)
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
def test_azure_ai_whisper_catalog_name_is_priced_per_second() -> None:
|
||||
one_second_cost: Final = _whisper_transcription_cost(1)
|
||||
one_hour_cost: Final = _whisper_transcription_cost(AN_HOUR_IN_SECONDS)
|
||||
assert one_second_cost > 0
|
||||
assert one_hour_cost == pytest.approx(AN_HOUR_IN_SECONDS * one_second_cost)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("catalog_name", CATALOG_NAMES)
|
||||
def test_azure_ai_catalog_entry_source_and_backup_match(catalog_name: str) -> None:
|
||||
main_entry = _cost_map_entry(MAIN_COST_MAP, catalog_name)
|
||||
backup_entry = _cost_map_entry(BACKUP_COST_MAP, catalog_name)
|
||||
|
||||
assert str(main_entry["source"]).startswith(AZURE_PRICING_PREFIX)
|
||||
assert backup_entry == main_entry
|
||||
|
||||
|
||||
def test_azure_ai_model_router_spellings_share_one_entry() -> None:
|
||||
underscore_entry = _cost_map_entry(MAIN_COST_MAP, "model_router")
|
||||
hyphen_entry = _cost_map_entry(MAIN_COST_MAP, "model-router")
|
||||
|
||||
assert {k: v for k, v in underscore_entry.items() if k != "comment"} == {
|
||||
k: v for k, v in hyphen_entry.items() if k != "comment"
|
||||
}
|
||||
Loading…
Add table
Reference in a new issue