diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 490f0288b00..a3ea03e6da0 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -36,6 +36,9 @@ from litellm.llms.anthropic.cost_calculation import ( from litellm.llms.azure.cost_calculation import ( cost_per_token as azure_openai_cost_per_token, ) +from litellm.llms.azure_ai.cost_calculator import ( + cost_per_token as azure_ai_cost_per_token, +) from litellm.llms.base_llm.search.transformation import SearchResponse from litellm.llms.bedrock.cost_calculation import ( cost_per_token as bedrock_cost_per_token, @@ -427,8 +430,8 @@ def cost_per_token( # noqa: PLR0915 return dashscope_cost_per_token(model=model, usage=usage_block) elif custom_llm_provider == "azure_ai": - return generic_cost_per_token( - model=model, usage=usage_block, custom_llm_provider=custom_llm_provider + return azure_ai_cost_per_token( + model=model, usage=usage_block, response_time_ms=response_time_ms ) else: model_info = _cached_get_model_info_helper( diff --git a/litellm/llms/azure_ai/cost_calculator.py b/litellm/llms/azure_ai/cost_calculator.py new file mode 100644 index 00000000000..1ffff0b40ba --- /dev/null +++ b/litellm/llms/azure_ai/cost_calculator.py @@ -0,0 +1,71 @@ +""" +Azure AI cost calculation helper. +Handles Azure AI Foundry Model Router flat cost and other Azure AI specific pricing. +""" + +from typing import Optional, Tuple + +from litellm._logging import verbose_logger +from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token +from litellm.types.utils import Usage + + +# Azure AI Foundry Model Router pricing +# Source: https://azure.microsoft.com/en-us/pricing/details/ai-services/ +AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = 0.14 # $0.14 per M input tokens + + +def _is_azure_model_router(model: str) -> bool: + """ + Check if the model is Azure AI Foundry Model Router. + + Args: + model: The model name (e.g., "azure-model-router", "model-router") + + Returns: + bool: True if this is a model router model + """ + model_lower = model.lower() + return "model-router" in model_lower or model_lower == "azure-model-router" + + +def cost_per_token( + model: str, usage: Usage, response_time_ms: Optional[float] = 0.0 +) -> Tuple[float, float]: + """ + Calculate the cost per token for Azure AI models. + + For Azure AI Foundry Model Router: + - Adds a flat cost of $0.14 per million input tokens + - Plus the cost of the actual model used (handled by generic_cost_per_token) + + Args: + model: str, the model name without provider prefix + usage: LiteLLM Usage block + response_time_ms: Optional response time in milliseconds + + Returns: + Tuple[float, float] - prompt_cost_in_usd, completion_cost_in_usd + """ + # Calculate base cost using generic cost calculator + prompt_cost, completion_cost = generic_cost_per_token( + model=model, + usage=usage, + custom_llm_provider="azure_ai", + ) + + # Add flat cost for Azure Model Router + if _is_azure_model_router(model): + # Flat cost per million input tokens + flat_cost_per_token = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000 + router_flat_cost = usage.prompt_tokens * flat_cost_per_token + + verbose_logger.debug( + f"Azure AI Model Router flat cost: ${router_flat_cost:.6f} " + f"({usage.prompt_tokens} tokens × ${flat_cost_per_token:.9f}/token)" + ) + + # Add flat cost to prompt cost + prompt_cost += router_flat_cost + + return prompt_cost, completion_cost diff --git a/tests/llm_translation/test_azure_ai.py b/tests/llm_translation/test_azure_ai.py index 1633fb4cc82..77af31b8ef3 100644 --- a/tests/llm_translation/test_azure_ai.py +++ b/tests/llm_translation/test_azure_ai.py @@ -373,6 +373,7 @@ def test_completion_azure_ai_gpt_4o_with_flexible_api_base(api_base): async def test_azure_ai_model_router(): """ Test Azure AI model router non-streaming response cost tracking. + Verifies that the flat cost of $0.14 per M input tokens is applied. """ litellm._turn_on_debug() response = await litellm.acompletion( @@ -387,6 +388,18 @@ async def test_azure_ai_model_router(): tracked_cost = response._hidden_params["response_cost"] assert tracked_cost > 0 print("Tracked cost: ", tracked_cost) + + # Verify flat cost is included + # Flat cost = prompt_tokens * $0.14 / 1M = prompt_tokens * 0.00000014 + usage = response.usage + if usage and usage.prompt_tokens: + expected_flat_cost = usage.prompt_tokens * 0.14 / 1_000_000 + print(f"Prompt tokens: {usage.prompt_tokens}") + print(f"Expected minimum flat cost: ${expected_flat_cost:.9f}") + # Total cost should be at least the flat cost + assert tracked_cost >= expected_flat_cost, ( + f"Cost ${tracked_cost:.9f} should be >= flat cost ${expected_flat_cost:.9f}" + ) @pytest.mark.asyncio diff --git a/tests/test_litellm/llms/azure_ai/test_cost_calculator.py b/tests/test_litellm/llms/azure_ai/test_cost_calculator.py new file mode 100644 index 00000000000..0ad9b81c756 --- /dev/null +++ b/tests/test_litellm/llms/azure_ai/test_cost_calculator.py @@ -0,0 +1,150 @@ +""" +Test Azure AI cost calculator, especially Model Router flat cost. +""" + +import pytest +from litellm.llms.azure_ai.cost_calculator import ( + AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS, + _is_azure_model_router, + cost_per_token, +) +from litellm.types.utils import Usage + + +class TestAzureModelRouterDetection: + """Test that we correctly identify Azure Model Router models.""" + + @pytest.mark.parametrize( + "model,expected", + [ + ("azure-model-router", True), + ("AZURE-MODEL-ROUTER", True), + ("model-router", True), + ("MODEL-ROUTER", True), + ("gpt-4o-mini-2024-07-18-model-router", True), + ("gpt-4o", False), + ("gpt-4o-mini", False), + ("claude-sonnet-4-5", False), + ], + ) + def test_is_azure_model_router(self, model: str, expected: bool): + """Test Azure Model Router detection.""" + assert _is_azure_model_router(model) == expected + + +class TestAzureModelRouterFlatCost: + """Test Azure AI Foundry Model Router flat cost calculation.""" + + def test_model_router_flat_cost_basic(self): + """Test that flat cost is added for Model Router requests.""" + model = "azure-model-router" + usage = Usage( + prompt_tokens=1000, + completion_tokens=500, + total_tokens=1500, + ) + + prompt_cost, completion_cost = cost_per_token(model=model, usage=usage) + + # Calculate expected flat cost + expected_flat_cost = ( + usage.prompt_tokens * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000 + ) + + # Flat cost should be $0.00014 (1000 tokens × $0.14 / 1M tokens) + assert expected_flat_cost == pytest.approx(0.00014, rel=1e-9) + + # Prompt cost should include the flat cost + # (plus any base cost from the actual model used, which might be 0 if not in model_cost) + assert prompt_cost >= expected_flat_cost + print( + f"Model Router flat cost for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}" + ) + print(f"Total prompt cost: ${prompt_cost:.6f}") + + def test_model_router_flat_cost_large_request(self): + """Test flat cost calculation for larger requests.""" + model = "model-router" + usage = Usage( + prompt_tokens=100_000, + completion_tokens=50_000, + total_tokens=150_000, + ) + + prompt_cost, completion_cost = cost_per_token(model=model, usage=usage) + + # Calculate expected flat cost + expected_flat_cost = ( + usage.prompt_tokens * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000 + ) + + # Flat cost should be $0.014 (100k tokens × $0.14 / 1M tokens) + assert expected_flat_cost == pytest.approx(0.014, rel=1e-9) + assert prompt_cost >= expected_flat_cost + print( + f"Model Router flat cost for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}" + ) + print(f"Total prompt cost: ${prompt_cost:.6f}") + + def test_model_router_flat_cost_1m_tokens(self): + """Test flat cost for exactly 1 million input tokens.""" + model = "azure-model-router" + usage = Usage( + prompt_tokens=1_000_000, + completion_tokens=100_000, + total_tokens=1_100_000, + ) + + prompt_cost, completion_cost = cost_per_token(model=model, usage=usage) + + # Calculate expected flat cost + expected_flat_cost = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS + + # Flat cost should be exactly $0.14 for 1M tokens + assert expected_flat_cost == pytest.approx(0.14, rel=1e-9) + assert prompt_cost >= expected_flat_cost + print(f"Model Router flat cost for 1M tokens: ${expected_flat_cost:.6f}") + print(f"Total prompt cost: ${prompt_cost:.6f}") + + def test_non_model_router_no_flat_cost(self): + """Test that non-Model Router models don't get the flat cost.""" + model = "gpt-4o" + usage = Usage( + prompt_tokens=1000, + completion_tokens=500, + total_tokens=1500, + ) + + prompt_cost, completion_cost = cost_per_token(model=model, usage=usage) + + # No flat cost should be added for non-Model Router models + # The cost might be 0 or based on the model's pricing + print(f"Non-Model Router prompt cost: ${prompt_cost:.6f}") + # We just ensure it doesn't crash and returns valid values + assert prompt_cost >= 0 + assert completion_cost >= 0 + + def test_model_router_with_cached_tokens(self): + """Test Model Router flat cost with cached tokens.""" + model = "azure-model-router" + usage = Usage( + prompt_tokens=2000, + completion_tokens=800, + total_tokens=2800, + cache_read_input_tokens=500, + cache_creation_input_tokens=200, + ) + + prompt_cost, completion_cost = cost_per_token(model=model, usage=usage) + + # Flat cost is based on ALL prompt tokens (including cached) + expected_flat_cost = ( + usage.prompt_tokens * AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000 + ) + + assert expected_flat_cost == pytest.approx(0.00028, rel=1e-9) + assert prompt_cost >= expected_flat_cost + print( + f"Model Router flat cost with caching for {usage.prompt_tokens} tokens: ${expected_flat_cost:.6f}" + ) + print(f"Total prompt cost: ${prompt_cost:.6f}")