mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
Move Azure Model Router flat cost to model_prices_and_context_window.json
- Added azure_ai/azure-model-router entry in model_prices_and_context_window.json with --.14/M tokens flat cost - Updated cost_calculator.py to read flat cost from model info instead of hardcoded constant - Updated tests to read pricing from model info - Pricing is now centrally managed in the JSON configuration file Co-authored-by: ishaan <ishaan@berri.ai>
This commit is contained in:
parent
9f2f8a4f65
commit
d56f97495c
3 changed files with 29 additions and 17 deletions
|
|
@ -8,11 +8,7 @@ from typing import Optional, Tuple
|
|||
from litellm._logging import verbose_logger
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
|
||||
# Azure AI Foundry Model Router pricing
|
||||
# Source: https://azure.microsoft.com/en-us/pricing/details/ai-services/
|
||||
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = 0.14 # $0.14 per M input tokens
|
||||
from litellm.utils import get_model_info
|
||||
|
||||
|
||||
def _is_azure_model_router(model: str) -> bool:
|
||||
|
|
@ -36,7 +32,7 @@ def cost_per_token(
|
|||
Calculate the cost per token for Azure AI models.
|
||||
|
||||
For Azure AI Foundry Model Router:
|
||||
- Adds a flat cost of $0.14 per million input tokens
|
||||
- Adds a flat cost of $0.14 per million input tokens (from model_prices_and_context_window.json)
|
||||
- Plus the cost of the actual model used (handled by generic_cost_per_token)
|
||||
|
||||
Args:
|
||||
|
|
@ -55,17 +51,21 @@ def cost_per_token(
|
|||
)
|
||||
|
||||
# Add flat cost for Azure Model Router
|
||||
# The flat cost is defined in model_prices_and_context_window.json for azure_ai/azure-model-router
|
||||
if _is_azure_model_router(model):
|
||||
# Flat cost per million input tokens
|
||||
flat_cost_per_token = AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS / 1_000_000
|
||||
router_flat_cost = usage.prompt_tokens * flat_cost_per_token
|
||||
# Get the model router pricing from model_prices_and_context_window.json
|
||||
model_info = get_model_info(model="azure-model-router", custom_llm_provider="azure_ai")
|
||||
router_flat_cost_per_token = model_info.get("input_cost_per_token", 0)
|
||||
|
||||
verbose_logger.debug(
|
||||
f"Azure AI Model Router flat cost: ${router_flat_cost:.6f} "
|
||||
f"({usage.prompt_tokens} tokens × ${flat_cost_per_token:.9f}/token)"
|
||||
)
|
||||
|
||||
# Add flat cost to prompt cost
|
||||
prompt_cost += router_flat_cost
|
||||
if router_flat_cost_per_token > 0:
|
||||
router_flat_cost = usage.prompt_tokens * router_flat_cost_per_token
|
||||
|
||||
verbose_logger.debug(
|
||||
f"Azure AI Model Router flat cost: ${router_flat_cost:.6f} "
|
||||
f"({usage.prompt_tokens} tokens × ${router_flat_cost_per_token:.9f}/token)"
|
||||
)
|
||||
|
||||
# Add flat cost to prompt cost
|
||||
prompt_cost += router_flat_cost
|
||||
|
||||
return prompt_cost, completion_cost
|
||||
|
|
|
|||
|
|
@ -1517,6 +1517,14 @@
|
|||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true
|
||||
},
|
||||
"azure_ai/azure-model-router": {
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"output_cost_per_token": 0,
|
||||
"litellm_provider": "azure_ai",
|
||||
"mode": "chat",
|
||||
"source": "https://azure.microsoft.com/en-us/pricing/details/ai-services/",
|
||||
"comment": "Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure"
|
||||
},
|
||||
"azure/eu/gpt-4o-2024-08-06": {
|
||||
"deprecation_date": "2026-02-27",
|
||||
"cache_read_input_token_cost": 1.375e-06,
|
||||
|
|
|
|||
|
|
@ -4,11 +4,15 @@ Test Azure AI cost calculator, especially Model Router flat cost.
|
|||
|
||||
import pytest
|
||||
from litellm.llms.azure_ai.cost_calculator import (
|
||||
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS,
|
||||
_is_azure_model_router,
|
||||
cost_per_token,
|
||||
)
|
||||
from litellm.types.utils import Usage
|
||||
from litellm.utils import get_model_info
|
||||
|
||||
# Get the flat cost from model_prices_and_context_window.json
|
||||
_model_info = get_model_info(model="azure-model-router", custom_llm_provider="azure_ai")
|
||||
AZURE_MODEL_ROUTER_FLAT_COST_PER_M_INPUT_TOKENS = _model_info.get("input_cost_per_token", 0) * 1_000_000
|
||||
|
||||
|
||||
class TestAzureModelRouterDetection:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue