mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(model_prices): gemini deep research input limits, vertex flash retirement dates, azure data zone gpt-6.1-sol pricing (#44633)
This commit is contained in:
parent
46d2c2a6ea
commit
126e79c967
3 changed files with 356 additions and 4 deletions
|
|
@ -8681,6 +8681,153 @@
|
|||
"supports_web_search": true,
|
||||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"azure/us/gpt-6.1-sol": {
|
||||
"cache_creation_input_token_cost": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-06,
|
||||
"cache_read_input_token_cost": 1.1e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 2.2e-07,
|
||||
"deprecation_date": "2028-03-11",
|
||||
"input_cost_per_token": 2.2e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.1e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 1.65e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
"supports_native_streaming": true,
|
||||
"supports_none_reasoning_effort": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"azure/eu/gpt-6.1-sol": {
|
||||
"cache_creation_input_token_cost": 3e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 6e-06,
|
||||
"cache_read_input_token_cost": 1.2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 2.4e-07,
|
||||
"deprecation_date": "2028-03-11",
|
||||
"input_cost_per_token": 2.4e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.8e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 1.8e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
"supports_native_streaming": true,
|
||||
"supports_none_reasoning_effort": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"azure/apac/gpt-6.1-sol": {
|
||||
"cache_creation_input_token_cost": 3e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 6e-06,
|
||||
"cache_read_input_token_cost": 1.2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 2.4e-07,
|
||||
"deprecation_date": "2028-03-11",
|
||||
"input_cost_per_token": 2.4e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.8e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 1.8e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
"supports_native_streaming": true,
|
||||
"supports_none_reasoning_effort": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"azure/gpt-chat-latest": {
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"deprecation_date": "2026-12-02",
|
||||
|
|
@ -28136,6 +28283,7 @@
|
|||
"input_cost_per_token_priority": 1.35e-06,
|
||||
"output_cost_per_token_priority": 6.75e-06,
|
||||
"cache_read_input_token_cost_priority": 1.35e-07,
|
||||
"deprecation_date": "2026-11-19",
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_low": 0.014,
|
||||
"search_context_size_medium": 0.014,
|
||||
|
|
@ -28195,6 +28343,7 @@
|
|||
"input_cost_per_token_priority": 1.35e-06,
|
||||
"output_cost_per_token_priority": 6.75e-06,
|
||||
"cache_read_input_token_cost_priority": 1.35e-07,
|
||||
"deprecation_date": "2027-01-28",
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_low": 0.014,
|
||||
"search_context_size_medium": 0.014,
|
||||
|
|
@ -28928,7 +29077,7 @@
|
|||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_batches": 1e-06,
|
||||
"litellm_provider": "gemini",
|
||||
"max_input_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65536,
|
||||
"max_tokens": 65536,
|
||||
"mode": "chat",
|
||||
|
|
@ -28965,7 +29114,7 @@
|
|||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_batches": 1e-06,
|
||||
"litellm_provider": "gemini",
|
||||
"max_input_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65536,
|
||||
"max_tokens": 65536,
|
||||
"mode": "chat",
|
||||
|
|
@ -30410,6 +30559,7 @@
|
|||
"input_cost_per_token_priority": 1.35e-06,
|
||||
"output_cost_per_token_priority": 6.75e-06,
|
||||
"cache_read_input_token_cost_priority": 1.35e-07,
|
||||
"deprecation_date": "2026-11-19",
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_low": 0.014,
|
||||
"search_context_size_medium": 0.014,
|
||||
|
|
@ -30469,6 +30619,7 @@
|
|||
"input_cost_per_token_priority": 1.35e-06,
|
||||
"output_cost_per_token_priority": 6.75e-06,
|
||||
"cache_read_input_token_cost_priority": 1.35e-07,
|
||||
"deprecation_date": "2027-01-28",
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_low": 0.014,
|
||||
"search_context_size_medium": 0.014,
|
||||
|
|
|
|||
|
|
@ -8681,6 +8681,153 @@
|
|||
"supports_web_search": true,
|
||||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"azure/us/gpt-6.1-sol": {
|
||||
"cache_creation_input_token_cost": 2.75e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-06,
|
||||
"cache_read_input_token_cost": 1.1e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 2.2e-07,
|
||||
"deprecation_date": "2028-03-11",
|
||||
"input_cost_per_token": 2.2e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.1e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 1.65e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
"supports_native_streaming": true,
|
||||
"supports_none_reasoning_effort": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"azure/eu/gpt-6.1-sol": {
|
||||
"cache_creation_input_token_cost": 3e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 6e-06,
|
||||
"cache_read_input_token_cost": 1.2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 2.4e-07,
|
||||
"deprecation_date": "2028-03-11",
|
||||
"input_cost_per_token": 2.4e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.8e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 1.8e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
"supports_native_streaming": true,
|
||||
"supports_none_reasoning_effort": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"azure/apac/gpt-6.1-sol": {
|
||||
"cache_creation_input_token_cost": 3e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 6e-06,
|
||||
"cache_read_input_token_cost": 1.2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 2.4e-07,
|
||||
"deprecation_date": "2028-03-11",
|
||||
"input_cost_per_token": 2.4e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.8e-06,
|
||||
"litellm_provider": "azure",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.2e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 1.8e-05,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
"supports_native_streaming": true,
|
||||
"supports_none_reasoning_effort": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"azure/gpt-chat-latest": {
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"deprecation_date": "2026-12-02",
|
||||
|
|
@ -28136,6 +28283,7 @@
|
|||
"input_cost_per_token_priority": 1.35e-06,
|
||||
"output_cost_per_token_priority": 6.75e-06,
|
||||
"cache_read_input_token_cost_priority": 1.35e-07,
|
||||
"deprecation_date": "2026-11-19",
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_low": 0.014,
|
||||
"search_context_size_medium": 0.014,
|
||||
|
|
@ -28195,6 +28343,7 @@
|
|||
"input_cost_per_token_priority": 1.35e-06,
|
||||
"output_cost_per_token_priority": 6.75e-06,
|
||||
"cache_read_input_token_cost_priority": 1.35e-07,
|
||||
"deprecation_date": "2027-01-28",
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_low": 0.014,
|
||||
"search_context_size_medium": 0.014,
|
||||
|
|
@ -28928,7 +29077,7 @@
|
|||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_batches": 1e-06,
|
||||
"litellm_provider": "gemini",
|
||||
"max_input_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65536,
|
||||
"max_tokens": 65536,
|
||||
"mode": "chat",
|
||||
|
|
@ -28965,7 +29114,7 @@
|
|||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_batches": 1e-06,
|
||||
"litellm_provider": "gemini",
|
||||
"max_input_tokens": 131072,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 65536,
|
||||
"max_tokens": 65536,
|
||||
"mode": "chat",
|
||||
|
|
@ -30410,6 +30559,7 @@
|
|||
"input_cost_per_token_priority": 1.35e-06,
|
||||
"output_cost_per_token_priority": 6.75e-06,
|
||||
"cache_read_input_token_cost_priority": 1.35e-07,
|
||||
"deprecation_date": "2026-11-19",
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_low": 0.014,
|
||||
"search_context_size_medium": 0.014,
|
||||
|
|
@ -30469,6 +30619,7 @@
|
|||
"input_cost_per_token_priority": 1.35e-06,
|
||||
"output_cost_per_token_priority": 6.75e-06,
|
||||
"cache_read_input_token_cost_priority": 1.35e-07,
|
||||
"deprecation_date": "2027-01-28",
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_low": 0.014,
|
||||
"search_context_size_medium": 0.014,
|
||||
|
|
|
|||
|
|
@ -232,6 +232,56 @@ def test_dated_variants_carry_base_alias_service_tier_pricing(prices: dict):
|
|||
)
|
||||
|
||||
|
||||
# Azure Foundry GPT-6.1 Sol announcement, 2026-09-29: https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-6-1-sol-in-microsoft-foundry-advanced-intelligence-optimized-for/4560811
|
||||
AZURE_GPT_6_1_SOL_REGIONAL_PREMIUM: Final = MappingProxyType({"us": 1.1, "eu": 1.2, "apac": 1.2})
|
||||
|
||||
|
||||
@pytest.mark.parametrize("region", sorted(AZURE_GPT_6_1_SOL_REGIONAL_PREMIUM))
|
||||
def test_azure_gpt_6_1_sol_data_zone_rows_charge_the_global_web_search_price(prices: dict, region: str):
|
||||
global_search_cost = prices["azure/gpt-6.1-sol"]["search_context_cost_per_query"]
|
||||
assert global_search_cost
|
||||
assert prices[f"azure/{region}/gpt-6.1-sol"].get("search_context_cost_per_query") == global_search_cost
|
||||
|
||||
|
||||
@pytest.mark.parametrize("region", sorted(AZURE_GPT_6_1_SOL_REGIONAL_PREMIUM))
|
||||
def test_azure_gpt_6_1_sol_data_zone_rows_price_requests_at_the_regional_premium(region: str):
|
||||
premium = AZURE_GPT_6_1_SOL_REGIONAL_PREMIUM[region]
|
||||
usage = {"prompt_tokens": 300_000, "completion_tokens": 1_000, "cache_read_input_tokens": 100_000}
|
||||
|
||||
def total_cost(model: str) -> float:
|
||||
prompt_cost, completion_cost = litellm.cost_per_token(
|
||||
model=model,
|
||||
custom_llm_provider="azure",
|
||||
prompt_tokens=usage["prompt_tokens"],
|
||||
completion_tokens=usage["completion_tokens"],
|
||||
cache_read_input_tokens=usage["cache_read_input_tokens"],
|
||||
)
|
||||
return prompt_cost + completion_cost
|
||||
|
||||
global_cost = total_cost("azure/gpt-6.1-sol")
|
||||
assert global_cost > 0
|
||||
assert total_cost(f"azure/{region}/gpt-6.1-sol") == pytest.approx(global_cost * premium)
|
||||
|
||||
|
||||
# https://ai.google.dev/gemini-api/docs/models/deep-research-preview-04-2026
|
||||
@pytest.mark.parametrize("model", ["gemini/deep-research-preview-04-2026", "gemini/deep-research-max-preview-04-2026"])
|
||||
def test_gemini_deep_research_previews_accept_the_documented_input_window(prices: dict, model: str):
|
||||
assert prices[model]["max_input_tokens"] == 1_048_576
|
||||
assert prices[model]["max_output_tokens"] == 65_536
|
||||
|
||||
|
||||
# https://cloud.google.com/vertex-ai/generative-ai/docs/learn/model-versions
|
||||
VERTEX_GEMINI_FLASH_RETIREMENT_DATES: Final = MappingProxyType(
|
||||
{"gemini-3.6-flash": "2026-11-19", "gemini-3.7-flash": "2027-01-28"}
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", sorted(VERTEX_GEMINI_FLASH_RETIREMENT_DATES))
|
||||
@pytest.mark.parametrize("prefix", ["", "vertex_ai/"])
|
||||
def test_vertex_gemini_flash_rows_carry_the_documented_retirement_date(prices: dict, model: str, prefix: str):
|
||||
assert prices[f"{prefix}{model}"]["deprecation_date"] == VERTEX_GEMINI_FLASH_RETIREMENT_DATES[model]
|
||||
|
||||
|
||||
OPENAI_REASONING_FAMILY_MARKERS = ("codex", "deep-research", "chat-latest")
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue