fix(model_prices): gemini deep research input limits, vertex flash retirement dates, azure data zone gpt-6.1-sol pricing (#44633)

This commit is contained in:
devin-ai-integration[bot] 2026-10-06 07:40:25 -07:00 • committed by GitHub
parent 46d2c2a6ea
commit 126e79c967
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 356 additions and 4 deletions

View file

@ -8681,6 +8681,153 @@
"supports_web_search": true,
"supports_xhigh_reasoning_effort": true
},
"azure/us/gpt-6.1-sol": {
"cache_creation_input_token_cost": 2.75e-06,
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-06,
"cache_read_input_token_cost": 1.1e-07,
"cache_read_input_token_cost_above_272k_tokens": 2.2e-07,
"deprecation_date": "2028-03-11",
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
"litellm_provider": "azure",
"max_input_tokens": 922000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.1e-05,
"output_cost_per_token_above_272k_tokens": 1.65e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_computer_use": true,
"supports_function_calling": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": false,
"supports_native_streaming": true,
"supports_none_reasoning_effort": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true,
"supports_xhigh_reasoning_effort": true
},
"azure/eu/gpt-6.1-sol": {
"cache_creation_input_token_cost": 3e-06,
"cache_creation_input_token_cost_above_272k_tokens": 6e-06,
"cache_read_input_token_cost": 1.2e-07,
"cache_read_input_token_cost_above_272k_tokens": 2.4e-07,
"deprecation_date": "2028-03-11",
"input_cost_per_token": 2.4e-06,
"input_cost_per_token_above_272k_tokens": 4.8e-06,
"litellm_provider": "azure",
"max_input_tokens": 922000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.2e-05,
"output_cost_per_token_above_272k_tokens": 1.8e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_computer_use": true,
"supports_function_calling": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": false,
"supports_native_streaming": true,
"supports_none_reasoning_effort": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true,
"supports_xhigh_reasoning_effort": true
},
"azure/apac/gpt-6.1-sol": {
"cache_creation_input_token_cost": 3e-06,
"cache_creation_input_token_cost_above_272k_tokens": 6e-06,
"cache_read_input_token_cost": 1.2e-07,
"cache_read_input_token_cost_above_272k_tokens": 2.4e-07,
"deprecation_date": "2028-03-11",
"input_cost_per_token": 2.4e-06,
"input_cost_per_token_above_272k_tokens": 4.8e-06,
"litellm_provider": "azure",
"max_input_tokens": 922000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.2e-05,
"output_cost_per_token_above_272k_tokens": 1.8e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_computer_use": true,
"supports_function_calling": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": false,
"supports_native_streaming": true,
"supports_none_reasoning_effort": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true,
"supports_xhigh_reasoning_effort": true
},
"azure/gpt-chat-latest": {
"cache_read_input_token_cost": 5e-07,
"deprecation_date": "2026-12-02",
@ -28136,6 +28283,7 @@
"input_cost_per_token_priority": 1.35e-06,
"output_cost_per_token_priority": 6.75e-06,
"cache_read_input_token_cost_priority": 1.35e-07,
"deprecation_date": "2026-11-19",
"search_context_cost_per_query": {
"search_context_size_low": 0.014,
"search_context_size_medium": 0.014,
@ -28195,6 +28343,7 @@
"input_cost_per_token_priority": 1.35e-06,
"output_cost_per_token_priority": 6.75e-06,
"cache_read_input_token_cost_priority": 1.35e-07,
"deprecation_date": "2027-01-28",
"search_context_cost_per_query": {
"search_context_size_low": 0.014,
"search_context_size_medium": 0.014,
@ -28928,7 +29077,7 @@
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"litellm_provider": "gemini",
"max_input_tokens": 131072,
"max_input_tokens": 1048576,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
@ -28965,7 +29114,7 @@
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"litellm_provider": "gemini",
"max_input_tokens": 131072,
"max_input_tokens": 1048576,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
@ -30410,6 +30559,7 @@
"input_cost_per_token_priority": 1.35e-06,
"output_cost_per_token_priority": 6.75e-06,
"cache_read_input_token_cost_priority": 1.35e-07,
"deprecation_date": "2026-11-19",
"search_context_cost_per_query": {
"search_context_size_low": 0.014,
"search_context_size_medium": 0.014,
@ -30469,6 +30619,7 @@
"input_cost_per_token_priority": 1.35e-06,
"output_cost_per_token_priority": 6.75e-06,
"cache_read_input_token_cost_priority": 1.35e-07,
"deprecation_date": "2027-01-28",
"search_context_cost_per_query": {
"search_context_size_low": 0.014,
"search_context_size_medium": 0.014,

View file

@ -8681,6 +8681,153 @@
"supports_web_search": true,
"supports_xhigh_reasoning_effort": true
},
"azure/us/gpt-6.1-sol": {
"cache_creation_input_token_cost": 2.75e-06,
"cache_creation_input_token_cost_above_272k_tokens": 5.5e-06,
"cache_read_input_token_cost": 1.1e-07,
"cache_read_input_token_cost_above_272k_tokens": 2.2e-07,
"deprecation_date": "2028-03-11",
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
"litellm_provider": "azure",
"max_input_tokens": 922000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.1e-05,
"output_cost_per_token_above_272k_tokens": 1.65e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_computer_use": true,
"supports_function_calling": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": false,
"supports_native_streaming": true,
"supports_none_reasoning_effort": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true,
"supports_xhigh_reasoning_effort": true
},
"azure/eu/gpt-6.1-sol": {
"cache_creation_input_token_cost": 3e-06,
"cache_creation_input_token_cost_above_272k_tokens": 6e-06,
"cache_read_input_token_cost": 1.2e-07,
"cache_read_input_token_cost_above_272k_tokens": 2.4e-07,
"deprecation_date": "2028-03-11",
"input_cost_per_token": 2.4e-06,
"input_cost_per_token_above_272k_tokens": 4.8e-06,
"litellm_provider": "azure",
"max_input_tokens": 922000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.2e-05,
"output_cost_per_token_above_272k_tokens": 1.8e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_computer_use": true,
"supports_function_calling": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": false,
"supports_native_streaming": true,
"supports_none_reasoning_effort": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true,
"supports_xhigh_reasoning_effort": true
},
"azure/apac/gpt-6.1-sol": {
"cache_creation_input_token_cost": 3e-06,
"cache_creation_input_token_cost_above_272k_tokens": 6e-06,
"cache_read_input_token_cost": 1.2e-07,
"cache_read_input_token_cost_above_272k_tokens": 2.4e-07,
"deprecation_date": "2028-03-11",
"input_cost_per_token": 2.4e-06,
"input_cost_per_token_above_272k_tokens": 4.8e-06,
"litellm_provider": "azure",
"max_input_tokens": 922000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 1.2e-05,
"output_cost_per_token_above_272k_tokens": 1.8e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"source": "https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models",
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_computer_use": true,
"supports_function_calling": true,
"supports_max_reasoning_effort": true,
"supports_minimal_reasoning_effort": false,
"supports_native_streaming": true,
"supports_none_reasoning_effort": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true,
"supports_xhigh_reasoning_effort": true
},
"azure/gpt-chat-latest": {
"cache_read_input_token_cost": 5e-07,
"deprecation_date": "2026-12-02",
@ -28136,6 +28283,7 @@
"input_cost_per_token_priority": 1.35e-06,
"output_cost_per_token_priority": 6.75e-06,
"cache_read_input_token_cost_priority": 1.35e-07,
"deprecation_date": "2026-11-19",
"search_context_cost_per_query": {
"search_context_size_low": 0.014,
"search_context_size_medium": 0.014,
@ -28195,6 +28343,7 @@
"input_cost_per_token_priority": 1.35e-06,
"output_cost_per_token_priority": 6.75e-06,
"cache_read_input_token_cost_priority": 1.35e-07,
"deprecation_date": "2027-01-28",
"search_context_cost_per_query": {
"search_context_size_low": 0.014,
"search_context_size_medium": 0.014,
@ -28928,7 +29077,7 @@
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"litellm_provider": "gemini",
"max_input_tokens": 131072,
"max_input_tokens": 1048576,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
@ -28965,7 +29114,7 @@
"input_cost_per_token": 2e-06,
"input_cost_per_token_batches": 1e-06,
"litellm_provider": "gemini",
"max_input_tokens": 131072,
"max_input_tokens": 1048576,
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
@ -30410,6 +30559,7 @@
"input_cost_per_token_priority": 1.35e-06,
"output_cost_per_token_priority": 6.75e-06,
"cache_read_input_token_cost_priority": 1.35e-07,
"deprecation_date": "2026-11-19",
"search_context_cost_per_query": {
"search_context_size_low": 0.014,
"search_context_size_medium": 0.014,
@ -30469,6 +30619,7 @@
"input_cost_per_token_priority": 1.35e-06,
"output_cost_per_token_priority": 6.75e-06,
"cache_read_input_token_cost_priority": 1.35e-07,
"deprecation_date": "2027-01-28",
"search_context_cost_per_query": {
"search_context_size_low": 0.014,
"search_context_size_medium": 0.014,

View file

@ -232,6 +232,56 @@ def test_dated_variants_carry_base_alias_service_tier_pricing(prices: dict):
)
# Azure Foundry GPT-6.1 Sol announcement, 2026-09-29: https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-gpt-6-1-sol-in-microsoft-foundry-advanced-intelligence-optimized-for/4560811
AZURE_GPT_6_1_SOL_REGIONAL_PREMIUM: Final = MappingProxyType({"us": 1.1, "eu": 1.2, "apac": 1.2})
@pytest.mark.parametrize("region", sorted(AZURE_GPT_6_1_SOL_REGIONAL_PREMIUM))
def test_azure_gpt_6_1_sol_data_zone_rows_charge_the_global_web_search_price(prices: dict, region: str):
global_search_cost = prices["azure/gpt-6.1-sol"]["search_context_cost_per_query"]
assert global_search_cost
assert prices[f"azure/{region}/gpt-6.1-sol"].get("search_context_cost_per_query") == global_search_cost
@pytest.mark.parametrize("region", sorted(AZURE_GPT_6_1_SOL_REGIONAL_PREMIUM))
def test_azure_gpt_6_1_sol_data_zone_rows_price_requests_at_the_regional_premium(region: str):
premium = AZURE_GPT_6_1_SOL_REGIONAL_PREMIUM[region]
usage = {"prompt_tokens": 300_000, "completion_tokens": 1_000, "cache_read_input_tokens": 100_000}
def total_cost(model: str) -> float:
prompt_cost, completion_cost = litellm.cost_per_token(
model=model,
custom_llm_provider="azure",
prompt_tokens=usage["prompt_tokens"],
completion_tokens=usage["completion_tokens"],
cache_read_input_tokens=usage["cache_read_input_tokens"],
)
return prompt_cost + completion_cost
global_cost = total_cost("azure/gpt-6.1-sol")
assert global_cost > 0
assert total_cost(f"azure/{region}/gpt-6.1-sol") == pytest.approx(global_cost * premium)
# https://ai.google.dev/gemini-api/docs/models/deep-research-preview-04-2026
@pytest.mark.parametrize("model", ["gemini/deep-research-preview-04-2026", "gemini/deep-research-max-preview-04-2026"])
def test_gemini_deep_research_previews_accept_the_documented_input_window(prices: dict, model: str):
assert prices[model]["max_input_tokens"] == 1_048_576
assert prices[model]["max_output_tokens"] == 65_536
# https://cloud.google.com/vertex-ai/generative-ai/docs/learn/model-versions
VERTEX_GEMINI_FLASH_RETIREMENT_DATES: Final = MappingProxyType(
{"gemini-3.6-flash": "2026-11-19", "gemini-3.7-flash": "2027-01-28"}
)
@pytest.mark.parametrize("model", sorted(VERTEX_GEMINI_FLASH_RETIREMENT_DATES))
@pytest.mark.parametrize("prefix", ["", "vertex_ai/"])
def test_vertex_gemini_flash_rows_carry_the_documented_retirement_date(prices: dict, model: str, prefix: str):
assert prices[f"{prefix}{model}"]["deprecation_date"] == VERTEX_GEMINI_FLASH_RETIREMENT_DATES[model]
OPENAI_REASONING_FAMILY_MARKERS = ("codex", "deep-research", "chat-latest")