diff --git a/litellm/llms/azure/chat/gpt_5_transformation.py b/litellm/llms/azure/chat/gpt_5_transformation.py index 6fdd277a04f..d1272c982dc 100644 --- a/litellm/llms/azure/chat/gpt_5_transformation.py +++ b/litellm/llms/azure/chat/gpt_5_transformation.py @@ -40,21 +40,7 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config): Accepts both explicit gpt-5 model names and the ``gpt5_series/`` prefix used for manual routing. """ - # The gpt-5-chat* family (gpt-5-chat, gpt-5-chat-latest, gpt-5-chat-2025-08-07, - # …) are regular chat models: they support temperature and tool_choice but NOT - # reasoning_effort. They must NOT be routed through the GPT-5 reasoning path. - # - # Versioned chat models such as gpt-5.3-chat and gpt-5.1-chat ARE reasoning - # models and must stay on the GPT-5 path. The distinguishing feature is that - # the gpt-5-chat family has a literal "-chat" immediately after "gpt-5" - # (i.e. "gpt-5-chat…"), while versioned chat models interpose a minor version - # number (i.e. "gpt-5.-chat"). - # - # Using a startswith("gpt-5-chat") prefix check on the normalized name (rather - # than a substring check) makes this boundary explicit and avoids any ambiguity - # if future model names coincidentally contain "gpt-5-chat" as an interior run. - _normalized: Final = model.split("/")[-1] # strip provider prefix, e.g. "azure/" - return ("gpt-5" in model and not _normalized.startswith("gpt-5-chat")) or "gpt5_series" in model + return OpenAIGPT5Config.is_model_gpt_5_model(model) or "gpt5_series" in model def get_supported_openai_params(self, model: str) -> list[str]: """Get supported parameters for Azure OpenAI GPT-5 models. diff --git a/litellm/llms/azure/chat/gpt_transformation.py b/litellm/llms/azure/chat/gpt_transformation.py index 0ac0662205a..ceeee8c12cb 100644 --- a/litellm/llms/azure/chat/gpt_transformation.py +++ b/litellm/llms/azure/chat/gpt_transformation.py @@ -139,7 +139,7 @@ class AzureOpenAIConfig(BaseConfig): name family needs the rename, including the ``gpt-5-chat*`` models that are excluded from the reasoning path by https://github.com/BerriAI/litellm/issues/13781. """ - return "gpt-5" in model or "gpt5_series" in model + return any(generation in model for generation in ("gpt-5", "gpt-6")) or "gpt5_series" in model def _is_response_format_supported_model(self, model: str) -> bool: """ diff --git a/litellm/llms/openai/chat/gpt_5_transformation.py b/litellm/llms/openai/chat/gpt_5_transformation.py index 0223be300b0..f6887387540 100644 --- a/litellm/llms/openai/chat/gpt_5_transformation.py +++ b/litellm/llms/openai/chat/gpt_5_transformation.py @@ -11,6 +11,8 @@ from litellm.utils import ( from .gpt_transformation import OpenAIGPTConfig +REASONING_GPT_GENERATIONS: Final = ("gpt-5", "gpt-6") + def _catalogue_declares_default_effort() -> bool: """Whether the loaded cost map carries default_reasoning_effort for ANY entry. @@ -87,7 +89,9 @@ class OpenAIGPT5Config(OpenAIGPTConfig): # than a substring check) makes this boundary explicit and avoids any ambiguity # if future model names coincidentally contain "gpt-5-chat" as an interior run. _normalized: Final = model.split("/")[-1] # strip provider prefix, e.g. "openai/" - return "gpt-5" in model and not _normalized.startswith("gpt-5-chat") + return any(generation in model for generation in REASONING_GPT_GENERATIONS) and not _normalized.startswith( + "gpt-5-chat" + ) @classmethod def is_model_gpt_5_search_model(cls, model: str) -> bool: @@ -120,8 +124,10 @@ class OpenAIGPT5Config(OpenAIGPTConfig): @classmethod def is_model_gpt_5_4_plus_model(cls, model: str) -> bool: - """Check if the model is gpt-5.4 or newer (5.4, 5.5, 5.6, etc., including pro).""" + """Check if the model is gpt-5.4 or newer (5.4, 5.5, 5.6, gpt-6, etc., including pro).""" model_name: Final = model.split("/")[-1] + if model_name.startswith("gpt-6"): + return True if not model_name.startswith("gpt-5."): return False try: diff --git a/litellm/llms/openai/responses/transformation.py b/litellm/llms/openai/responses/transformation.py index 01313e95878..85222144b82 100644 --- a/litellm/llms/openai/responses/transformation.py +++ b/litellm/llms/openai/responses/transformation.py @@ -22,6 +22,7 @@ from litellm.types.responses.main import * from litellm.types.router import GenericLiteLLMParams from litellm.types.utils import LlmProviders +from ..chat.gpt_5_transformation import OpenAIGPT5Config from ..common_utils import OpenAIError from ..workload_identity import get_workload_identity_bearer_token, resolve_openai_workload_identity_config @@ -88,7 +89,7 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig): parts: Final = model.split("/") if len(parts) > 1 and parts[0] not in ("openai",): return False - return "gpt-5" in model and "gpt-5-chat" not in model + return OpenAIGPT5Config.is_model_gpt_5_model(model) @staticmethod def _supports_reasoning_effort_none(model: str) -> bool: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d8a8f84b032..f44f01634d5 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -29585,6 +29585,74 @@ "supports_computer_use": true, "supports_parallel_function_calling": true }, + "gpt-6-astra": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_272k_tokens": 2.5e-05, + "cache_creation_input_token_cost_above_272k_tokens_flex": 1.25e-05, + "cache_creation_input_token_cost_above_272k_tokens_priority": 5e-05, + "cache_creation_input_token_cost_flex": 6.25e-06, + "cache_creation_input_token_cost_priority": 2.5e-05, + "cache_read_input_token_cost": 1e-06, + "cache_read_input_token_cost_above_272k_tokens": 2e-06, + "cache_read_input_token_cost_above_272k_tokens_flex": 1e-06, + "cache_read_input_token_cost_above_272k_tokens_priority": 4e-06, + "cache_read_input_token_cost_flex": 5e-07, + "cache_read_input_token_cost_priority": 2e-06, + "input_cost_per_token": 1e-05, + "input_cost_per_token_above_272k_tokens": 2e-05, + "input_cost_per_token_above_272k_tokens_flex": 1e-05, + "input_cost_per_token_above_272k_tokens_priority": 4e-05, + "input_cost_per_token_batches": 5e-06, + "input_cost_per_token_flex": 5e-06, + "input_cost_per_token_priority": 2e-05, + "litellm_provider": "openai", + "max_input_tokens": 922000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "output_cost_per_token_above_272k_tokens": 7.5e-05, + "output_cost_per_token_above_272k_tokens_flex": 3.75e-05, + "output_cost_per_token_above_272k_tokens_priority": 0.00015, + "output_cost_per_token_batches": 2.5e-05, + "output_cost_per_token_flex": 2.5e-05, + "output_cost_per_token_priority": 0.0001, + "regional_processing_uplift_multiplier_eu": 1.1, + "regional_processing_uplift_multiplier_us": 1.1, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_computer_use": true, + "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, + "supports_native_streaming": true, + "supports_none_reasoning_effort": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_cache_breakpoint": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true, + "supports_xhigh_reasoning_effort": true + }, "daybreak-red-latest": { "cache_creation_input_token_cost": 1.5625e-05, "cache_creation_input_token_cost_above_272k_tokens": 3.125e-05, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d8a8f84b032..f44f01634d5 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -29585,6 +29585,74 @@ "supports_computer_use": true, "supports_parallel_function_calling": true }, + "gpt-6-astra": { + "cache_creation_input_token_cost": 1.25e-05, + "cache_creation_input_token_cost_above_272k_tokens": 2.5e-05, + "cache_creation_input_token_cost_above_272k_tokens_flex": 1.25e-05, + "cache_creation_input_token_cost_above_272k_tokens_priority": 5e-05, + "cache_creation_input_token_cost_flex": 6.25e-06, + "cache_creation_input_token_cost_priority": 2.5e-05, + "cache_read_input_token_cost": 1e-06, + "cache_read_input_token_cost_above_272k_tokens": 2e-06, + "cache_read_input_token_cost_above_272k_tokens_flex": 1e-06, + "cache_read_input_token_cost_above_272k_tokens_priority": 4e-06, + "cache_read_input_token_cost_flex": 5e-07, + "cache_read_input_token_cost_priority": 2e-06, + "input_cost_per_token": 1e-05, + "input_cost_per_token_above_272k_tokens": 2e-05, + "input_cost_per_token_above_272k_tokens_flex": 1e-05, + "input_cost_per_token_above_272k_tokens_priority": 4e-05, + "input_cost_per_token_batches": 5e-06, + "input_cost_per_token_flex": 5e-06, + "input_cost_per_token_priority": 2e-05, + "litellm_provider": "openai", + "max_input_tokens": 922000, + "max_output_tokens": 128000, + "max_tokens": 128000, + "mode": "chat", + "output_cost_per_token": 5e-05, + "output_cost_per_token_above_272k_tokens": 7.5e-05, + "output_cost_per_token_above_272k_tokens_flex": 3.75e-05, + "output_cost_per_token_above_272k_tokens_priority": 0.00015, + "output_cost_per_token_batches": 2.5e-05, + "output_cost_per_token_flex": 2.5e-05, + "output_cost_per_token_priority": 0.0001, + "regional_processing_uplift_multiplier_eu": 1.1, + "regional_processing_uplift_multiplier_us": 1.1, + "search_context_cost_per_query": { + "search_context_size_high": 0.01, + "search_context_size_low": 0.01, + "search_context_size_medium": 0.01 + }, + "supported_endpoints": [ + "/v1/chat/completions", + "/v1/batch", + "/v1/responses" + ], + "supported_modalities": [ + "text", + "image" + ], + "supported_output_modalities": [ + "text" + ], + "supports_computer_use": true, + "supports_function_calling": true, + "supports_minimal_reasoning_effort": false, + "supports_native_streaming": true, + "supports_none_reasoning_effort": true, + "supports_parallel_function_calling": true, + "supports_pdf_input": true, + "supports_prompt_cache_breakpoint": true, + "supports_prompt_caching": true, + "supports_reasoning": true, + "supports_response_schema": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_vision": true, + "supports_web_search": true, + "supports_xhigh_reasoning_effort": true + }, "daybreak-red-latest": { "cache_creation_input_token_cost": 1.5625e-05, "cache_creation_input_token_cost_above_272k_tokens": 3.125e-05, diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index b7f0ca1efe1..0acf423a605 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -1615,6 +1615,71 @@ def test_generic_cost_per_token_gpt56_terra_cache_costs_by_tier_and_context(_loc assert prompt_cost == pytest.approx(expected_prompt_cost) +@pytest.mark.parametrize( + "service_tier,prompt_tokens,input_rate,cache_write_rate,cache_read_rate,output_rate", + [ + (None, 100000, 1e-5, 1.25e-5, 1e-6, 5e-5), + ("flex", 100000, 5e-6, 6.25e-6, 5e-7, 2.5e-5), + ("priority", 100000, 2e-5, 2.5e-5, 2e-6, 1e-4), + (None, 300000, 2e-5, 2.5e-5, 2e-6, 7.5e-5), + ("flex", 300000, 1e-5, 1.25e-5, 1e-6, 3.75e-5), + ("priority", 300000, 4e-5, 5e-5, 4e-6, 1.5e-4), + ], +) +def test_generic_cost_per_token_gpt6_astra_by_tier_and_context( + _local_model_cost_map, + service_tier, + prompt_tokens, + input_rate, + cache_write_rate, + cache_read_rate, + output_rate, +): + """gpt-6-astra: $10/$50 per 1M with $1 cache read and $12.50 cache write, flex at half, + priority at double, and the GPT-5.6 long-context multipliers (2x input and cache, 1.5x + output) once the prompt passes 272K tokens.""" + cached_tokens = 50000 + cache_write_tokens = 40000 + text_tokens = prompt_tokens - cached_tokens - cache_write_tokens + completion_tokens = 1000 + usage = Usage( + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + total_tokens=prompt_tokens + completion_tokens, + prompt_tokens_details=PromptTokensDetailsWrapper( + cached_tokens=cached_tokens, cache_write_tokens=cache_write_tokens + ), + ) + + prompt_cost, completion_cost = generic_cost_per_token( + model="gpt-6-astra", + usage=usage, + custom_llm_provider="openai", + service_tier=service_tier, + ) + + assert prompt_cost == pytest.approx( + text_tokens * input_rate + + cached_tokens * cache_read_rate + + cache_write_tokens * cache_write_rate + ) + assert completion_cost == pytest.approx(completion_tokens * output_rate) + + +def test_batch_cost_gpt6_astra_is_half_the_standard_rate(_local_model_cost_map): + from litellm.cost_calculator import batch_cost_calculator + + prompt_cost, completion_cost = batch_cost_calculator( + usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500), + model="gpt-6-astra", + custom_llm_provider="openai", + model_info=litellm.get_model_info("gpt-6-astra"), + ) + + assert prompt_cost == pytest.approx(1000 * 5e-6) + assert completion_cost == pytest.approx(500 * 2.5e-5) + + @pytest.mark.parametrize("model", ["gpt-5.6-cyber", "daybreak-red-latest"]) @pytest.mark.parametrize( "prompt_tokens,input_rate,cache_write_rate,cache_read_rate,output_rate", diff --git a/tests/test_litellm/llms/azure/chat/test_azure_chat_gpt_transformation.py b/tests/test_litellm/llms/azure/chat/test_azure_chat_gpt_transformation.py index 4e6b9ed0188..4f785607bde 100644 --- a/tests/test_litellm/llms/azure/chat/test_azure_chat_gpt_transformation.py +++ b/tests/test_litellm/llms/azure/chat/test_azure_chat_gpt_transformation.py @@ -139,6 +139,7 @@ def test_transform_request_drops_tool_reference_parts(): ("gpt-5-chat-latest", "max_completion_tokens", "max_tokens"), ("gpt-5-chat-2025-08-07", "max_completion_tokens", "max_tokens"), ("gpt-5", "max_completion_tokens", "max_tokens"), + ("gpt-6-astra", "max_completion_tokens", "max_tokens"), ("o3-mini", "max_completion_tokens", "max_tokens"), ("gpt-4o", "max_tokens", "max_completion_tokens"), ], diff --git a/tests/test_litellm/llms/openai/responses/test_openai_responses_transformation.py b/tests/test_litellm/llms/openai/responses/test_openai_responses_transformation.py index b0ffd1845fe..81abfbec3c4 100644 --- a/tests/test_litellm/llms/openai/responses/test_openai_responses_transformation.py +++ b/tests/test_litellm/llms/openai/responses/test_openai_responses_transformation.py @@ -1718,6 +1718,8 @@ class TestResponsesSurfaceSharesTheEffortRule: ("gpt-5.6-sol", None, False), ("gpt-5.6-terra", "none", True), ("gpt-5.6-terra", "medium", False), + ("gpt-6-astra", None, False), + ("gpt-6-astra", "none", True), ], ) def test_temperature_follows_the_resolved_effort( diff --git a/tests/test_litellm/llms/openai/test_is_model_gpt_5_model.py b/tests/test_litellm/llms/openai/test_is_model_gpt_5_model.py index 1095819c98c..3a2bf71363f 100644 --- a/tests/test_litellm/llms/openai/test_is_model_gpt_5_model.py +++ b/tests/test_litellm/llms/openai/test_is_model_gpt_5_model.py @@ -54,6 +54,8 @@ GPT5_MODELS = [ "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna", + "gpt-6-astra", + "openai/gpt-6-astra", "gpt-5.1-chat", # versioned chat — THE KEY REGRESSION CASE "gpt-5.2-chat", # versioned chat — also a regression case "gpt-5.3-chat", # versioned chat — THE KEY REGRESSION CASE @@ -128,6 +130,8 @@ GPT5_4_PLUS_MODELS = [ "gpt-5.6-terra", "gpt-5.6-luna", "openai/gpt-5.6-sol", + "gpt-6-astra", + "openai/gpt-6-astra", ] GPT5_PRE_5_4_MODELS = [ diff --git a/tests/test_litellm/test_gpt_6_astra_model_metadata.py b/tests/test_litellm/test_gpt_6_astra_model_metadata.py new file mode 100644 index 00000000000..faf2d443846 --- /dev/null +++ b/tests/test_litellm/test_gpt_6_astra_model_metadata.py @@ -0,0 +1,48 @@ +import json +from pathlib import Path + +import litellm +from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider +from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config +from litellm.types.utils import LlmProviders +from litellm.utils import ProviderConfigManager + +REPO_ROOT = Path(__file__).parents[2] +MAIN_PATH = REPO_ROOT / "model_prices_and_context_window.json" +BACKUP_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json" +MODEL = "gpt-6-astra" + + +def _load(path): + with open(path) as f: + return json.load(f) + + +def test_gpt_6_astra_backup_matches_main(): + main_entry = _load(MAIN_PATH).get(MODEL) + assert main_entry is not None, f"{MODEL} missing from model_prices_and_context_window.json" + assert _load(BACKUP_PATH).get(MODEL) == main_entry + + +def test_gpt_6_astra_routes_to_openai_on_the_gpt_5_reasoning_path(): + routed_model, provider, _, _ = get_llm_provider(model=f"openai/{MODEL}") + assert (routed_model, provider) == (MODEL, "openai") + + config = ProviderConfigManager.get_provider_chat_config(model=MODEL, provider=LlmProviders.OPENAI) + assert isinstance(config, OpenAIGPT5Config) + + +def test_gpt_6_astra_maps_max_tokens_and_drops_temperature_like_gpt_5(): + mapped = litellm.get_optional_params( + model=MODEL, + custom_llm_provider="openai", + max_tokens=100, + temperature=0.2, + reasoning_effort="high", + drop_params=True, + ) + + assert mapped["max_completion_tokens"] == 100 + assert "max_tokens" not in mapped + assert "temperature" not in mapped + assert mapped["reasoning_effort"] == "high" diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index 3c8bf142835..f21d5373b74 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -860,7 +860,7 @@ def test_responses_api_bridge_check_gpt_5_4_tools_with_default_reasoning_routes_ assert model_info.get("mode") == "responses" -@pytest.mark.parametrize("model_name", ["gpt-5.6-sol", "gpt-5.6-luna", "gpt-5.6-terra"]) +@pytest.mark.parametrize("model_name", ["gpt-5.6-sol", "gpt-5.6-luna", "gpt-5.6-terra", "gpt-6-astra"]) def test_responses_api_bridge_check_gpt_5_6_tools_with_default_reasoning_routes_to_responses( monkeypatch, model_name ): diff --git a/tests/test_litellm/test_openai_service_tier_long_context_pricing.py b/tests/test_litellm/test_openai_service_tier_long_context_pricing.py index c0860a5b55f..70e9c2720b8 100644 --- a/tests/test_litellm/test_openai_service_tier_long_context_pricing.py +++ b/tests/test_litellm/test_openai_service_tier_long_context_pricing.py @@ -52,6 +52,12 @@ PRIORITY_LONG_CONTEXT = { "cache_read_input_token_cost_above_272k_tokens_priority": 8e-08, "cache_creation_input_token_cost_above_272k_tokens_priority": 1e-06, }, + "gpt-6-astra": { + "input_cost_per_token_above_272k_tokens_priority": 4e-05, + "output_cost_per_token_above_272k_tokens_priority": 0.00015, + "cache_read_input_token_cost_above_272k_tokens_priority": 4e-06, + "cache_creation_input_token_cost_above_272k_tokens_priority": 5e-05, + }, } EXPECTED = {**FLEX_LONG_CONTEXT, **PRIORITY_LONG_CONTEXT} @@ -114,6 +120,7 @@ TIERED_COST_CASES = [ ("gpt-5.6-sol", "priority", 1.6e-05, 6e-05), ("gpt-5.6-terra", "priority", 8e-06, 3.6e-05), ("gpt-5.6-luna", "priority", 8e-07, 3.6e-06), + ("gpt-6-astra", "priority", 4e-05, 0.00015), ]