feat(openai): add gpt-6-astra pricing and route it through the GPT-5 reasoning config

This commit is contained in:
mateo-berri 2026-09-03 10:56:22 -07:00
parent 7256bd307a
commit ec721f97f3
13 changed files with 276 additions and 20 deletions

View file

@ -40,21 +40,7 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config):
Accepts both explicit gpt-5 model names and the ``gpt5_series/`` prefix
used for manual routing.
"""
# The gpt-5-chat* family (gpt-5-chat, gpt-5-chat-latest, gpt-5-chat-2025-08-07,
# …) are regular chat models: they support temperature and tool_choice but NOT
# reasoning_effort. They must NOT be routed through the GPT-5 reasoning path.
#
# Versioned chat models such as gpt-5.3-chat and gpt-5.1-chat ARE reasoning
# models and must stay on the GPT-5 path. The distinguishing feature is that
# the gpt-5-chat family has a literal "-chat" immediately after "gpt-5"
# (i.e. "gpt-5-chat…"), while versioned chat models interpose a minor version
# number (i.e. "gpt-5.<digit>-chat").
#
# Using a startswith("gpt-5-chat") prefix check on the normalized name (rather
# than a substring check) makes this boundary explicit and avoids any ambiguity
# if future model names coincidentally contain "gpt-5-chat" as an interior run.
_normalized: Final = model.split("/")[-1] # strip provider prefix, e.g. "azure/"
return ("gpt-5" in model and not _normalized.startswith("gpt-5-chat")) or "gpt5_series" in model
return OpenAIGPT5Config.is_model_gpt_5_model(model) or "gpt5_series" in model
def get_supported_openai_params(self, model: str) -> list[str]:
"""Get supported parameters for Azure OpenAI GPT-5 models.

View file

@ -139,7 +139,7 @@ class AzureOpenAIConfig(BaseConfig):
name family needs the rename, including the ``gpt-5-chat*`` models that are excluded from
the reasoning path by https://github.com/BerriAI/litellm/issues/13781.
"""
return "gpt-5" in model or "gpt5_series" in model
return any(generation in model for generation in ("gpt-5", "gpt-6")) or "gpt5_series" in model
def _is_response_format_supported_model(self, model: str) -> bool:
"""

View file

@ -11,6 +11,8 @@ from litellm.utils import (
from .gpt_transformation import OpenAIGPTConfig
REASONING_GPT_GENERATIONS: Final = ("gpt-5", "gpt-6")
def _catalogue_declares_default_effort() -> bool:
"""Whether the loaded cost map carries default_reasoning_effort for ANY entry.
@ -87,7 +89,9 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
# than a substring check) makes this boundary explicit and avoids any ambiguity
# if future model names coincidentally contain "gpt-5-chat" as an interior run.
_normalized: Final = model.split("/")[-1] # strip provider prefix, e.g. "openai/"
return "gpt-5" in model and not _normalized.startswith("gpt-5-chat")
return any(generation in model for generation in REASONING_GPT_GENERATIONS) and not _normalized.startswith(
"gpt-5-chat"
)
@classmethod
def is_model_gpt_5_search_model(cls, model: str) -> bool:
@ -120,8 +124,10 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
@classmethod
def is_model_gpt_5_4_plus_model(cls, model: str) -> bool:
"""Check if the model is gpt-5.4 or newer (5.4, 5.5, 5.6, etc., including pro)."""
"""Check if the model is gpt-5.4 or newer (5.4, 5.5, 5.6, gpt-6, etc., including pro)."""
model_name: Final = model.split("/")[-1]
if model_name.startswith("gpt-6"):
return True
if not model_name.startswith("gpt-5."):
return False
try:

View file

@ -22,6 +22,7 @@ from litellm.types.responses.main import *
from litellm.types.router import GenericLiteLLMParams
from litellm.types.utils import LlmProviders
from ..chat.gpt_5_transformation import OpenAIGPT5Config
from ..common_utils import OpenAIError
from ..workload_identity import get_workload_identity_bearer_token, resolve_openai_workload_identity_config
@ -88,7 +89,7 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
parts: Final = model.split("/")
if len(parts) > 1 and parts[0] not in ("openai",):
return False
return "gpt-5" in model and "gpt-5-chat" not in model
return OpenAIGPT5Config.is_model_gpt_5_model(model)
@staticmethod
def _supports_reasoning_effort_none(model: str) -> bool:

View file

@ -29585,6 +29585,74 @@
"supports_computer_use": true,
"supports_parallel_function_calling": true
},
"gpt-6-astra": {
"cache_creation_input_token_cost": 1.25e-05,
"cache_creation_input_token_cost_above_272k_tokens": 2.5e-05,
"cache_creation_input_token_cost_above_272k_tokens_flex": 1.25e-05,
"cache_creation_input_token_cost_above_272k_tokens_priority": 5e-05,
"cache_creation_input_token_cost_flex": 6.25e-06,
"cache_creation_input_token_cost_priority": 2.5e-05,
"cache_read_input_token_cost": 1e-06,
"cache_read_input_token_cost_above_272k_tokens": 2e-06,
"cache_read_input_token_cost_above_272k_tokens_flex": 1e-06,
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
"cache_read_input_token_cost_flex": 5e-07,
"cache_read_input_token_cost_priority": 2e-06,
"input_cost_per_token": 1e-05,
"input_cost_per_token_above_272k_tokens": 2e-05,
"input_cost_per_token_above_272k_tokens_flex": 1e-05,
"input_cost_per_token_above_272k_tokens_priority": 4e-05,
"input_cost_per_token_batches": 5e-06,
"input_cost_per_token_flex": 5e-06,
"input_cost_per_token_priority": 2e-05,
"litellm_provider": "openai",
"max_input_tokens": 922000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 5e-05,
"output_cost_per_token_above_272k_tokens": 7.5e-05,
"output_cost_per_token_above_272k_tokens_flex": 3.75e-05,
"output_cost_per_token_above_272k_tokens_priority": 0.00015,
"output_cost_per_token_batches": 2.5e-05,
"output_cost_per_token_flex": 2.5e-05,
"output_cost_per_token_priority": 0.0001,
"regional_processing_uplift_multiplier_eu": 1.1,
"regional_processing_uplift_multiplier_us": 1.1,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": false,
"supports_native_streaming": true,
"supports_none_reasoning_effort": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_cache_breakpoint": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true,
"supports_xhigh_reasoning_effort": true
},
"daybreak-red-latest": {
"cache_creation_input_token_cost": 1.5625e-05,
"cache_creation_input_token_cost_above_272k_tokens": 3.125e-05,

View file

@ -29585,6 +29585,74 @@
"supports_computer_use": true,
"supports_parallel_function_calling": true
},
"gpt-6-astra": {
"cache_creation_input_token_cost": 1.25e-05,
"cache_creation_input_token_cost_above_272k_tokens": 2.5e-05,
"cache_creation_input_token_cost_above_272k_tokens_flex": 1.25e-05,
"cache_creation_input_token_cost_above_272k_tokens_priority": 5e-05,
"cache_creation_input_token_cost_flex": 6.25e-06,
"cache_creation_input_token_cost_priority": 2.5e-05,
"cache_read_input_token_cost": 1e-06,
"cache_read_input_token_cost_above_272k_tokens": 2e-06,
"cache_read_input_token_cost_above_272k_tokens_flex": 1e-06,
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
"cache_read_input_token_cost_flex": 5e-07,
"cache_read_input_token_cost_priority": 2e-06,
"input_cost_per_token": 1e-05,
"input_cost_per_token_above_272k_tokens": 2e-05,
"input_cost_per_token_above_272k_tokens_flex": 1e-05,
"input_cost_per_token_above_272k_tokens_priority": 4e-05,
"input_cost_per_token_batches": 5e-06,
"input_cost_per_token_flex": 5e-06,
"input_cost_per_token_priority": 2e-05,
"litellm_provider": "openai",
"max_input_tokens": 922000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 5e-05,
"output_cost_per_token_above_272k_tokens": 7.5e-05,
"output_cost_per_token_above_272k_tokens_flex": 3.75e-05,
"output_cost_per_token_above_272k_tokens_priority": 0.00015,
"output_cost_per_token_batches": 2.5e-05,
"output_cost_per_token_flex": 2.5e-05,
"output_cost_per_token_priority": 0.0001,
"regional_processing_uplift_multiplier_eu": 1.1,
"regional_processing_uplift_multiplier_us": 1.1,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supported_endpoints": [
"/v1/chat/completions",
"/v1/batch",
"/v1/responses"
],
"supported_modalities": [
"text",
"image"
],
"supported_output_modalities": [
"text"
],
"supports_computer_use": true,
"supports_function_calling": true,
"supports_minimal_reasoning_effort": false,
"supports_native_streaming": true,
"supports_none_reasoning_effort": true,
"supports_parallel_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_cache_breakpoint": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_vision": true,
"supports_web_search": true,
"supports_xhigh_reasoning_effort": true
},
"daybreak-red-latest": {
"cache_creation_input_token_cost": 1.5625e-05,
"cache_creation_input_token_cost_above_272k_tokens": 3.125e-05,

View file

@ -1615,6 +1615,71 @@ def test_generic_cost_per_token_gpt56_terra_cache_costs_by_tier_and_context(_loc
assert prompt_cost == pytest.approx(expected_prompt_cost)
@pytest.mark.parametrize(
"service_tier,prompt_tokens,input_rate,cache_write_rate,cache_read_rate,output_rate",
[
(None, 100000, 1e-5, 1.25e-5, 1e-6, 5e-5),
("flex", 100000, 5e-6, 6.25e-6, 5e-7, 2.5e-5),
("priority", 100000, 2e-5, 2.5e-5, 2e-6, 1e-4),
(None, 300000, 2e-5, 2.5e-5, 2e-6, 7.5e-5),
("flex", 300000, 1e-5, 1.25e-5, 1e-6, 3.75e-5),
("priority", 300000, 4e-5, 5e-5, 4e-6, 1.5e-4),
],
)
def test_generic_cost_per_token_gpt6_astra_by_tier_and_context(
_local_model_cost_map,
service_tier,
prompt_tokens,
input_rate,
cache_write_rate,
cache_read_rate,
output_rate,
):
"""gpt-6-astra: $10/$50 per 1M with $1 cache read and $12.50 cache write, flex at half,
priority at double, and the GPT-5.6 long-context multipliers (2x input and cache, 1.5x
output) once the prompt passes 272K tokens."""
cached_tokens = 50000
cache_write_tokens = 40000
text_tokens = prompt_tokens - cached_tokens - cache_write_tokens
completion_tokens = 1000
usage = Usage(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
total_tokens=prompt_tokens + completion_tokens,
prompt_tokens_details=PromptTokensDetailsWrapper(
cached_tokens=cached_tokens, cache_write_tokens=cache_write_tokens
),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="gpt-6-astra",
usage=usage,
custom_llm_provider="openai",
service_tier=service_tier,
)
assert prompt_cost == pytest.approx(
text_tokens * input_rate
+ cached_tokens * cache_read_rate
+ cache_write_tokens * cache_write_rate
)
assert completion_cost == pytest.approx(completion_tokens * output_rate)
def test_batch_cost_gpt6_astra_is_half_the_standard_rate(_local_model_cost_map):
from litellm.cost_calculator import batch_cost_calculator
prompt_cost, completion_cost = batch_cost_calculator(
usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
model="gpt-6-astra",
custom_llm_provider="openai",
model_info=litellm.get_model_info("gpt-6-astra"),
)
assert prompt_cost == pytest.approx(1000 * 5e-6)
assert completion_cost == pytest.approx(500 * 2.5e-5)
@pytest.mark.parametrize("model", ["gpt-5.6-cyber", "daybreak-red-latest"])
@pytest.mark.parametrize(
"prompt_tokens,input_rate,cache_write_rate,cache_read_rate,output_rate",

View file

@ -139,6 +139,7 @@ def test_transform_request_drops_tool_reference_parts():
("gpt-5-chat-latest", "max_completion_tokens", "max_tokens"),
("gpt-5-chat-2025-08-07", "max_completion_tokens", "max_tokens"),
("gpt-5", "max_completion_tokens", "max_tokens"),
("gpt-6-astra", "max_completion_tokens", "max_tokens"),
("o3-mini", "max_completion_tokens", "max_tokens"),
("gpt-4o", "max_tokens", "max_completion_tokens"),
],

View file

@ -1718,6 +1718,8 @@ class TestResponsesSurfaceSharesTheEffortRule:
("gpt-5.6-sol", None, False),
("gpt-5.6-terra", "none", True),
("gpt-5.6-terra", "medium", False),
("gpt-6-astra", None, False),
("gpt-6-astra", "none", True),
],
)
def test_temperature_follows_the_resolved_effort(

View file

@ -54,6 +54,8 @@ GPT5_MODELS = [
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-5.6-luna",
"gpt-6-astra",
"openai/gpt-6-astra",
"gpt-5.1-chat", # versioned chat — THE KEY REGRESSION CASE
"gpt-5.2-chat", # versioned chat — also a regression case
"gpt-5.3-chat", # versioned chat — THE KEY REGRESSION CASE
@ -128,6 +130,8 @@ GPT5_4_PLUS_MODELS = [
"gpt-5.6-terra",
"gpt-5.6-luna",
"openai/gpt-5.6-sol",
"gpt-6-astra",
"openai/gpt-6-astra",
]
GPT5_PRE_5_4_MODELS = [

View file

@ -0,0 +1,48 @@
import json
from pathlib import Path
import litellm
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config
from litellm.types.utils import LlmProviders
from litellm.utils import ProviderConfigManager
REPO_ROOT = Path(__file__).parents[2]
MAIN_PATH = REPO_ROOT / "model_prices_and_context_window.json"
BACKUP_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
MODEL = "gpt-6-astra"
def _load(path):
with open(path) as f:
return json.load(f)
def test_gpt_6_astra_backup_matches_main():
main_entry = _load(MAIN_PATH).get(MODEL)
assert main_entry is not None, f"{MODEL} missing from model_prices_and_context_window.json"
assert _load(BACKUP_PATH).get(MODEL) == main_entry
def test_gpt_6_astra_routes_to_openai_on_the_gpt_5_reasoning_path():
routed_model, provider, _, _ = get_llm_provider(model=f"openai/{MODEL}")
assert (routed_model, provider) == (MODEL, "openai")
config = ProviderConfigManager.get_provider_chat_config(model=MODEL, provider=LlmProviders.OPENAI)
assert isinstance(config, OpenAIGPT5Config)
def test_gpt_6_astra_maps_max_tokens_and_drops_temperature_like_gpt_5():
mapped = litellm.get_optional_params(
model=MODEL,
custom_llm_provider="openai",
max_tokens=100,
temperature=0.2,
reasoning_effort="high",
drop_params=True,
)
assert mapped["max_completion_tokens"] == 100
assert "max_tokens" not in mapped
assert "temperature" not in mapped
assert mapped["reasoning_effort"] == "high"

View file

@ -860,7 +860,7 @@ def test_responses_api_bridge_check_gpt_5_4_tools_with_default_reasoning_routes_
assert model_info.get("mode") == "responses"
@pytest.mark.parametrize("model_name", ["gpt-5.6-sol", "gpt-5.6-luna", "gpt-5.6-terra"])
@pytest.mark.parametrize("model_name", ["gpt-5.6-sol", "gpt-5.6-luna", "gpt-5.6-terra", "gpt-6-astra"])
def test_responses_api_bridge_check_gpt_5_6_tools_with_default_reasoning_routes_to_responses(
monkeypatch, model_name
):

View file

@ -52,6 +52,12 @@ PRIORITY_LONG_CONTEXT = {
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
"cache_creation_input_token_cost_above_272k_tokens_priority": 1e-06,
},
"gpt-6-astra": {
"input_cost_per_token_above_272k_tokens_priority": 4e-05,
"output_cost_per_token_above_272k_tokens_priority": 0.00015,
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
"cache_creation_input_token_cost_above_272k_tokens_priority": 5e-05,
},
}
EXPECTED = {**FLEX_LONG_CONTEXT, **PRIORITY_LONG_CONTEXT}
@ -114,6 +120,7 @@ TIERED_COST_CASES = [
("gpt-5.6-sol", "priority", 1.6e-05, 6e-05),
("gpt-5.6-terra", "priority", 8e-06, 3.6e-05),
("gpt-5.6-luna", "priority", 8e-07, 3.6e-06),
("gpt-6-astra", "priority", 4e-05, 0.00015),
]