mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
feat(openai): add gpt-6-astra pricing and route it through the GPT-5 reasoning config
This commit is contained in:
parent
7256bd307a
commit
ec721f97f3
13 changed files with 276 additions and 20 deletions
|
|
@ -40,21 +40,7 @@ class AzureOpenAIGPT5Config(AzureOpenAIConfig, OpenAIGPT5Config):
|
|||
Accepts both explicit gpt-5 model names and the ``gpt5_series/`` prefix
|
||||
used for manual routing.
|
||||
"""
|
||||
# The gpt-5-chat* family (gpt-5-chat, gpt-5-chat-latest, gpt-5-chat-2025-08-07,
|
||||
# …) are regular chat models: they support temperature and tool_choice but NOT
|
||||
# reasoning_effort. They must NOT be routed through the GPT-5 reasoning path.
|
||||
#
|
||||
# Versioned chat models such as gpt-5.3-chat and gpt-5.1-chat ARE reasoning
|
||||
# models and must stay on the GPT-5 path. The distinguishing feature is that
|
||||
# the gpt-5-chat family has a literal "-chat" immediately after "gpt-5"
|
||||
# (i.e. "gpt-5-chat…"), while versioned chat models interpose a minor version
|
||||
# number (i.e. "gpt-5.<digit>-chat").
|
||||
#
|
||||
# Using a startswith("gpt-5-chat") prefix check on the normalized name (rather
|
||||
# than a substring check) makes this boundary explicit and avoids any ambiguity
|
||||
# if future model names coincidentally contain "gpt-5-chat" as an interior run.
|
||||
_normalized: Final = model.split("/")[-1] # strip provider prefix, e.g. "azure/"
|
||||
return ("gpt-5" in model and not _normalized.startswith("gpt-5-chat")) or "gpt5_series" in model
|
||||
return OpenAIGPT5Config.is_model_gpt_5_model(model) or "gpt5_series" in model
|
||||
|
||||
def get_supported_openai_params(self, model: str) -> list[str]:
|
||||
"""Get supported parameters for Azure OpenAI GPT-5 models.
|
||||
|
|
|
|||
|
|
@ -139,7 +139,7 @@ class AzureOpenAIConfig(BaseConfig):
|
|||
name family needs the rename, including the ``gpt-5-chat*`` models that are excluded from
|
||||
the reasoning path by https://github.com/BerriAI/litellm/issues/13781.
|
||||
"""
|
||||
return "gpt-5" in model or "gpt5_series" in model
|
||||
return any(generation in model for generation in ("gpt-5", "gpt-6")) or "gpt5_series" in model
|
||||
|
||||
def _is_response_format_supported_model(self, model: str) -> bool:
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -11,6 +11,8 @@ from litellm.utils import (
|
|||
|
||||
from .gpt_transformation import OpenAIGPTConfig
|
||||
|
||||
REASONING_GPT_GENERATIONS: Final = ("gpt-5", "gpt-6")
|
||||
|
||||
|
||||
def _catalogue_declares_default_effort() -> bool:
|
||||
"""Whether the loaded cost map carries default_reasoning_effort for ANY entry.
|
||||
|
|
@ -87,7 +89,9 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
|
|||
# than a substring check) makes this boundary explicit and avoids any ambiguity
|
||||
# if future model names coincidentally contain "gpt-5-chat" as an interior run.
|
||||
_normalized: Final = model.split("/")[-1] # strip provider prefix, e.g. "openai/"
|
||||
return "gpt-5" in model and not _normalized.startswith("gpt-5-chat")
|
||||
return any(generation in model for generation in REASONING_GPT_GENERATIONS) and not _normalized.startswith(
|
||||
"gpt-5-chat"
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def is_model_gpt_5_search_model(cls, model: str) -> bool:
|
||||
|
|
@ -120,8 +124,10 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
|
|||
|
||||
@classmethod
|
||||
def is_model_gpt_5_4_plus_model(cls, model: str) -> bool:
|
||||
"""Check if the model is gpt-5.4 or newer (5.4, 5.5, 5.6, etc., including pro)."""
|
||||
"""Check if the model is gpt-5.4 or newer (5.4, 5.5, 5.6, gpt-6, etc., including pro)."""
|
||||
model_name: Final = model.split("/")[-1]
|
||||
if model_name.startswith("gpt-6"):
|
||||
return True
|
||||
if not model_name.startswith("gpt-5."):
|
||||
return False
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -22,6 +22,7 @@ from litellm.types.responses.main import *
|
|||
from litellm.types.router import GenericLiteLLMParams
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
||||
from ..chat.gpt_5_transformation import OpenAIGPT5Config
|
||||
from ..common_utils import OpenAIError
|
||||
from ..workload_identity import get_workload_identity_bearer_token, resolve_openai_workload_identity_config
|
||||
|
||||
|
|
@ -88,7 +89,7 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
parts: Final = model.split("/")
|
||||
if len(parts) > 1 and parts[0] not in ("openai",):
|
||||
return False
|
||||
return "gpt-5" in model and "gpt-5-chat" not in model
|
||||
return OpenAIGPT5Config.is_model_gpt_5_model(model)
|
||||
|
||||
@staticmethod
|
||||
def _supports_reasoning_effort_none(model: str) -> bool:
|
||||
|
|
|
|||
|
|
@ -29585,6 +29585,74 @@
|
|||
"supports_computer_use": true,
|
||||
"supports_parallel_function_calling": true
|
||||
},
|
||||
"gpt-6-astra": {
|
||||
"cache_creation_input_token_cost": 1.25e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 2.5e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_flex": 1.25e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 5e-05,
|
||||
"cache_creation_input_token_cost_flex": 6.25e-06,
|
||||
"cache_creation_input_token_cost_priority": 2.5e-05,
|
||||
"cache_read_input_token_cost": 1e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 2e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 1e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
|
||||
"cache_read_input_token_cost_flex": 5e-07,
|
||||
"cache_read_input_token_cost_priority": 2e-06,
|
||||
"input_cost_per_token": 1e-05,
|
||||
"input_cost_per_token_above_272k_tokens": 2e-05,
|
||||
"input_cost_per_token_above_272k_tokens_flex": 1e-05,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 4e-05,
|
||||
"input_cost_per_token_batches": 5e-06,
|
||||
"input_cost_per_token_flex": 5e-06,
|
||||
"input_cost_per_token_priority": 2e-05,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 5e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 7.5e-05,
|
||||
"output_cost_per_token_above_272k_tokens_flex": 3.75e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 0.00015,
|
||||
"output_cost_per_token_batches": 2.5e-05,
|
||||
"output_cost_per_token_flex": 2.5e-05,
|
||||
"output_cost_per_token_priority": 0.0001,
|
||||
"regional_processing_uplift_multiplier_eu": 1.1,
|
||||
"regional_processing_uplift_multiplier_us": 1.1,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
"supports_native_streaming": true,
|
||||
"supports_none_reasoning_effort": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_cache_breakpoint": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"daybreak-red-latest": {
|
||||
"cache_creation_input_token_cost": 1.5625e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 3.125e-05,
|
||||
|
|
|
|||
|
|
@ -29585,6 +29585,74 @@
|
|||
"supports_computer_use": true,
|
||||
"supports_parallel_function_calling": true
|
||||
},
|
||||
"gpt-6-astra": {
|
||||
"cache_creation_input_token_cost": 1.25e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 2.5e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_flex": 1.25e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 5e-05,
|
||||
"cache_creation_input_token_cost_flex": 6.25e-06,
|
||||
"cache_creation_input_token_cost_priority": 2.5e-05,
|
||||
"cache_read_input_token_cost": 1e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 2e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 1e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
|
||||
"cache_read_input_token_cost_flex": 5e-07,
|
||||
"cache_read_input_token_cost_priority": 2e-06,
|
||||
"input_cost_per_token": 1e-05,
|
||||
"input_cost_per_token_above_272k_tokens": 2e-05,
|
||||
"input_cost_per_token_above_272k_tokens_flex": 1e-05,
|
||||
"input_cost_per_token_above_272k_tokens_priority": 4e-05,
|
||||
"input_cost_per_token_batches": 5e-06,
|
||||
"input_cost_per_token_flex": 5e-06,
|
||||
"input_cost_per_token_priority": 2e-05,
|
||||
"litellm_provider": "openai",
|
||||
"max_input_tokens": 922000,
|
||||
"max_output_tokens": 128000,
|
||||
"max_tokens": 128000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 5e-05,
|
||||
"output_cost_per_token_above_272k_tokens": 7.5e-05,
|
||||
"output_cost_per_token_above_272k_tokens_flex": 3.75e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 0.00015,
|
||||
"output_cost_per_token_batches": 2.5e-05,
|
||||
"output_cost_per_token_flex": 2.5e-05,
|
||||
"output_cost_per_token_priority": 0.0001,
|
||||
"regional_processing_uplift_multiplier_eu": 1.1,
|
||||
"regional_processing_uplift_multiplier_us": 1.1,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.01,
|
||||
"search_context_size_low": 0.01,
|
||||
"search_context_size_medium": 0.01
|
||||
},
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/batch",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_computer_use": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
"supports_native_streaming": true,
|
||||
"supports_none_reasoning_effort": true,
|
||||
"supports_parallel_function_calling": true,
|
||||
"supports_pdf_input": true,
|
||||
"supports_prompt_cache_breakpoint": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"supports_web_search": true,
|
||||
"supports_xhigh_reasoning_effort": true
|
||||
},
|
||||
"daybreak-red-latest": {
|
||||
"cache_creation_input_token_cost": 1.5625e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens": 3.125e-05,
|
||||
|
|
|
|||
|
|
@ -1615,6 +1615,71 @@ def test_generic_cost_per_token_gpt56_terra_cache_costs_by_tier_and_context(_loc
|
|||
assert prompt_cost == pytest.approx(expected_prompt_cost)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"service_tier,prompt_tokens,input_rate,cache_write_rate,cache_read_rate,output_rate",
|
||||
[
|
||||
(None, 100000, 1e-5, 1.25e-5, 1e-6, 5e-5),
|
||||
("flex", 100000, 5e-6, 6.25e-6, 5e-7, 2.5e-5),
|
||||
("priority", 100000, 2e-5, 2.5e-5, 2e-6, 1e-4),
|
||||
(None, 300000, 2e-5, 2.5e-5, 2e-6, 7.5e-5),
|
||||
("flex", 300000, 1e-5, 1.25e-5, 1e-6, 3.75e-5),
|
||||
("priority", 300000, 4e-5, 5e-5, 4e-6, 1.5e-4),
|
||||
],
|
||||
)
|
||||
def test_generic_cost_per_token_gpt6_astra_by_tier_and_context(
|
||||
_local_model_cost_map,
|
||||
service_tier,
|
||||
prompt_tokens,
|
||||
input_rate,
|
||||
cache_write_rate,
|
||||
cache_read_rate,
|
||||
output_rate,
|
||||
):
|
||||
"""gpt-6-astra: $10/$50 per 1M with $1 cache read and $12.50 cache write, flex at half,
|
||||
priority at double, and the GPT-5.6 long-context multipliers (2x input and cache, 1.5x
|
||||
output) once the prompt passes 272K tokens."""
|
||||
cached_tokens = 50000
|
||||
cache_write_tokens = 40000
|
||||
text_tokens = prompt_tokens - cached_tokens - cache_write_tokens
|
||||
completion_tokens = 1000
|
||||
usage = Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
cached_tokens=cached_tokens, cache_write_tokens=cache_write_tokens
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model="gpt-6-astra",
|
||||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
service_tier=service_tier,
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(
|
||||
text_tokens * input_rate
|
||||
+ cached_tokens * cache_read_rate
|
||||
+ cache_write_tokens * cache_write_rate
|
||||
)
|
||||
assert completion_cost == pytest.approx(completion_tokens * output_rate)
|
||||
|
||||
|
||||
def test_batch_cost_gpt6_astra_is_half_the_standard_rate(_local_model_cost_map):
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
prompt_cost, completion_cost = batch_cost_calculator(
|
||||
usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
|
||||
model="gpt-6-astra",
|
||||
custom_llm_provider="openai",
|
||||
model_info=litellm.get_model_info("gpt-6-astra"),
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(1000 * 5e-6)
|
||||
assert completion_cost == pytest.approx(500 * 2.5e-5)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["gpt-5.6-cyber", "daybreak-red-latest"])
|
||||
@pytest.mark.parametrize(
|
||||
"prompt_tokens,input_rate,cache_write_rate,cache_read_rate,output_rate",
|
||||
|
|
|
|||
|
|
@ -139,6 +139,7 @@ def test_transform_request_drops_tool_reference_parts():
|
|||
("gpt-5-chat-latest", "max_completion_tokens", "max_tokens"),
|
||||
("gpt-5-chat-2025-08-07", "max_completion_tokens", "max_tokens"),
|
||||
("gpt-5", "max_completion_tokens", "max_tokens"),
|
||||
("gpt-6-astra", "max_completion_tokens", "max_tokens"),
|
||||
("o3-mini", "max_completion_tokens", "max_tokens"),
|
||||
("gpt-4o", "max_tokens", "max_completion_tokens"),
|
||||
],
|
||||
|
|
|
|||
|
|
@ -1718,6 +1718,8 @@ class TestResponsesSurfaceSharesTheEffortRule:
|
|||
("gpt-5.6-sol", None, False),
|
||||
("gpt-5.6-terra", "none", True),
|
||||
("gpt-5.6-terra", "medium", False),
|
||||
("gpt-6-astra", None, False),
|
||||
("gpt-6-astra", "none", True),
|
||||
],
|
||||
)
|
||||
def test_temperature_follows_the_resolved_effort(
|
||||
|
|
|
|||
|
|
@ -54,6 +54,8 @@ GPT5_MODELS = [
|
|||
"gpt-5.6-sol",
|
||||
"gpt-5.6-terra",
|
||||
"gpt-5.6-luna",
|
||||
"gpt-6-astra",
|
||||
"openai/gpt-6-astra",
|
||||
"gpt-5.1-chat", # versioned chat — THE KEY REGRESSION CASE
|
||||
"gpt-5.2-chat", # versioned chat — also a regression case
|
||||
"gpt-5.3-chat", # versioned chat — THE KEY REGRESSION CASE
|
||||
|
|
@ -128,6 +130,8 @@ GPT5_4_PLUS_MODELS = [
|
|||
"gpt-5.6-terra",
|
||||
"gpt-5.6-luna",
|
||||
"openai/gpt-5.6-sol",
|
||||
"gpt-6-astra",
|
||||
"openai/gpt-6-astra",
|
||||
]
|
||||
|
||||
GPT5_PRE_5_4_MODELS = [
|
||||
|
|
|
|||
48
tests/test_litellm/test_gpt_6_astra_model_metadata.py
Normal file
48
tests/test_litellm/test_gpt_6_astra_model_metadata.py
Normal file
|
|
@ -0,0 +1,48 @@
|
|||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
||||
from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config
|
||||
from litellm.types.utils import LlmProviders
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[2]
|
||||
MAIN_PATH = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
BACKUP_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
|
||||
MODEL = "gpt-6-astra"
|
||||
|
||||
|
||||
def _load(path):
|
||||
with open(path) as f:
|
||||
return json.load(f)
|
||||
|
||||
|
||||
def test_gpt_6_astra_backup_matches_main():
|
||||
main_entry = _load(MAIN_PATH).get(MODEL)
|
||||
assert main_entry is not None, f"{MODEL} missing from model_prices_and_context_window.json"
|
||||
assert _load(BACKUP_PATH).get(MODEL) == main_entry
|
||||
|
||||
|
||||
def test_gpt_6_astra_routes_to_openai_on_the_gpt_5_reasoning_path():
|
||||
routed_model, provider, _, _ = get_llm_provider(model=f"openai/{MODEL}")
|
||||
assert (routed_model, provider) == (MODEL, "openai")
|
||||
|
||||
config = ProviderConfigManager.get_provider_chat_config(model=MODEL, provider=LlmProviders.OPENAI)
|
||||
assert isinstance(config, OpenAIGPT5Config)
|
||||
|
||||
|
||||
def test_gpt_6_astra_maps_max_tokens_and_drops_temperature_like_gpt_5():
|
||||
mapped = litellm.get_optional_params(
|
||||
model=MODEL,
|
||||
custom_llm_provider="openai",
|
||||
max_tokens=100,
|
||||
temperature=0.2,
|
||||
reasoning_effort="high",
|
||||
drop_params=True,
|
||||
)
|
||||
|
||||
assert mapped["max_completion_tokens"] == 100
|
||||
assert "max_tokens" not in mapped
|
||||
assert "temperature" not in mapped
|
||||
assert mapped["reasoning_effort"] == "high"
|
||||
|
|
@ -860,7 +860,7 @@ def test_responses_api_bridge_check_gpt_5_4_tools_with_default_reasoning_routes_
|
|||
assert model_info.get("mode") == "responses"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model_name", ["gpt-5.6-sol", "gpt-5.6-luna", "gpt-5.6-terra"])
|
||||
@pytest.mark.parametrize("model_name", ["gpt-5.6-sol", "gpt-5.6-luna", "gpt-5.6-terra", "gpt-6-astra"])
|
||||
def test_responses_api_bridge_check_gpt_5_6_tools_with_default_reasoning_routes_to_responses(
|
||||
monkeypatch, model_name
|
||||
):
|
||||
|
|
|
|||
|
|
@ -52,6 +52,12 @@ PRIORITY_LONG_CONTEXT = {
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 1e-06,
|
||||
},
|
||||
"gpt-6-astra": {
|
||||
"input_cost_per_token_above_272k_tokens_priority": 4e-05,
|
||||
"output_cost_per_token_above_272k_tokens_priority": 0.00015,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": 5e-05,
|
||||
},
|
||||
}
|
||||
|
||||
EXPECTED = {**FLEX_LONG_CONTEXT, **PRIORITY_LONG_CONTEXT}
|
||||
|
|
@ -114,6 +120,7 @@ TIERED_COST_CASES = [
|
|||
("gpt-5.6-sol", "priority", 1.6e-05, 6e-05),
|
||||
("gpt-5.6-terra", "priority", 8e-06, 3.6e-05),
|
||||
("gpt-5.6-luna", "priority", 8e-07, 3.6e-06),
|
||||
("gpt-6-astra", "priority", 4e-05, 0.00015),
|
||||
]
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue