feat(openai): add 272K tier pricing for GPT-5.4/5.4-pro

Prompts >272K input tokens priced at 2x input, 1.5x output for full session
(standard, batch, flex). Applies to models with 1.05M context window (gpt-5.4,
gpt-5.4-pro).

- Add input/output_cost_per_token_above_272k_tokens to model_prices
- Add above_272k fields to ModelInfoBase and get_model_info extraction
- Add test_generic_cost_per_token_gpt54_above_272k_tokens

Made-with: Cursor
This commit is contained in:
Sameer Kankute 2026-03-06 22:26:14 +05:30
parent 4c26c1847b
commit 3c80351b57
6 changed files with 4414 additions and 815 deletions

File diff suppressed because it is too large Load diff

View file

@ -1,47 +1,61 @@
import json
import time
from enum import Enum
from typing import (TYPE_CHECKING, Any, Dict, List, Literal, Mapping, Optional,
Union)
from typing import TYPE_CHECKING, Any, Dict, List, Literal, Mapping, Optional, Union
from openai._models import BaseModel as OpenAIObject
from openai.types.audio.transcription_create_params import \
FileTypes as FileTypes # type: ignore
from openai.types.audio.transcription_create_params import (
FileTypes as FileTypes, # type: ignore
)
from openai.types.chat.chat_completion import ChatCompletion as ChatCompletion
from openai.types.completion_usage import (CompletionTokensDetails,
CompletionUsage,
PromptTokensDetails)
from openai.types.completion_usage import (
CompletionTokensDetails,
CompletionUsage,
PromptTokensDetails,
)
from openai.types.moderation import Categories as Categories
from openai.types.moderation import \
CategoryAppliedInputTypes as CategoryAppliedInputTypes
from openai.types.moderation import (
CategoryAppliedInputTypes as CategoryAppliedInputTypes,
)
from openai.types.moderation import CategoryScores as CategoryScores
from openai.types.moderation_create_response import Moderation as Moderation
from openai.types.moderation_create_response import \
ModerationCreateResponse as ModerationCreateResponse
from openai.types.moderation_create_response import (
ModerationCreateResponse as ModerationCreateResponse,
)
from pydantic import BaseModel, ConfigDict, Field, PrivateAttr, model_validator
from typing_extensions import Required, TypedDict
from litellm._uuid import uuid
from litellm.types.llms.base import (BaseLiteLLMOpenAIResponseObject,
LiteLLMPydanticObjectBase)
from litellm.types.llms.base import (
BaseLiteLLMOpenAIResponseObject,
LiteLLMPydanticObjectBase,
)
from litellm.types.mcp import MCPServerCostInfo
from ..litellm_core_utils.core_helpers import map_finish_reason
from .agents import LiteLLMSendMessageResponse
from .guardrails import GuardrailEventHooks
from .llms.anthropic_messages.anthropic_response import \
AnthropicMessagesResponse
from .llms.anthropic_messages.anthropic_response import AnthropicMessagesResponse
from .llms.base import HiddenParams
from .llms.openai import (AllMessageValues, Batch, ChatCompletionAnnotation,
ChatCompletionRedactedThinkingBlock,
ChatCompletionThinkingBlock,
ChatCompletionToolCallChunk, ChatCompletionToolParam,
ChatCompletionUsageBlock, FileSearchTool,
FineTuningJob, ImageURLListItem,
OpenAIChatCompletionChunk,
OpenAIChatCompletionFinishReason, OpenAIFileObject,
OpenAIRealtimeStreamList, ResponsesAPIResponse,
WebSearchOptions)
from .llms.openai import (
AllMessageValues,
Batch,
ChatCompletionAnnotation,
ChatCompletionRedactedThinkingBlock,
ChatCompletionThinkingBlock,
ChatCompletionToolCallChunk,
ChatCompletionToolParam,
ChatCompletionUsageBlock,
FileSearchTool,
FineTuningJob,
ImageURLListItem,
OpenAIChatCompletionChunk,
OpenAIChatCompletionFinishReason,
OpenAIFileObject,
OpenAIRealtimeStreamList,
ResponsesAPIResponse,
WebSearchOptions,
)
from .rerank import RerankResponse as RerankResponse
if TYPE_CHECKING:
@ -150,12 +164,16 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
float
] # OpenAI priority service tier pricing
cache_read_input_token_cost_above_200k_tokens: Optional[float]
cache_read_input_token_cost_above_272k_tokens: Optional[float]
input_cost_per_character: Optional[float] # only for vertex ai models
input_cost_per_audio_token: Optional[float]
input_cost_per_token_above_128k_tokens: Optional[float] # only for vertex ai models
input_cost_per_token_above_200k_tokens: Optional[
float
] # only for vertex ai gemini-2.5-pro models
input_cost_per_token_above_272k_tokens: Optional[
float
] # GPT-5.4/5.4-pro: prompts >272K priced at 2x input
input_cost_per_character_above_128k_tokens: Optional[
float
] # only for vertex ai models
@ -180,6 +198,9 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
output_cost_per_token_above_200k_tokens: Optional[
float
] # only for vertex ai gemini-2.5-pro models
output_cost_per_token_above_272k_tokens: Optional[
float
] # GPT-5.4/5.4-pro: prompts >272K priced at 1.5x output
output_cost_per_character_above_128k_tokens: Optional[
float
] # only for vertex ai models

View file

@ -5599,6 +5599,9 @@ def _get_model_info_helper( # noqa: PLR0915
cache_read_input_token_cost_above_200k_tokens=_model_info.get(
"cache_read_input_token_cost_above_200k_tokens", None
),
cache_read_input_token_cost_above_272k_tokens=_model_info.get(
"cache_read_input_token_cost_above_272k_tokens", None
),
cache_read_input_token_cost_flex=_model_info.get(
"cache_read_input_token_cost_flex", None
),
@ -5617,6 +5620,9 @@ def _get_model_info_helper( # noqa: PLR0915
input_cost_per_token_above_200k_tokens=_model_info.get(
"input_cost_per_token_above_200k_tokens", None
),
input_cost_per_token_above_272k_tokens=_model_info.get(
"input_cost_per_token_above_272k_tokens", None
),
input_cost_per_query=_model_info.get("input_cost_per_query", None),
input_cost_per_second=_model_info.get("input_cost_per_second", None),
input_cost_per_audio_token=_model_info.get(
@ -5663,6 +5669,9 @@ def _get_model_info_helper( # noqa: PLR0915
output_cost_per_token_above_200k_tokens=_model_info.get(
"output_cost_per_token_above_200k_tokens", None
),
output_cost_per_token_above_272k_tokens=_model_info.get(
"output_cost_per_token_above_272k_tokens", None
),
output_cost_per_second=_model_info.get("output_cost_per_second", None),
output_cost_per_video_per_second=_model_info.get(
"output_cost_per_video_per_second", None

File diff suppressed because it is too large Load diff

View file

@ -23,8 +23,8 @@ sys.path.insert(
) # Adds the parent directory to the system path
from litellm.litellm_core_utils.llm_cost_calc.utils import (
_calculate_input_cost,
PromptTokensDetailsResult,
_calculate_input_cost,
calculate_cache_writing_cost,
generic_cost_per_token,
)
@ -239,6 +239,32 @@ def test_generic_cost_per_token_above_200k_tokens():
)
def test_generic_cost_per_token_gpt54_above_272k_tokens():
"""GPT-5.4/5.4-pro: prompts >272K input tokens priced at 2x input, 1.5x output."""
model = "gpt-5.4"
custom_llm_provider = "openai"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_cost_map = litellm.model_cost[model]
prompt_tokens = 273000 # Above 272K threshold
completion_tokens = 1000
usage = Usage(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
total_tokens=prompt_tokens + completion_tokens,
)
prompt_cost, completion_cost = generic_cost_per_token(
model=model,
usage=usage,
custom_llm_provider=custom_llm_provider,
)
expected_prompt = model_cost_map["input_cost_per_token_above_272k_tokens"] * prompt_tokens
expected_completion = model_cost_map["output_cost_per_token_above_272k_tokens"] * completion_tokens
assert round(prompt_cost, 10) == round(expected_prompt, 10)
assert round(completion_cost, 10) == round(expected_completion, 10)
def test_generic_cost_per_token_anthropic_prompt_caching():
model = "claude-sonnet-4@20250514"
usage = Usage(

View file

@ -13,12 +13,12 @@ sys.path.insert(
import litellm
from litellm.proxy.utils import is_valid_api_key
from litellm.types.utils import (
CallTypes,
Delta,
LlmProviders,
ModelResponseStream,
StreamingChoices,
)
from litellm.types.utils import CallTypes
from litellm.utils import (
ProviderConfigManager,
TextCompletionStreamWrapper,
@ -96,6 +96,19 @@ def test_supports_function_calling_github_anthropic_alias():
)
def test_supports_function_calling_deepinfra_llama():
"""Test that deepinfra Llama models correctly report function calling support.
Regression test for https://github.com/BerriAI/litellm/issues/22619
"""
assert (
litellm.utils.supports_function_calling(
model="deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo"
)
is True
)
def test_supports_function_calling_unknown_github_alias_returns_false():
assert (
litellm.utils.supports_function_calling(
@ -500,6 +513,8 @@ def validate_model_cost_values(model_data, exceptions=None):
"output_cost_per_token_above_128k_tokens",
"input_cost_per_token_above_200k_tokens",
"output_cost_per_token_above_200k_tokens",
"input_cost_per_token_above_272k_tokens",
"output_cost_per_token_above_272k_tokens",
"input_cost_per_character_above_128k_tokens",
"output_cost_per_character_above_128k_tokens",
"input_cost_per_image_above_128k_tokens",
@ -590,8 +605,10 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"cache_creation_input_token_cost_above_200k_tokens": {"type": "number"},
"cache_read_input_token_cost": {"type": "number"},
"cache_read_input_token_cost_above_200k_tokens": {"type": "number"},
"cache_read_input_token_cost_above_272k_tokens": {"type": "number"},
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": {"type": "number"},
"cache_read_input_audio_token_cost": {"type": "number"},
"cache_read_input_token_cost_per_audio_token": {"type": "number"},
"cache_read_input_image_token_cost": {"type": "number"},
"deprecation_date": {"type": "string"},
"input_cost_per_audio_per_second": {"type": "number"},
@ -604,12 +621,20 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"input_cost_per_image_above_128k_tokens": {"type": "number"},
"input_cost_per_image_token": {"type": "number"},
"input_cost_per_token_above_200k_tokens": {"type": "number"},
"input_cost_per_token_above_272k_tokens": {"type": "number"},
"cache_read_input_token_cost_flex": {"type": "number"},
"cache_read_input_token_cost_priority": {"type": "number"},
"cache_read_input_token_cost_above_200k_tokens_priority": {"type": "number"},
"cache_read_input_token_cost_above_272k_tokens_priority": {"type": "number"},
"input_cost_per_token_flex": {"type": "number"},
"input_cost_per_token_priority": {"type": "number"},
"input_cost_per_token_above_200k_tokens_priority": {"type": "number"},
"input_cost_per_token_above_272k_tokens_priority": {"type": "number"},
"input_cost_per_audio_token_priority": {"type": "number"},
"output_cost_per_token_flex": {"type": "number"},
"output_cost_per_token_priority": {"type": "number"},
"output_cost_per_token_above_200k_tokens_priority": {"type": "number"},
"output_cost_per_token_above_272k_tokens_priority": {"type": "number"},
"input_cost_per_pixel": {"type": "number"},
"input_cost_per_query": {"type": "number"},
"input_cost_per_request": {"type": "number"},
@ -644,6 +669,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"max_video_length": {"type": "number"},
"max_videos_per_prompt": {"type": "number"},
"metadata": {"type": "object"},
"provider_specific_entry": {"type": "object"},
"mode": {
"type": "string",
"enum": [
@ -658,6 +684,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"video_generation",
"moderation",
"rerank",
"realtime",
"responses",
"ocr",
"search",
@ -674,6 +701,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"output_cost_per_token": {"type": "number"},
"output_cost_per_token_above_128k_tokens": {"type": "number"},
"output_cost_per_token_above_200k_tokens": {"type": "number"},
"output_cost_per_token_above_272k_tokens": {"type": "number"},
"output_cost_per_image_above_1024_and_1024_pixels": {"type": "number"},
"output_cost_per_image_above_1024_and_1024_pixels_and_premium_image": {
"type": "number"
@ -697,6 +725,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_audio_input": {"type": "boolean"},
"supports_audio_output": {"type": "boolean"},
"supports_embedding_image_input": {"type": "boolean"},
"supports_code_execution": {"type": "boolean"},
"supports_file_search": {"type": "boolean"},
"supports_function_calling": {"type": "boolean"},
"supports_image_input": {"type": "boolean"},
"supports_parallel_function_calling": {"type": "boolean"},
@ -710,10 +740,13 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_web_search": {"type": "boolean"},
"supports_url_context": {"type": "boolean"},
"supports_reasoning": {"type": "boolean"},
"supports_none_reasoning_effort": {"type": "boolean"},
"supports_xhigh_reasoning_effort": {"type": "boolean"},
"supports_service_tier": {"type": "boolean"},
"supports_preset": {"type": "boolean"},
"tool_use_system_prompt_tokens": {"type": "number"},
"tpm": {"type": "number"},
"provider_specific_entry": {"type": "object"},
"supported_endpoints": {
"type": "array",
"items": {
@ -802,8 +835,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
},
}
prod_json = "./model_prices_and_context_window.json"
# prod_json = "../../model_prices_and_context_window.json"
prod_json = os.path.join(os.path.dirname(__file__), "..", "..", "model_prices_and_context_window.json")
with open(prod_json, "r") as model_prices_file:
actual_json = json.load(model_prices_file)
assert isinstance(actual_json, dict)
@ -1134,20 +1166,25 @@ def test_pre_process_non_default_params(model, custom_llm_provider):
provider_config=provider_config,
)
print(processed_non_default_params)
# Vertex AI / Gemini uses Pydantic's model_json_schema() which doesn't
# include additionalProperties: False (Gemini rejects it). Other
# providers use OpenAI's to_strict_json_schema() which does.
expected_schema = {
"properties": {
"x": {"title": "X", "type": "string"},
"y": {"title": "Y", "type": "string"},
},
"required": ["x", "y"],
"title": "ResponseFormat",
"type": "object",
}
if custom_llm_provider not in ("vertex_ai", "vertex_ai_beta", "gemini"):
expected_schema["additionalProperties"] = False
assert processed_non_default_params == {
"response_format": {
"type": "json_schema",
"json_schema": {
"schema": {
"properties": {
"x": {"title": "X", "type": "string"},
"y": {"title": "Y", "type": "string"},
},
"required": ["x", "y"],
"title": "ResponseFormat",
"type": "object",
"additionalProperties": False,
},
"schema": expected_schema,
"name": "ResponseFormat",
"strict": True,
},
@ -2337,7 +2374,7 @@ def test_register_model_with_scientific_notation():
Test that the register_model function can handle scientific notation in the model name.
"""
import uuid
# Use a truly unique model name with uuid to avoid conflicts when tests run in parallel
test_model_name = f"test-scientific-notation-model-{uuid.uuid4().hex[:12]}"
@ -2372,6 +2409,64 @@ def test_register_model_with_scientific_notation():
_invalidate_model_cost_lowercase_map()
def test_register_model_openrouter_without_slash():
"""
Test that register_model handles openrouter models without '/' in the name.
Fixes https://github.com/BerriAI/litellm/issues/18936
Previously, the code did `split_string[1]` which would fail with IndexError
when the model name didn't contain '/'. Now it uses `split_string[-1]` which
always works.
"""
# Clear any existing entries
litellm.openrouter_models.discard("my-custom-alias")
litellm.openrouter_models.discard("gpt-4")
litellm.openrouter_models.discard("openai/gpt-4")
# Test 1: Model name without '/' (this was the bug - would raise IndexError)
litellm.register_model(
{
"my-custom-alias": {
"max_tokens": 8192,
"input_cost_per_token": 0.00001,
"output_cost_per_token": 0.00002,
"litellm_provider": "openrouter",
"mode": "chat",
},
}
)
assert "my-custom-alias" in litellm.openrouter_models
# Test 2: Model name with single '/' (openrouter/model format)
litellm.register_model(
{
"openrouter/gpt-4": {
"max_tokens": 8192,
"input_cost_per_token": 0.00001,
"output_cost_per_token": 0.00002,
"litellm_provider": "openrouter",
"mode": "chat",
},
}
)
assert "gpt-4" in litellm.openrouter_models
# Test 3: Model name with double '/' (openrouter/provider/model format)
litellm.register_model(
{
"openrouter/openai/gpt-4-turbo": {
"max_tokens": 8192,
"input_cost_per_token": 0.00001,
"output_cost_per_token": 0.00002,
"litellm_provider": "openrouter",
"mode": "chat",
},
}
)
assert "openai/gpt-4-turbo" in litellm.openrouter_models
def test_reasoning_content_preserved_in_text_completion_wrapper():
"""Ensure reasoning_content is copied from delta to text_choices."""
chunk = ModelResponseStream(
@ -2981,8 +3076,8 @@ class TestProxyLoggingBudgetAlerts:
via metadata.soft_budget_alerting_emails to work even when global alerting is disabled.
"""
from litellm.caching.caching import DualCache
from litellm.proxy.utils import ProxyLogging
from litellm.proxy._types import CallInfo, Litellm_EntityType
from litellm.proxy.utils import ProxyLogging
proxy_logging = ProxyLogging(user_api_key_cache=DualCache())
proxy_logging.alerting = None # Global alerting is disabled
@ -3018,8 +3113,8 @@ class TestProxyLoggingBudgetAlerts:
and do not send emails when alerting is None.
"""
from litellm.caching.caching import DualCache
from litellm.proxy.utils import ProxyLogging
from litellm.proxy._types import CallInfo, Litellm_EntityType
from litellm.proxy.utils import ProxyLogging
proxy_logging = ProxyLogging(user_api_key_cache=DualCache())
proxy_logging.alerting = None
@ -3050,8 +3145,8 @@ class TestProxyLoggingBudgetAlerts:
Test that soft_budget alerts with empty alert_emails list still respect alerting=None.
"""
from litellm.caching.caching import DualCache
from litellm.proxy.utils import ProxyLogging
from litellm.proxy._types import CallInfo, Litellm_EntityType
from litellm.proxy.utils import ProxyLogging
proxy_logging = ProxyLogging(user_api_key_cache=DualCache())
proxy_logging.alerting = None
@ -3554,3 +3649,43 @@ class TestMetadataNoneHandling:
litellm_params = {"metadata": None}
metadata = litellm_params.get("metadata") or {}
assert metadata == {}
class TestValidateAndFixThinkingParam:
"""Tests for validate_and_fix_thinking_param."""
def test_none_returns_none(self):
from litellm.utils import validate_and_fix_thinking_param
assert validate_and_fix_thinking_param(thinking=None) is None
def test_already_snake_case(self):
from litellm.utils import validate_and_fix_thinking_param
thinking = {"type": "enabled", "budget_tokens": 32000}
result = validate_and_fix_thinking_param(thinking=thinking)
assert result == {"type": "enabled", "budget_tokens": 32000}
def test_camel_case_normalized(self):
from litellm.utils import validate_and_fix_thinking_param
thinking = {"type": "enabled", "budgetTokens": 32000}
result = validate_and_fix_thinking_param(thinking=thinking)
assert result == {"type": "enabled", "budget_tokens": 32000}
assert "budgetTokens" not in result
def test_both_keys_snake_case_wins(self):
from litellm.utils import validate_and_fix_thinking_param
thinking = {"type": "enabled", "budget_tokens": 10000, "budgetTokens": 50000}
result = validate_and_fix_thinking_param(thinking=thinking)
assert result == {"type": "enabled", "budget_tokens": 10000}
assert "budgetTokens" not in result
def test_original_dict_not_mutated(self):
from litellm.utils import validate_and_fix_thinking_param
thinking = {"type": "enabled", "budgetTokens": 32000}
validate_and_fix_thinking_param(thinking=thinking)
assert "budgetTokens" in thinking
assert "budget_tokens" not in thinking