From 9819e4e0f8c161a3d7fef916a9bf73c1d068dd56 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:39:35 -0700 Subject: [PATCH] fix(cost): add the gpt-5.5-pro batch long-context tier and ignore malformed batch tier keys --- basedpyright-code-budget.json | 2 +- litellm/cost_calculator.py | 4 +- .../litellm_core_utils/llm_cost_calc/utils.py | 7 ++- ...odel_prices_and_context_window_backup.json | 4 ++ model_prices_and_context_window.json | 4 ++ tests/test_litellm/test_cost_calculator.py | 48 +++++++++++++++++++ 6 files changed, 62 insertions(+), 7 deletions(-) diff --git a/basedpyright-code-budget.json b/basedpyright-code-budget.json index b876cf2d69a..0653ebe63c0 100644 --- a/basedpyright-code-budget.json +++ b/basedpyright-code-budget.json @@ -3,7 +3,7 @@ "limit": 13429 }, "reportArgumentType": { - "limit": 2198 + "limit": 1913 }, "reportAssignmentType": { "limit": 319 diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index e76947f5295..096d70d7a60 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -27,11 +27,11 @@ from litellm.litellm_core_utils.llm_cost_calc.usage_object_transformation import from litellm.litellm_core_utils.llm_cost_calc.utils import ( CostCalculatorUtils, _generic_cost_per_character, - _get_batch_cost_rates, _get_regional_uplift_multiplier, _get_service_tier_cost_key, calculate_cost_component, generic_cost_per_token, + get_batch_cost_rates, get_billable_input_tokens, get_token_type_cost_breakdown, parse_prompt_tokens_details, @@ -2242,7 +2242,7 @@ def batch_cost_calculator( if not model_info: return 0.0, 0.0 - input_cost_per_token_batches, output_cost_per_token_batches = _get_batch_cost_rates( + input_cost_per_token_batches, output_cost_per_token_batches = get_batch_cost_rates( model_info, usage, custom_llm_provider ) input_cost_per_token: Final = model_info.get("input_cost_per_token") diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index acb0251b021..2dddf726321 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -62,6 +62,7 @@ _SERVICE_TIER_TO_COST_KEY_SUFFIX: Final[Mapping[str, str]] = MappingProxyType( _INCLUSIVE_THRESHOLD_PROVIDERS: Final = frozenset({"xai"}) _BATCH_KEY_SUFFIX: Final = "_batches" +_BATCH_TIER_INPUT_KEY: Final = re.compile(r"^input_cost_per_token_above_\d+k?_tokens_batches$") _NON_STANDARD_THRESHOLD_SUFFIXES: Final = (*_SERVICE_TIER_SUFFIXES, _BATCH_KEY_SUFFIX) @@ -252,14 +253,12 @@ def _batch_rate(model_info: ModelInfo, key: str) -> float | None: return value if isinstance(value, (int, float)) else None -def _get_batch_cost_rates( +def get_batch_cost_rates( model_info: ModelInfo, usage: Usage, custom_llm_provider: str | None ) -> tuple[float | None, float | None]: inclusive: Final = _uses_inclusive_token_thresholds(custom_llm_provider) tier_input_keys: Final = tuple( - key - for key, value in model_info.items() - if key.startswith("input_cost_per_token_above_") and key.endswith(_BATCH_KEY_SUFFIX) and value is not None + key for key, value in model_info.items() if _BATCH_TIER_INPUT_KEY.match(key) and value is not None ) crossed_input_key: Final = next( ( diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index e4a67f2cf8b..acb988ef953 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -30839,6 +30839,7 @@ "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, "input_cost_per_token_batches": 1.5e-05, + "input_cost_per_token_above_272k_tokens_batches": 3e-05, "litellm_provider": "openai", "max_input_tokens": 1050000, "max_output_tokens": 128000, @@ -30848,6 +30849,7 @@ "output_cost_per_token_above_272k_tokens": 0.00027, "output_cost_per_token_flex": 9e-05, "output_cost_per_token_batches": 9e-05, + "output_cost_per_token_above_272k_tokens_batches": 0.000135, "regional_processing_uplift_multiplier_eu": 1.1, "regional_processing_uplift_multiplier_us": 1.1, "search_context_cost_per_query": { @@ -30889,6 +30891,7 @@ "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, "input_cost_per_token_batches": 1.5e-05, + "input_cost_per_token_above_272k_tokens_batches": 3e-05, "litellm_provider": "openai", "max_input_tokens": 1050000, "max_output_tokens": 128000, @@ -30898,6 +30901,7 @@ "output_cost_per_token_above_272k_tokens": 0.00027, "output_cost_per_token_flex": 9e-05, "output_cost_per_token_batches": 9e-05, + "output_cost_per_token_above_272k_tokens_batches": 0.000135, "regional_processing_uplift_multiplier_eu": 1.1, "regional_processing_uplift_multiplier_us": 1.1, "search_context_cost_per_query": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index e4a67f2cf8b..acb988ef953 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -30839,6 +30839,7 @@ "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, "input_cost_per_token_batches": 1.5e-05, + "input_cost_per_token_above_272k_tokens_batches": 3e-05, "litellm_provider": "openai", "max_input_tokens": 1050000, "max_output_tokens": 128000, @@ -30848,6 +30849,7 @@ "output_cost_per_token_above_272k_tokens": 0.00027, "output_cost_per_token_flex": 9e-05, "output_cost_per_token_batches": 9e-05, + "output_cost_per_token_above_272k_tokens_batches": 0.000135, "regional_processing_uplift_multiplier_eu": 1.1, "regional_processing_uplift_multiplier_us": 1.1, "search_context_cost_per_query": { @@ -30889,6 +30891,7 @@ "input_cost_per_token_above_272k_tokens": 6e-05, "input_cost_per_token_flex": 1.5e-05, "input_cost_per_token_batches": 1.5e-05, + "input_cost_per_token_above_272k_tokens_batches": 3e-05, "litellm_provider": "openai", "max_input_tokens": 1050000, "max_output_tokens": 128000, @@ -30898,6 +30901,7 @@ "output_cost_per_token_above_272k_tokens": 0.00027, "output_cost_per_token_flex": 9e-05, "output_cost_per_token_batches": 9e-05, + "output_cost_per_token_above_272k_tokens_batches": 0.000135, "regional_processing_uplift_multiplier_eu": 1.1, "regional_processing_uplift_multiplier_us": 1.1, "search_context_cost_per_query": { diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index ffdc72d7383..9259d28fbac 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -1,6 +1,7 @@ import json from pathlib import Path +from typing import cast import pytest @@ -4546,3 +4547,50 @@ def test_regular_path_never_bills_the_batch_tier_keys(_local_model_cost_map, mon assert prompt_cost == pytest.approx(300_035 * 2e-6) assert completion_cost == pytest.approx(64 * 8e-6) + + +def _lacks_half_price_batch_tier(entry: dict) -> bool: + return any( + entry.get(f"{side}_cost_per_token_above_272k_tokens_batches") + != entry[f"{side}_cost_per_token_above_272k_tokens"] / 2 + for side in ("input", "output") + ) + + +def test_every_openai_entry_with_long_context_and_batch_rates_carries_the_batch_tier(_local_model_cost_map): + offenders = [ + name + for name, entry in litellm.model_cost.items() + if isinstance(entry, dict) + and entry.get("litellm_provider") == "openai" + and entry.get("input_cost_per_token_above_272k_tokens") is not None + and entry.get("input_cost_per_token_batches") is not None + and _lacks_half_price_batch_tier(entry) + ] + + assert offenders == [] + + +def test_batch_cost_calculator_ignores_malformed_batch_tier_keys(): + from litellm.cost_calculator import batch_cost_calculator + + usage = Usage(prompt_tokens=300_035, completion_tokens=64, total_tokens=300_099) + model_info = cast( + ModelInfo, + { + "input_cost_per_token": 2e-6, + "output_cost_per_token": 8e-6, + "input_cost_per_token_batches": 1e-6, + "output_cost_per_token_batches": 4e-6, + "input_cost_per_token_above_272k_tokens_batches": 2e-6, + "output_cost_per_token_above_272k_tokens_batches": 6e-6, + "input_cost_per_token_above_lots_tokens_batches": 1.0, + }, + ) + + prompt_cost, completion_cost = batch_cost_calculator( + usage=usage, model="gpt-5.4", custom_llm_provider="openai", model_info=model_info + ) + + assert prompt_cost == pytest.approx(300_035 * 2e-6) + assert completion_cost == pytest.approx(64 * 6e-6)