diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index 3203c35b64d..21ed49afb54 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -169,6 +169,7 @@ COST_DESCRIPTIONS: dict[str, str] = { "cache_read_input_token_cost": "USD per prompt token served from the provider's prompt cache.", "input_cost_per_token_batches": "USD per prompt token via the provider's batch API.", "output_cost_per_token_batches": "USD per generated token via the provider's batch API.", + "cache_read_input_token_cost_batches": "USD per cached prompt token via the provider's batch API.", } diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 096d70d7a60..a43f1af0940 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -2242,15 +2242,13 @@ def batch_cost_calculator( if not model_info: return 0.0, 0.0 - input_cost_per_token_batches, output_cost_per_token_batches = get_batch_cost_rates( - model_info, usage, custom_llm_provider - ) + batch_rates: Final = get_batch_cost_rates(model_info, usage, custom_llm_provider) input_cost_per_token: Final = model_info.get("input_cost_per_token") output_cost_per_token: Final = model_info.get("output_cost_per_token") total_prompt_cost = 0.0 total_completion_cost = 0.0 - if input_cost_per_token_batches is not None: - total_prompt_cost = usage.prompt_tokens * input_cost_per_token_batches + if batch_rates.input is not None: + total_prompt_cost = _batch_prompt_cost(usage, batch_rates.input, batch_rates.cache_read) elif input_cost_per_token: details: Final = parse_prompt_tokens_details(usage) cache_read_tokens: Final = details["cache_hit_tokens"] @@ -2269,8 +2267,8 @@ def batch_cost_calculator( cache_creation_cost: Final = model_info.get("cache_creation_input_token_cost") or input_cost_per_token total_prompt_cost += cache_creation_tokens * cache_creation_cost / 2 - if output_cost_per_token_batches is not None: - total_completion_cost = usage.completion_tokens * output_cost_per_token_batches + if batch_rates.output is not None: + total_completion_cost = usage.completion_tokens * batch_rates.output elif output_cost_per_token: total_completion_cost = ( usage.completion_tokens * (output_cost_per_token) / 2 @@ -2284,6 +2282,13 @@ def batch_cost_calculator( return total_prompt_cost, total_completion_cost +def _batch_prompt_cost(usage: Usage, input_rate: float, cache_read_rate: float | None) -> float: + if cache_read_rate is None: + return usage.prompt_tokens * input_rate + cached_tokens: Final = parse_prompt_tokens_details(usage)["cache_hit_tokens"] + return get_billable_input_tokens(usage) * input_rate + cached_tokens * cache_read_rate + + def _attribute_value(obj: object, name: str) -> object: return getattr(obj, name) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index 2dddf726321..2b9fb004bb1 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -248,14 +248,31 @@ def _prompt_exceeds_threshold(prompt_tokens: int, threshold: float, inclusive: b return prompt_tokens > threshold or (inclusive and prompt_tokens == threshold) +@dataclass(frozen=True, slots=True) +class BatchCostRates: + input: float | None + output: float | None + cache_read: float | None + + def _batch_rate(model_info: ModelInfo, key: str) -> float | None: value: Final = model_info.get(key) - return value if isinstance(value, (int, float)) else None + if isinstance(value, (int, float)): + return float(value) + if not isinstance(value, str): + return None + try: + return float(value) + except ValueError: + return None -def get_batch_cost_rates( - model_info: ModelInfo, usage: Usage, custom_llm_provider: str | None -) -> tuple[float | None, float | None]: +def _batch_tier_rate(model_info: ModelInfo, tier_key: str, flat_key: str) -> float | None: + tier_rate: Final = _batch_rate(model_info, tier_key) + return _batch_rate(model_info, flat_key) if tier_rate is None else tier_rate + + +def get_batch_cost_rates(model_info: ModelInfo, usage: Usage, custom_llm_provider: str | None) -> BatchCostRates: inclusive: Final = _uses_inclusive_token_thresholds(custom_llm_provider) tier_input_keys: Final = tuple( key for key, value in model_info.items() if _BATCH_TIER_INPUT_KEY.match(key) and value is not None @@ -268,12 +285,23 @@ def get_batch_cost_rates( ), None, ) - flat_output_rate: Final = _batch_rate(model_info, "output_cost_per_token_batches") if crossed_input_key is None: - return _batch_rate(model_info, "input_cost_per_token_batches"), flat_output_rate - tier_input_rate: Final = _batch_rate(model_info, crossed_input_key) - tier_output_rate: Final = _batch_rate(model_info, crossed_input_key.replace("input_", "output_", 1)) - return tier_input_rate, flat_output_rate if tier_output_rate is None else tier_output_rate + return BatchCostRates( + input=_batch_rate(model_info, "input_cost_per_token_batches"), + output=_batch_rate(model_info, "output_cost_per_token_batches"), + cache_read=_batch_rate(model_info, "cache_read_input_token_cost_batches"), + ) + return BatchCostRates( + input=_batch_tier_rate(model_info, crossed_input_key, "input_cost_per_token_batches"), + output=_batch_tier_rate( + model_info, crossed_input_key.replace("input_", "output_", 1), "output_cost_per_token_batches" + ), + cache_read=_batch_tier_rate( + model_info, + crossed_input_key.replace("input_cost_per_token", "cache_read_input_token_cost", 1), + "cache_read_input_token_cost_batches", + ), + ) def _select_priced_tier(model_info: ModelInfo, usage: Usage) -> dict | None: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index acb988ef953..8c862c6ead5 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -30152,6 +30152,8 @@ "cache_read_input_token_cost_above_272k_tokens": 2e-06, "cache_read_input_token_cost_above_272k_tokens_flex": 1e-06, "cache_read_input_token_cost_above_272k_tokens_priority": 4e-06, + "cache_read_input_token_cost_batches": 5e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 1e-06, "cache_read_input_token_cost_flex": 5e-07, "cache_read_input_token_cost_priority": 2e-06, "input_cost_per_token": 1e-05, @@ -30223,6 +30225,8 @@ "cache_read_input_token_cost_above_272k_tokens": 8e-07, "cache_read_input_token_cost_above_272k_tokens_flex": 4e-07, "cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06, + "cache_read_input_token_cost_batches": 2e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 4e-07, "cache_read_input_token_cost_flex": 2e-07, "cache_read_input_token_cost_priority": 8e-07, "input_cost_per_token": 4e-06, @@ -30292,6 +30296,8 @@ "cache_read_input_token_cost_above_272k_tokens": 8e-07, "cache_read_input_token_cost_above_272k_tokens_flex": 4e-07, "cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06, + "cache_read_input_token_cost_batches": 2e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 4e-07, "cache_read_input_token_cost_flex": 2e-07, "cache_read_input_token_cost_priority": 8e-07, "input_cost_per_token": 4e-06, @@ -30362,6 +30368,8 @@ "cache_read_input_token_cost_above_272k_tokens": 4e-07, "cache_read_input_token_cost_above_272k_tokens_flex": 2e-07, "cache_read_input_token_cost_above_272k_tokens_priority": 8e-07, + "cache_read_input_token_cost_batches": 1e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 2e-07, "cache_read_input_token_cost_flex": 1e-07, "cache_read_input_token_cost_priority": 4e-07, "input_cost_per_token": 2e-06, @@ -30431,6 +30439,8 @@ "cache_read_input_token_cost_above_272k_tokens": 4e-08, "cache_read_input_token_cost_above_272k_tokens_flex": 2e-08, "cache_read_input_token_cost_above_272k_tokens_priority": 8e-08, + "cache_read_input_token_cost_batches": 1e-08, + "cache_read_input_token_cost_above_272k_tokens_batches": 2e-08, "cache_read_input_token_cost_flex": 1e-08, "cache_read_input_token_cost_priority": 4e-08, "input_cost_per_token": 2e-07, @@ -30717,6 +30727,8 @@ "gpt-5.5": { "cache_read_input_token_cost": 5e-07, "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_batches": 2.5e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 5e-07, "cache_read_input_token_cost_flex": 2.5e-07, "cache_read_input_token_cost_priority": 1.25e-06, "input_cost_per_token": 5e-06, @@ -30776,6 +30788,8 @@ "gpt-5.5-2026-04-23": { "cache_read_input_token_cost": 5e-07, "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_batches": 2.5e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 5e-07, "cache_read_input_token_cost_flex": 2.5e-07, "cache_read_input_token_cost_priority": 1.25e-06, "input_cost_per_token": 5e-06, @@ -30939,6 +30953,8 @@ "gpt-5.4": { "cache_read_input_token_cost": 2.5e-07, "cache_read_input_token_cost_above_272k_tokens": 5e-07, + "cache_read_input_token_cost_batches": 1.25e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07, "cache_read_input_token_cost_flex": 1.3e-07, "cache_read_input_token_cost_priority": 5e-07, "input_cost_per_token": 2.5e-06, @@ -30993,6 +31009,8 @@ "gpt-5.4-2026-03-05": { "cache_read_input_token_cost": 2.5e-07, "cache_read_input_token_cost_above_272k_tokens": 5e-07, + "cache_read_input_token_cost_batches": 1.25e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07, "cache_read_input_token_cost_flex": 1.3e-07, "cache_read_input_token_cost_priority": 5e-07, "input_cost_per_token": 2.5e-06, diff --git a/litellm/types/utils.py b/litellm/types/utils.py index f61183542df..c2e0179e103 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -256,6 +256,8 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False): cache_read_input_token_cost_above_272k_tokens_priority: float | None cache_read_input_token_cost_above_272k_tokens_flex: float | None cache_read_input_token_cost_above_512k_tokens: float | None + cache_read_input_token_cost_batches: ReadOnly[float | None] + cache_read_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None] # Smallest prefix this model will actually cache, whatever caching mechanism its provider uses. # Absent means the provider-agnostic default applies; see MINIMUM_PROMPT_CACHE_TOKEN_COUNT. prompt_cache_min_tokens: int | None @@ -3517,6 +3519,8 @@ class CustomPricingLiteLLMParams(MirroredPricingParams): cache_read_input_token_cost_above_200k_tokens_priority: float | None = None cache_read_input_token_cost_above_272k_tokens_priority: float | None = None cache_read_input_token_cost_above_272k_tokens_flex: float | None = None + cache_read_input_token_cost_batches: float | None = None + cache_read_input_token_cost_above_272k_tokens_batches: float | None = None cache_read_input_audio_token_cost: float | None = None input_cost_per_character_above_128k_tokens: float | None = None input_cost_per_audio_token: float | None = None diff --git a/litellm/utils.py b/litellm/utils.py index ca18af8ee97..b041d44b7d5 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -5818,6 +5818,10 @@ def _get_model_info_helper( cache_read_input_token_cost_flex=_model_info.get("cache_read_input_token_cost_flex", None), cache_read_input_token_cost_priority=_model_info.get("cache_read_input_token_cost_priority", None), cache_read_input_token_cost_ultrafast=_model_info.get("cache_read_input_token_cost_ultrafast", None), + cache_read_input_token_cost_batches=_model_info.get("cache_read_input_token_cost_batches"), + cache_read_input_token_cost_above_272k_tokens_batches=_model_info.get( + "cache_read_input_token_cost_above_272k_tokens_batches" + ), cache_creation_input_token_cost_above_1hr=_model_info.get( "cache_creation_input_token_cost_above_1hr", None ), diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index acb988ef953..8c862c6ead5 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -30152,6 +30152,8 @@ "cache_read_input_token_cost_above_272k_tokens": 2e-06, "cache_read_input_token_cost_above_272k_tokens_flex": 1e-06, "cache_read_input_token_cost_above_272k_tokens_priority": 4e-06, + "cache_read_input_token_cost_batches": 5e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 1e-06, "cache_read_input_token_cost_flex": 5e-07, "cache_read_input_token_cost_priority": 2e-06, "input_cost_per_token": 1e-05, @@ -30223,6 +30225,8 @@ "cache_read_input_token_cost_above_272k_tokens": 8e-07, "cache_read_input_token_cost_above_272k_tokens_flex": 4e-07, "cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06, + "cache_read_input_token_cost_batches": 2e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 4e-07, "cache_read_input_token_cost_flex": 2e-07, "cache_read_input_token_cost_priority": 8e-07, "input_cost_per_token": 4e-06, @@ -30292,6 +30296,8 @@ "cache_read_input_token_cost_above_272k_tokens": 8e-07, "cache_read_input_token_cost_above_272k_tokens_flex": 4e-07, "cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06, + "cache_read_input_token_cost_batches": 2e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 4e-07, "cache_read_input_token_cost_flex": 2e-07, "cache_read_input_token_cost_priority": 8e-07, "input_cost_per_token": 4e-06, @@ -30362,6 +30368,8 @@ "cache_read_input_token_cost_above_272k_tokens": 4e-07, "cache_read_input_token_cost_above_272k_tokens_flex": 2e-07, "cache_read_input_token_cost_above_272k_tokens_priority": 8e-07, + "cache_read_input_token_cost_batches": 1e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 2e-07, "cache_read_input_token_cost_flex": 1e-07, "cache_read_input_token_cost_priority": 4e-07, "input_cost_per_token": 2e-06, @@ -30431,6 +30439,8 @@ "cache_read_input_token_cost_above_272k_tokens": 4e-08, "cache_read_input_token_cost_above_272k_tokens_flex": 2e-08, "cache_read_input_token_cost_above_272k_tokens_priority": 8e-08, + "cache_read_input_token_cost_batches": 1e-08, + "cache_read_input_token_cost_above_272k_tokens_batches": 2e-08, "cache_read_input_token_cost_flex": 1e-08, "cache_read_input_token_cost_priority": 4e-08, "input_cost_per_token": 2e-07, @@ -30717,6 +30727,8 @@ "gpt-5.5": { "cache_read_input_token_cost": 5e-07, "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_batches": 2.5e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 5e-07, "cache_read_input_token_cost_flex": 2.5e-07, "cache_read_input_token_cost_priority": 1.25e-06, "input_cost_per_token": 5e-06, @@ -30776,6 +30788,8 @@ "gpt-5.5-2026-04-23": { "cache_read_input_token_cost": 5e-07, "cache_read_input_token_cost_above_272k_tokens": 1e-06, + "cache_read_input_token_cost_batches": 2.5e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 5e-07, "cache_read_input_token_cost_flex": 2.5e-07, "cache_read_input_token_cost_priority": 1.25e-06, "input_cost_per_token": 5e-06, @@ -30939,6 +30953,8 @@ "gpt-5.4": { "cache_read_input_token_cost": 2.5e-07, "cache_read_input_token_cost_above_272k_tokens": 5e-07, + "cache_read_input_token_cost_batches": 1.25e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07, "cache_read_input_token_cost_flex": 1.3e-07, "cache_read_input_token_cost_priority": 5e-07, "input_cost_per_token": 2.5e-06, @@ -30993,6 +31009,8 @@ "gpt-5.4-2026-03-05": { "cache_read_input_token_cost": 2.5e-07, "cache_read_input_token_cost_above_272k_tokens": 5e-07, + "cache_read_input_token_cost_batches": 1.25e-07, + "cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07, "cache_read_input_token_cost_flex": 1.3e-07, "cache_read_input_token_cost_priority": 5e-07, "input_cost_per_token": 2.5e-06, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index 1970ddef6e0..615836095c2 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -163,6 +163,11 @@ "minimum": 0, "description": "Rate applied once the prompt exceeds the token threshold in the field name." }, + "cache_read_input_token_cost_above_272k_tokens_batches": { + "type": "number", + "minimum": 0, + "description": "Batch API rate applied once the prompt exceeds the token threshold in the field name." + }, "cache_read_input_token_cost_above_272k_tokens_flex": { "type": "number", "minimum": 0, @@ -178,6 +183,11 @@ "minimum": 0, "description": "Rate applied once the prompt exceeds the token threshold in the field name." }, + "cache_read_input_token_cost_batches": { + "type": "number", + "minimum": 0, + "description": "USD per cached prompt token via the provider's batch API." + }, "cache_read_input_token_cost_flex": { "type": "number", "minimum": 0, diff --git a/tests/test_litellm/batches/test_batch_utils.py b/tests/test_litellm/batches/test_batch_utils.py index 4be668fb485..b3420d275df 100644 --- a/tests/test_litellm/batches/test_batch_utils.py +++ b/tests/test_litellm/batches/test_batch_utils.py @@ -1745,3 +1745,42 @@ def test_unparsable_bedrock_batch_usage_warns(caplog): assert usage.total_tokens == 0 assert "does not understand" in caplog.text assert "inputTextTokenCount" in caplog.text + + +def test_total_cost_bills_cached_tokens_per_line_at_the_batch_cached_rate(): + responses_row = _success_row( + usage={ + "input_tokens": 300_000, + "output_tokens": 10, + "total_tokens": 300_010, + "input_tokens_details": {"cached_tokens": 299_000}, + } + ) + chat_row = _success_row(usage={**_usage(100, 10), "prompt_tokens_details": {"cached_tokens": 60}}) + + result = bu._aggregate_batch_cost_usage_models( + entries=[responses_row, chat_row], + custom_llm_provider="openai", + model_info=ModelInfo( + key="lit-batch-cached-tier", + max_tokens=None, + max_input_tokens=None, + max_output_tokens=None, + input_cost_per_token=2e-6, + output_cost_per_token=8e-6, + cache_read_input_token_cost=1e-6, + litellm_provider="openai", + mode="chat", + supported_openai_params=None, + input_cost_per_token_batches=1e-6, + output_cost_per_token_batches=4e-6, + cache_read_input_token_cost_batches=5e-7, + input_cost_per_token_above_272k_tokens_batches=2e-6, + output_cost_per_token_above_272k_tokens_batches=6e-6, + cache_read_input_token_cost_above_272k_tokens_batches=1e-6, + ), + ) + + long_line = 1_000 * 2e-6 + 299_000 * 1e-6 + 10 * 6e-6 + short_line = 40 * 1e-6 + 60 * 5e-7 + 10 * 4e-6 + assert result.cost == pytest.approx(long_line + short_line) diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index 6554aac0975..a3d8815fbd6 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -1,4 +1,5 @@ import json +from typing import cast import pytest from fastapi.testclient import TestClient @@ -1776,7 +1777,7 @@ def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map): sol = litellm.model_cost["gpt-5.6-sol"] cost_fields = sorted(field for field in sol if "cost" in field) - assert len(cost_fields) == 29 + assert len(cost_fields) == 31 for field in cost_fields: assert alias.get(field) == sol.get(field), field @@ -4766,3 +4767,62 @@ def test_route_image_generation_cost_falls_back_to_requested_size(monkeypatch, r ) assert cost == expected_cost + + +def _batch_rates_model_info(**rates: object) -> ModelInfo: + return cast(ModelInfo, dict(rates)) + + +def test_get_batch_cost_rates_parses_string_rates(): + from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates + + rates = get_batch_cost_rates( + _batch_rates_model_info( + input_cost_per_token_batches="1e-06", + output_cost_per_token_batches="4e-06", + cache_read_input_token_cost_batches="1e-07", + input_cost_per_token_above_272k_tokens_batches="2e-06", + output_cost_per_token_above_272k_tokens_batches="6e-06", + cache_read_input_token_cost_above_272k_tokens_batches="2e-07", + ), + Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001), + "openai", + ) + + assert (rates.input, rates.output, rates.cache_read) == (2e-6, 6e-6, 2e-7) + + +def test_get_batch_cost_rates_falls_back_to_the_flat_rates_when_tier_rates_are_unparsable(): + from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates + + rates = get_batch_cost_rates( + _batch_rates_model_info( + input_cost_per_token_batches=1e-6, + output_cost_per_token_batches=4e-6, + cache_read_input_token_cost_batches=1e-7, + input_cost_per_token_above_272k_tokens_batches="two", + output_cost_per_token_above_272k_tokens_batches="six", + cache_read_input_token_cost_above_272k_tokens_batches="none", + ), + Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001), + "openai", + ) + + assert (rates.input, rates.output, rates.cache_read) == (1e-6, 4e-6, 1e-7) + + +def test_get_batch_cost_rates_has_no_cached_rate_without_a_cached_batch_key(): + from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates + + rates = get_batch_cost_rates( + _batch_rates_model_info( + input_cost_per_token_batches=1e-6, + output_cost_per_token_batches=4e-6, + cache_read_input_token_cost=2e-7, + input_cost_per_token_above_272k_tokens_batches=2e-6, + ), + Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001), + "openai", + ) + + assert (rates.input, rates.output, rates.cache_read) == (2e-6, 4e-6, None) diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 9259d28fbac..4d7d6a13949 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -4523,6 +4523,8 @@ def test_get_model_info_exposes_the_long_context_batch_tier(_local_model_cost_ma assert info["input_cost_per_token_above_272k_tokens_batches"] == 2.5e-6 assert info["output_cost_per_token_above_272k_tokens_batches"] == 1.125e-5 + assert info["cache_read_input_token_cost_batches"] == 1.25e-7 + assert info["cache_read_input_token_cost_above_272k_tokens_batches"] == 2.5e-7 def test_regular_path_never_bills_the_batch_tier_keys(_local_model_cost_map, monkeypatch): @@ -4594,3 +4596,80 @@ def test_batch_cost_calculator_ignores_malformed_batch_tier_keys(): assert prompt_cost == pytest.approx(300_035 * 2e-6) assert completion_cost == pytest.approx(64 * 6e-6) + + +def _cached_usage(prompt_tokens: int, cached_tokens: int, completion_tokens: int) -> Usage: + return Usage( + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + total_tokens=prompt_tokens + completion_tokens, + prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens), + ) + + +def test_batch_cost_calculator_bills_cached_tokens_at_the_long_context_batch_cached_rate(_local_model_cost_map): + from litellm.cost_calculator import batch_cost_calculator + + prompt_cost, completion_cost = batch_cost_calculator( + usage=_cached_usage(300_048, 300_045, 11), model="gpt-5.6-luna", custom_llm_provider="openai" + ) + + assert prompt_cost == pytest.approx(3 * 2e-7 + 300_045 * 2e-8) + assert completion_cost == pytest.approx(11 * 9e-7) + + +def test_batch_cost_calculator_bills_cached_tokens_at_the_flat_batch_cached_rate_at_or_below_272k( + _local_model_cost_map, +): + from litellm.cost_calculator import batch_cost_calculator + + prompt_cost, _ = batch_cost_calculator( + usage=_cached_usage(1_000, 900, 4), model="gpt-5.6-luna", custom_llm_provider="openai" + ) + + assert prompt_cost == pytest.approx(100 * 1e-7 + 900 * 1e-8) + + +def test_batch_cost_calculator_bills_cached_tokens_at_the_batch_input_rate_without_a_cached_batch_rate( + _local_model_cost_map, +): + from litellm.cost_calculator import batch_cost_calculator + + prompt_cost, _ = batch_cost_calculator( + usage=_cached_usage(300_048, 300_045, 11), model="gpt-5.5-pro", custom_llm_provider="openai" + ) + + assert prompt_cost == pytest.approx(300_048 * 3e-5) + + +_OPENAI_ENTRIES_WITH_CACHED_BATCH_RATES = frozenset( + { + "gpt-6-astra", + "gpt-5.6", + "gpt-5.6-sol", + "gpt-5.6-terra", + "gpt-5.6-luna", + "gpt-5.5", + "gpt-5.5-2026-04-23", + "gpt-5.4", + "gpt-5.4-2026-03-05", + } +) + + +def test_openai_cached_batch_rates_are_half_the_standard_cached_rates(_local_model_cost_map): + carriers = { + name: entry + for name, entry in litellm.model_cost.items() + if isinstance(entry, dict) + and entry.get("litellm_provider") == "openai" + and entry.get("cache_read_input_token_cost_batches") is not None + } + + assert _OPENAI_ENTRIES_WITH_CACHED_BATCH_RATES <= set(carriers) + for entry in carriers.values(): + assert entry["cache_read_input_token_cost_batches"] == entry["cache_read_input_token_cost"] / 2 + assert ( + entry["cache_read_input_token_cost_above_272k_tokens_batches"] + == entry["cache_read_input_token_cost_above_272k_tokens"] / 2 + ) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 0f2d91899b5..4c833b34a0c 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -927,6 +927,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "type": "number" }, "cache_read_input_token_cost_above_512k_tokens": {"type": "number"}, + "cache_read_input_token_cost_batches": {"type": "number"}, + "cache_read_input_token_cost_above_272k_tokens_batches": {"type": "number"}, "cache_creation_input_token_cost_above_1hr_above_200k_tokens": { "type": "number" }, diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 1adaef29f37..9bc513e38c6 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -29336,12 +29336,16 @@ export interface components { cache_read_input_token_cost_above_200k_tokens_priority?: number | null; /** Cache Read Input Token Cost Above 272K Tokens */ cache_read_input_token_cost_above_272k_tokens?: number | null; + /** Cache Read Input Token Cost Above 272K Tokens Batches */ + cache_read_input_token_cost_above_272k_tokens_batches?: number | null; /** Cache Read Input Token Cost Above 272K Tokens Flex */ cache_read_input_token_cost_above_272k_tokens_flex?: number | null; /** Cache Read Input Token Cost Above 272K Tokens Priority */ cache_read_input_token_cost_above_272k_tokens_priority?: number | null; /** Cache Read Input Token Cost Above 512K Tokens */ cache_read_input_token_cost_above_512k_tokens?: number | null; + /** Cache Read Input Token Cost Batches */ + cache_read_input_token_cost_batches?: number | null; /** Cache Read Input Token Cost Flex */ cache_read_input_token_cost_flex?: number | null; /** Cache Read Input Token Cost Priority */ @@ -39470,12 +39474,16 @@ export interface components { cache_read_input_token_cost_above_200k_tokens_priority?: number | null; /** Cache Read Input Token Cost Above 272K Tokens */ cache_read_input_token_cost_above_272k_tokens?: number | null; + /** Cache Read Input Token Cost Above 272K Tokens Batches */ + cache_read_input_token_cost_above_272k_tokens_batches?: number | null; /** Cache Read Input Token Cost Above 272K Tokens Flex */ cache_read_input_token_cost_above_272k_tokens_flex?: number | null; /** Cache Read Input Token Cost Above 272K Tokens Priority */ cache_read_input_token_cost_above_272k_tokens_priority?: number | null; /** Cache Read Input Token Cost Above 512K Tokens */ cache_read_input_token_cost_above_512k_tokens?: number | null; + /** Cache Read Input Token Cost Batches */ + cache_read_input_token_cost_batches?: number | null; /** Cache Read Input Token Cost Flex */ cache_read_input_token_cost_flex?: number | null; /** Cache Read Input Token Cost Priority */