fix(cost): bill cached batch tokens at OpenAI's cached batch rate

Adds cache_read_input_token_cost_batches and
cache_read_input_token_cost_above_272k_tokens_batches for the tiered
OpenAI entries at half the standard cached rate, bills cached batch
tokens at that rate per output line, and parses string-valued batch
rates in deployment-level model_info.
This commit is contained in:
mateo-berri 2026-09-05 00:25:45 -07:00
parent 9819e4e0f8
commit ab3c924174
13 changed files with 293 additions and 17 deletions

View file

@ -169,6 +169,7 @@ COST_DESCRIPTIONS: dict[str, str] = {
"cache_read_input_token_cost": "USD per prompt token served from the provider's prompt cache.",
"input_cost_per_token_batches": "USD per prompt token via the provider's batch API.",
"output_cost_per_token_batches": "USD per generated token via the provider's batch API.",
"cache_read_input_token_cost_batches": "USD per cached prompt token via the provider's batch API.",
}

View file

@ -2242,15 +2242,13 @@ def batch_cost_calculator(
if not model_info:
return 0.0, 0.0
input_cost_per_token_batches, output_cost_per_token_batches = get_batch_cost_rates(
model_info, usage, custom_llm_provider
)
batch_rates: Final = get_batch_cost_rates(model_info, usage, custom_llm_provider)
input_cost_per_token: Final = model_info.get("input_cost_per_token")
output_cost_per_token: Final = model_info.get("output_cost_per_token")
total_prompt_cost = 0.0
total_completion_cost = 0.0
if input_cost_per_token_batches is not None:
total_prompt_cost = usage.prompt_tokens * input_cost_per_token_batches
if batch_rates.input is not None:
total_prompt_cost = _batch_prompt_cost(usage, batch_rates.input, batch_rates.cache_read)
elif input_cost_per_token:
details: Final = parse_prompt_tokens_details(usage)
cache_read_tokens: Final = details["cache_hit_tokens"]
@ -2269,8 +2267,8 @@ def batch_cost_calculator(
cache_creation_cost: Final = model_info.get("cache_creation_input_token_cost") or input_cost_per_token
total_prompt_cost += cache_creation_tokens * cache_creation_cost / 2
if output_cost_per_token_batches is not None:
total_completion_cost = usage.completion_tokens * output_cost_per_token_batches
if batch_rates.output is not None:
total_completion_cost = usage.completion_tokens * batch_rates.output
elif output_cost_per_token:
total_completion_cost = (
usage.completion_tokens * (output_cost_per_token) / 2
@ -2284,6 +2282,13 @@ def batch_cost_calculator(
return total_prompt_cost, total_completion_cost
def _batch_prompt_cost(usage: Usage, input_rate: float, cache_read_rate: float | None) -> float:
if cache_read_rate is None:
return usage.prompt_tokens * input_rate
cached_tokens: Final = parse_prompt_tokens_details(usage)["cache_hit_tokens"]
return get_billable_input_tokens(usage) * input_rate + cached_tokens * cache_read_rate
def _attribute_value(obj: object, name: str) -> object:
return getattr(obj, name)

View file

@ -248,14 +248,31 @@ def _prompt_exceeds_threshold(prompt_tokens: int, threshold: float, inclusive: b
return prompt_tokens > threshold or (inclusive and prompt_tokens == threshold)
@dataclass(frozen=True, slots=True)
class BatchCostRates:
input: float | None
output: float | None
cache_read: float | None
def _batch_rate(model_info: ModelInfo, key: str) -> float | None:
value: Final = model_info.get(key)
return value if isinstance(value, (int, float)) else None
if isinstance(value, (int, float)):
return float(value)
if not isinstance(value, str):
return None
try:
return float(value)
except ValueError:
return None
def get_batch_cost_rates(
model_info: ModelInfo, usage: Usage, custom_llm_provider: str | None
) -> tuple[float | None, float | None]:
def _batch_tier_rate(model_info: ModelInfo, tier_key: str, flat_key: str) -> float | None:
tier_rate: Final = _batch_rate(model_info, tier_key)
return _batch_rate(model_info, flat_key) if tier_rate is None else tier_rate
def get_batch_cost_rates(model_info: ModelInfo, usage: Usage, custom_llm_provider: str | None) -> BatchCostRates:
inclusive: Final = _uses_inclusive_token_thresholds(custom_llm_provider)
tier_input_keys: Final = tuple(
key for key, value in model_info.items() if _BATCH_TIER_INPUT_KEY.match(key) and value is not None
@ -268,12 +285,23 @@ def get_batch_cost_rates(
),
None,
)
flat_output_rate: Final = _batch_rate(model_info, "output_cost_per_token_batches")
if crossed_input_key is None:
return _batch_rate(model_info, "input_cost_per_token_batches"), flat_output_rate
tier_input_rate: Final = _batch_rate(model_info, crossed_input_key)
tier_output_rate: Final = _batch_rate(model_info, crossed_input_key.replace("input_", "output_", 1))
return tier_input_rate, flat_output_rate if tier_output_rate is None else tier_output_rate
return BatchCostRates(
input=_batch_rate(model_info, "input_cost_per_token_batches"),
output=_batch_rate(model_info, "output_cost_per_token_batches"),
cache_read=_batch_rate(model_info, "cache_read_input_token_cost_batches"),
)
return BatchCostRates(
input=_batch_tier_rate(model_info, crossed_input_key, "input_cost_per_token_batches"),
output=_batch_tier_rate(
model_info, crossed_input_key.replace("input_", "output_", 1), "output_cost_per_token_batches"
),
cache_read=_batch_tier_rate(
model_info,
crossed_input_key.replace("input_cost_per_token", "cache_read_input_token_cost", 1),
"cache_read_input_token_cost_batches",
),
)
def _select_priced_tier(model_info: ModelInfo, usage: Usage) -> dict | None:

View file

@ -30152,6 +30152,8 @@
"cache_read_input_token_cost_above_272k_tokens": 2e-06,
"cache_read_input_token_cost_above_272k_tokens_flex": 1e-06,
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
"cache_read_input_token_cost_batches": 5e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 1e-06,
"cache_read_input_token_cost_flex": 5e-07,
"cache_read_input_token_cost_priority": 2e-06,
"input_cost_per_token": 1e-05,
@ -30223,6 +30225,8 @@
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
"cache_read_input_token_cost_batches": 2e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
"cache_read_input_token_cost_flex": 2e-07,
"cache_read_input_token_cost_priority": 8e-07,
"input_cost_per_token": 4e-06,
@ -30292,6 +30296,8 @@
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
"cache_read_input_token_cost_batches": 2e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
"cache_read_input_token_cost_flex": 2e-07,
"cache_read_input_token_cost_priority": 8e-07,
"input_cost_per_token": 4e-06,
@ -30362,6 +30368,8 @@
"cache_read_input_token_cost_above_272k_tokens": 4e-07,
"cache_read_input_token_cost_above_272k_tokens_flex": 2e-07,
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-07,
"cache_read_input_token_cost_flex": 1e-07,
"cache_read_input_token_cost_priority": 4e-07,
"input_cost_per_token": 2e-06,
@ -30431,6 +30439,8 @@
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
"cache_read_input_token_cost_above_272k_tokens_flex": 2e-08,
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
"cache_read_input_token_cost_batches": 1e-08,
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-08,
"cache_read_input_token_cost_flex": 1e-08,
"cache_read_input_token_cost_priority": 4e-08,
"input_cost_per_token": 2e-07,
@ -30717,6 +30727,8 @@
"gpt-5.5": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
"cache_read_input_token_cost_batches": 2.5e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 5e-07,
"cache_read_input_token_cost_flex": 2.5e-07,
"cache_read_input_token_cost_priority": 1.25e-06,
"input_cost_per_token": 5e-06,
@ -30776,6 +30788,8 @@
"gpt-5.5-2026-04-23": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
"cache_read_input_token_cost_batches": 2.5e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 5e-07,
"cache_read_input_token_cost_flex": 2.5e-07,
"cache_read_input_token_cost_priority": 1.25e-06,
"input_cost_per_token": 5e-06,
@ -30939,6 +30953,8 @@
"gpt-5.4": {
"cache_read_input_token_cost": 2.5e-07,
"cache_read_input_token_cost_above_272k_tokens": 5e-07,
"cache_read_input_token_cost_batches": 1.25e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07,
"cache_read_input_token_cost_flex": 1.3e-07,
"cache_read_input_token_cost_priority": 5e-07,
"input_cost_per_token": 2.5e-06,
@ -30993,6 +31009,8 @@
"gpt-5.4-2026-03-05": {
"cache_read_input_token_cost": 2.5e-07,
"cache_read_input_token_cost_above_272k_tokens": 5e-07,
"cache_read_input_token_cost_batches": 1.25e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07,
"cache_read_input_token_cost_flex": 1.3e-07,
"cache_read_input_token_cost_priority": 5e-07,
"input_cost_per_token": 2.5e-06,

View file

@ -256,6 +256,8 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
cache_read_input_token_cost_above_272k_tokens_priority: float | None
cache_read_input_token_cost_above_272k_tokens_flex: float | None
cache_read_input_token_cost_above_512k_tokens: float | None
cache_read_input_token_cost_batches: ReadOnly[float | None]
cache_read_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None]
# Smallest prefix this model will actually cache, whatever caching mechanism its provider uses.
# Absent means the provider-agnostic default applies; see MINIMUM_PROMPT_CACHE_TOKEN_COUNT.
prompt_cache_min_tokens: int | None
@ -3517,6 +3519,8 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
cache_read_input_token_cost_above_200k_tokens_priority: float | None = None
cache_read_input_token_cost_above_272k_tokens_priority: float | None = None
cache_read_input_token_cost_above_272k_tokens_flex: float | None = None
cache_read_input_token_cost_batches: float | None = None
cache_read_input_token_cost_above_272k_tokens_batches: float | None = None
cache_read_input_audio_token_cost: float | None = None
input_cost_per_character_above_128k_tokens: float | None = None
input_cost_per_audio_token: float | None = None

View file

@ -5818,6 +5818,10 @@ def _get_model_info_helper(
cache_read_input_token_cost_flex=_model_info.get("cache_read_input_token_cost_flex", None),
cache_read_input_token_cost_priority=_model_info.get("cache_read_input_token_cost_priority", None),
cache_read_input_token_cost_ultrafast=_model_info.get("cache_read_input_token_cost_ultrafast", None),
cache_read_input_token_cost_batches=_model_info.get("cache_read_input_token_cost_batches"),
cache_read_input_token_cost_above_272k_tokens_batches=_model_info.get(
"cache_read_input_token_cost_above_272k_tokens_batches"
),
cache_creation_input_token_cost_above_1hr=_model_info.get(
"cache_creation_input_token_cost_above_1hr", None
),

View file

@ -30152,6 +30152,8 @@
"cache_read_input_token_cost_above_272k_tokens": 2e-06,
"cache_read_input_token_cost_above_272k_tokens_flex": 1e-06,
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
"cache_read_input_token_cost_batches": 5e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 1e-06,
"cache_read_input_token_cost_flex": 5e-07,
"cache_read_input_token_cost_priority": 2e-06,
"input_cost_per_token": 1e-05,
@ -30223,6 +30225,8 @@
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
"cache_read_input_token_cost_batches": 2e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
"cache_read_input_token_cost_flex": 2e-07,
"cache_read_input_token_cost_priority": 8e-07,
"input_cost_per_token": 4e-06,
@ -30292,6 +30296,8 @@
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
"cache_read_input_token_cost_batches": 2e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
"cache_read_input_token_cost_flex": 2e-07,
"cache_read_input_token_cost_priority": 8e-07,
"input_cost_per_token": 4e-06,
@ -30362,6 +30368,8 @@
"cache_read_input_token_cost_above_272k_tokens": 4e-07,
"cache_read_input_token_cost_above_272k_tokens_flex": 2e-07,
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-07,
"cache_read_input_token_cost_flex": 1e-07,
"cache_read_input_token_cost_priority": 4e-07,
"input_cost_per_token": 2e-06,
@ -30431,6 +30439,8 @@
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
"cache_read_input_token_cost_above_272k_tokens_flex": 2e-08,
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
"cache_read_input_token_cost_batches": 1e-08,
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-08,
"cache_read_input_token_cost_flex": 1e-08,
"cache_read_input_token_cost_priority": 4e-08,
"input_cost_per_token": 2e-07,
@ -30717,6 +30727,8 @@
"gpt-5.5": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
"cache_read_input_token_cost_batches": 2.5e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 5e-07,
"cache_read_input_token_cost_flex": 2.5e-07,
"cache_read_input_token_cost_priority": 1.25e-06,
"input_cost_per_token": 5e-06,
@ -30776,6 +30788,8 @@
"gpt-5.5-2026-04-23": {
"cache_read_input_token_cost": 5e-07,
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
"cache_read_input_token_cost_batches": 2.5e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 5e-07,
"cache_read_input_token_cost_flex": 2.5e-07,
"cache_read_input_token_cost_priority": 1.25e-06,
"input_cost_per_token": 5e-06,
@ -30939,6 +30953,8 @@
"gpt-5.4": {
"cache_read_input_token_cost": 2.5e-07,
"cache_read_input_token_cost_above_272k_tokens": 5e-07,
"cache_read_input_token_cost_batches": 1.25e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07,
"cache_read_input_token_cost_flex": 1.3e-07,
"cache_read_input_token_cost_priority": 5e-07,
"input_cost_per_token": 2.5e-06,
@ -30993,6 +31009,8 @@
"gpt-5.4-2026-03-05": {
"cache_read_input_token_cost": 2.5e-07,
"cache_read_input_token_cost_above_272k_tokens": 5e-07,
"cache_read_input_token_cost_batches": 1.25e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07,
"cache_read_input_token_cost_flex": 1.3e-07,
"cache_read_input_token_cost_priority": 5e-07,
"input_cost_per_token": 2.5e-06,

View file

@ -163,6 +163,11 @@
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_read_input_token_cost_above_272k_tokens_batches": {
"type": "number",
"minimum": 0,
"description": "Batch API rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_read_input_token_cost_above_272k_tokens_flex": {
"type": "number",
"minimum": 0,
@ -178,6 +183,11 @@
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_read_input_token_cost_batches": {
"type": "number",
"minimum": 0,
"description": "USD per cached prompt token via the provider's batch API."
},
"cache_read_input_token_cost_flex": {
"type": "number",
"minimum": 0,

View file

@ -1745,3 +1745,42 @@ def test_unparsable_bedrock_batch_usage_warns(caplog):
assert usage.total_tokens == 0
assert "does not understand" in caplog.text
assert "inputTextTokenCount" in caplog.text
def test_total_cost_bills_cached_tokens_per_line_at_the_batch_cached_rate():
responses_row = _success_row(
usage={
"input_tokens": 300_000,
"output_tokens": 10,
"total_tokens": 300_010,
"input_tokens_details": {"cached_tokens": 299_000},
}
)
chat_row = _success_row(usage={**_usage(100, 10), "prompt_tokens_details": {"cached_tokens": 60}})
result = bu._aggregate_batch_cost_usage_models(
entries=[responses_row, chat_row],
custom_llm_provider="openai",
model_info=ModelInfo(
key="lit-batch-cached-tier",
max_tokens=None,
max_input_tokens=None,
max_output_tokens=None,
input_cost_per_token=2e-6,
output_cost_per_token=8e-6,
cache_read_input_token_cost=1e-6,
litellm_provider="openai",
mode="chat",
supported_openai_params=None,
input_cost_per_token_batches=1e-6,
output_cost_per_token_batches=4e-6,
cache_read_input_token_cost_batches=5e-7,
input_cost_per_token_above_272k_tokens_batches=2e-6,
output_cost_per_token_above_272k_tokens_batches=6e-6,
cache_read_input_token_cost_above_272k_tokens_batches=1e-6,
),
)
long_line = 1_000 * 2e-6 + 299_000 * 1e-6 + 10 * 6e-6
short_line = 40 * 1e-6 + 60 * 5e-7 + 10 * 4e-6
assert result.cost == pytest.approx(long_line + short_line)

View file

@ -1,4 +1,5 @@
import json
from typing import cast
import pytest
from fastapi.testclient import TestClient
@ -1776,7 +1777,7 @@ def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map):
sol = litellm.model_cost["gpt-5.6-sol"]
cost_fields = sorted(field for field in sol if "cost" in field)
assert len(cost_fields) == 29
assert len(cost_fields) == 31
for field in cost_fields:
assert alias.get(field) == sol.get(field), field
@ -4766,3 +4767,62 @@ def test_route_image_generation_cost_falls_back_to_requested_size(monkeypatch, r
)
assert cost == expected_cost
def _batch_rates_model_info(**rates: object) -> ModelInfo:
return cast(ModelInfo, dict(rates))
def test_get_batch_cost_rates_parses_string_rates():
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
rates = get_batch_cost_rates(
_batch_rates_model_info(
input_cost_per_token_batches="1e-06",
output_cost_per_token_batches="4e-06",
cache_read_input_token_cost_batches="1e-07",
input_cost_per_token_above_272k_tokens_batches="2e-06",
output_cost_per_token_above_272k_tokens_batches="6e-06",
cache_read_input_token_cost_above_272k_tokens_batches="2e-07",
),
Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001),
"openai",
)
assert (rates.input, rates.output, rates.cache_read) == (2e-6, 6e-6, 2e-7)
def test_get_batch_cost_rates_falls_back_to_the_flat_rates_when_tier_rates_are_unparsable():
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
rates = get_batch_cost_rates(
_batch_rates_model_info(
input_cost_per_token_batches=1e-6,
output_cost_per_token_batches=4e-6,
cache_read_input_token_cost_batches=1e-7,
input_cost_per_token_above_272k_tokens_batches="two",
output_cost_per_token_above_272k_tokens_batches="six",
cache_read_input_token_cost_above_272k_tokens_batches="none",
),
Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001),
"openai",
)
assert (rates.input, rates.output, rates.cache_read) == (1e-6, 4e-6, 1e-7)
def test_get_batch_cost_rates_has_no_cached_rate_without_a_cached_batch_key():
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
rates = get_batch_cost_rates(
_batch_rates_model_info(
input_cost_per_token_batches=1e-6,
output_cost_per_token_batches=4e-6,
cache_read_input_token_cost=2e-7,
input_cost_per_token_above_272k_tokens_batches=2e-6,
),
Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001),
"openai",
)
assert (rates.input, rates.output, rates.cache_read) == (2e-6, 4e-6, None)

View file

@ -4523,6 +4523,8 @@ def test_get_model_info_exposes_the_long_context_batch_tier(_local_model_cost_ma
assert info["input_cost_per_token_above_272k_tokens_batches"] == 2.5e-6
assert info["output_cost_per_token_above_272k_tokens_batches"] == 1.125e-5
assert info["cache_read_input_token_cost_batches"] == 1.25e-7
assert info["cache_read_input_token_cost_above_272k_tokens_batches"] == 2.5e-7
def test_regular_path_never_bills_the_batch_tier_keys(_local_model_cost_map, monkeypatch):
@ -4594,3 +4596,80 @@ def test_batch_cost_calculator_ignores_malformed_batch_tier_keys():
assert prompt_cost == pytest.approx(300_035 * 2e-6)
assert completion_cost == pytest.approx(64 * 6e-6)
def _cached_usage(prompt_tokens: int, cached_tokens: int, completion_tokens: int) -> Usage:
return Usage(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
total_tokens=prompt_tokens + completion_tokens,
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens),
)
def test_batch_cost_calculator_bills_cached_tokens_at_the_long_context_batch_cached_rate(_local_model_cost_map):
from litellm.cost_calculator import batch_cost_calculator
prompt_cost, completion_cost = batch_cost_calculator(
usage=_cached_usage(300_048, 300_045, 11), model="gpt-5.6-luna", custom_llm_provider="openai"
)
assert prompt_cost == pytest.approx(3 * 2e-7 + 300_045 * 2e-8)
assert completion_cost == pytest.approx(11 * 9e-7)
def test_batch_cost_calculator_bills_cached_tokens_at_the_flat_batch_cached_rate_at_or_below_272k(
_local_model_cost_map,
):
from litellm.cost_calculator import batch_cost_calculator
prompt_cost, _ = batch_cost_calculator(
usage=_cached_usage(1_000, 900, 4), model="gpt-5.6-luna", custom_llm_provider="openai"
)
assert prompt_cost == pytest.approx(100 * 1e-7 + 900 * 1e-8)
def test_batch_cost_calculator_bills_cached_tokens_at_the_batch_input_rate_without_a_cached_batch_rate(
_local_model_cost_map,
):
from litellm.cost_calculator import batch_cost_calculator
prompt_cost, _ = batch_cost_calculator(
usage=_cached_usage(300_048, 300_045, 11), model="gpt-5.5-pro", custom_llm_provider="openai"
)
assert prompt_cost == pytest.approx(300_048 * 3e-5)
_OPENAI_ENTRIES_WITH_CACHED_BATCH_RATES = frozenset(
{
"gpt-6-astra",
"gpt-5.6",
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-5.6-luna",
"gpt-5.5",
"gpt-5.5-2026-04-23",
"gpt-5.4",
"gpt-5.4-2026-03-05",
}
)
def test_openai_cached_batch_rates_are_half_the_standard_cached_rates(_local_model_cost_map):
carriers = {
name: entry
for name, entry in litellm.model_cost.items()
if isinstance(entry, dict)
and entry.get("litellm_provider") == "openai"
and entry.get("cache_read_input_token_cost_batches") is not None
}
assert _OPENAI_ENTRIES_WITH_CACHED_BATCH_RATES <= set(carriers)
for entry in carriers.values():
assert entry["cache_read_input_token_cost_batches"] == entry["cache_read_input_token_cost"] / 2
assert (
entry["cache_read_input_token_cost_above_272k_tokens_batches"]
== entry["cache_read_input_token_cost_above_272k_tokens"] / 2
)

View file

@ -927,6 +927,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"type": "number"
},
"cache_read_input_token_cost_above_512k_tokens": {"type": "number"},
"cache_read_input_token_cost_batches": {"type": "number"},
"cache_read_input_token_cost_above_272k_tokens_batches": {"type": "number"},
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": {
"type": "number"
},

View file

@ -29336,12 +29336,16 @@ export interface components {
cache_read_input_token_cost_above_200k_tokens_priority?: number | null;
/** Cache Read Input Token Cost Above 272K Tokens */
cache_read_input_token_cost_above_272k_tokens?: number | null;
/** Cache Read Input Token Cost Above 272K Tokens Batches */
cache_read_input_token_cost_above_272k_tokens_batches?: number | null;
/** Cache Read Input Token Cost Above 272K Tokens Flex */
cache_read_input_token_cost_above_272k_tokens_flex?: number | null;
/** Cache Read Input Token Cost Above 272K Tokens Priority */
cache_read_input_token_cost_above_272k_tokens_priority?: number | null;
/** Cache Read Input Token Cost Above 512K Tokens */
cache_read_input_token_cost_above_512k_tokens?: number | null;
/** Cache Read Input Token Cost Batches */
cache_read_input_token_cost_batches?: number | null;
/** Cache Read Input Token Cost Flex */
cache_read_input_token_cost_flex?: number | null;
/** Cache Read Input Token Cost Priority */
@ -39470,12 +39474,16 @@ export interface components {
cache_read_input_token_cost_above_200k_tokens_priority?: number | null;
/** Cache Read Input Token Cost Above 272K Tokens */
cache_read_input_token_cost_above_272k_tokens?: number | null;
/** Cache Read Input Token Cost Above 272K Tokens Batches */
cache_read_input_token_cost_above_272k_tokens_batches?: number | null;
/** Cache Read Input Token Cost Above 272K Tokens Flex */
cache_read_input_token_cost_above_272k_tokens_flex?: number | null;
/** Cache Read Input Token Cost Above 272K Tokens Priority */
cache_read_input_token_cost_above_272k_tokens_priority?: number | null;
/** Cache Read Input Token Cost Above 512K Tokens */
cache_read_input_token_cost_above_512k_tokens?: number | null;
/** Cache Read Input Token Cost Batches */
cache_read_input_token_cost_batches?: number | null;
/** Cache Read Input Token Cost Flex */
cache_read_input_token_cost_flex?: number | null;
/** Cache Read Input Token Cost Priority */