fix(cost): bill batch cache writes at the batch cache-write rate and carry published batch rates for one-sided deployments

OpenAI's Batch table prices cache writes for gpt-6-astra, gpt-5.6, gpt-5.6-sol, gpt-5.6-terra and gpt-5.6-luna at half the standard cache-write rate, so the cost map gains cache_creation_input_token_cost_batches and its above_272k tier for those entries and batch cost pulls written tokens out of the input bucket at that rate; models without the key keep billing writes at the batch input rate.

A deployment declaring only one side of its batch pricing now carries every published batch rate of the other side (tier, cached, cache write), its own keys win, and a lone tier, cached or cache-write batch key counts as declared pricing instead of being ignored.
This commit is contained in:
mateo-berri 2026-09-05 02:39:06 -07:00
parent ab3c924174
commit 3f723bb918
14 changed files with 313 additions and 19 deletions

View file

@ -170,6 +170,9 @@ COST_DESCRIPTIONS: dict[str, str] = {
"input_cost_per_token_batches": "USD per prompt token via the provider's batch API.",
"output_cost_per_token_batches": "USD per generated token via the provider's batch API.",
"cache_read_input_token_cost_batches": "USD per cached prompt token via the provider's batch API.",
"cache_creation_input_token_cost_batches": (
"USD per token written to the provider's prompt cache via its batch API."
),
}

View file

@ -2248,7 +2248,9 @@ def batch_cost_calculator(
total_prompt_cost = 0.0
total_completion_cost = 0.0
if batch_rates.input is not None:
total_prompt_cost = _batch_prompt_cost(usage, batch_rates.input, batch_rates.cache_read)
total_prompt_cost = _batch_prompt_cost(
usage, batch_rates.input, batch_rates.cache_read, batch_rates.cache_creation
)
elif input_cost_per_token:
details: Final = parse_prompt_tokens_details(usage)
cache_read_tokens: Final = details["cache_hit_tokens"]
@ -2282,11 +2284,17 @@ def batch_cost_calculator(
return total_prompt_cost, total_completion_cost
def _batch_prompt_cost(usage: Usage, input_rate: float, cache_read_rate: float | None) -> float:
if cache_read_rate is None:
return usage.prompt_tokens * input_rate
cached_tokens: Final = parse_prompt_tokens_details(usage)["cache_hit_tokens"]
return get_billable_input_tokens(usage) * input_rate + cached_tokens * cache_read_rate
def _batch_prompt_cost(
usage: Usage, input_rate: float, cache_read_rate: float | None, cache_creation_rate: float | None
) -> float:
details: Final = parse_prompt_tokens_details(usage)
cached_tokens: Final = details["cache_hit_tokens"] if cache_read_rate is not None else 0
written_tokens: Final = details["cache_creation_tokens"] if cache_creation_rate is not None else 0
return (
(usage.prompt_tokens - cached_tokens - written_tokens) * input_rate
+ cached_tokens * (cache_read_rate or 0.0)
+ written_tokens * (cache_creation_rate or 0.0)
)
def _attribute_value(obj: object, name: str) -> object:

View file

@ -330,7 +330,36 @@ _DEPLOYMENT_PRICING_KEYS: Final = (
"output_cost_per_token",
"input_cost_per_token_batches",
"output_cost_per_token_batches",
"input_cost_per_token_above_272k_tokens_batches",
"output_cost_per_token_above_272k_tokens_batches",
"cache_read_input_token_cost_batches",
"cache_read_input_token_cost_above_272k_tokens_batches",
"cache_creation_input_token_cost_batches",
"cache_creation_input_token_cost_above_272k_tokens_batches",
)
_INPUT_PRICING_KEY_PREFIXES: Final = (
"input_cost_per_token",
"cache_read_input_token_cost",
"cache_creation_input_token_cost",
)
_OUTPUT_PRICING_KEY_PREFIXES: Final = ("output_cost_per_token",)
_BATCH_PRICING_KEY_SUFFIX: Final = "_batches"
_NO_CARRIED_RATES: Final[Mapping[str, object]] = MappingProxyType({})
def _published_direction(
published: ModelInfo, registered: Mapping[str, object], flat_key: str, prefixes: tuple[str, ...]
) -> Mapping[str, object]:
return MappingProxyType(
{
key: value
for key, value in published.items()
if registered.get(key) is None
and (key == flat_key or (key.startswith(prefixes) and key.endswith(_BATCH_PRICING_KEY_SUFFIX)))
}
)
def deployment_pricing_model_info(model_id: str | None, deployment_model: str | None) -> ModelInfo | None:
@ -342,10 +371,11 @@ def deployment_pricing_model_info(model_id: str | None, deployment_model: str |
get_model_info fills absent costs with 0, so asking it directly cannot
tell "configured as free" apart from "no pricing configured". A deployment
may declare only one side of its pricing, so the side it leaves out keeps
the model's published rates instead of billing as zero. Ownership is per
token direction: declaring either rate for a direction takes that whole
direction, so a published batch rate can never displace a standard rate
the deployment configured itself.
the model's published standard rate and every published batch rate for
that direction (flat, long-context tier, cached, cache write) instead of
billing as zero. Ownership is per token direction: declaring either rate
for a direction takes that whole direction, so a published batch rate can
never displace a standard rate the deployment configured itself.
"""
if model_id is None:
return None
@ -366,13 +396,22 @@ def deployment_pricing_model_info(model_id: str | None, deployment_model: str |
registered.get("output_cost_per_token") is not None
or registered.get("output_cost_per_token_batches") is not None
)
if not declares_input:
merged["input_cost_per_token"] = published.get("input_cost_per_token")
merged["input_cost_per_token_batches"] = published.get("input_cost_per_token_batches")
if not declares_output:
merged["output_cost_per_token"] = published.get("output_cost_per_token")
merged["output_cost_per_token_batches"] = published.get("output_cost_per_token_batches")
return merged
carried_input: Final = (
_NO_CARRIED_RATES
if declares_input
else _published_direction(published, registered, "input_cost_per_token", _INPUT_PRICING_KEY_PREFIXES)
)
carried_output: Final = (
_NO_CARRIED_RATES
if declares_output
else _published_direction(published, registered, "output_cost_per_token", _OUTPUT_PRICING_KEY_PREFIXES)
)
priced: Final[ModelInfo] = { # pyright: ignore[reportAssignmentType] # carried keys are ModelInfo rates
**merged,
**carried_input,
**carried_output,
}
return priced
def _published_pricing(deployment_model: str | None) -> ModelInfo | None:

View file

@ -253,6 +253,7 @@ class BatchCostRates:
input: float | None
output: float | None
cache_read: float | None
cache_creation: float | None
def _batch_rate(model_info: ModelInfo, key: str) -> float | None:
@ -290,6 +291,7 @@ def get_batch_cost_rates(model_info: ModelInfo, usage: Usage, custom_llm_provide
input=_batch_rate(model_info, "input_cost_per_token_batches"),
output=_batch_rate(model_info, "output_cost_per_token_batches"),
cache_read=_batch_rate(model_info, "cache_read_input_token_cost_batches"),
cache_creation=_batch_rate(model_info, "cache_creation_input_token_cost_batches"),
)
return BatchCostRates(
input=_batch_tier_rate(model_info, crossed_input_key, "input_cost_per_token_batches"),
@ -301,6 +303,11 @@ def get_batch_cost_rates(model_info: ModelInfo, usage: Usage, custom_llm_provide
crossed_input_key.replace("input_cost_per_token", "cache_read_input_token_cost", 1),
"cache_read_input_token_cost_batches",
),
cache_creation=_batch_tier_rate(
model_info,
crossed_input_key.replace("input_cost_per_token", "cache_creation_input_token_cost", 1),
"cache_creation_input_token_cost_batches",
),
)

View file

@ -30154,6 +30154,8 @@
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
"cache_read_input_token_cost_batches": 5e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 1e-06,
"cache_creation_input_token_cost_batches": 6.25e-06,
"cache_creation_input_token_cost_above_272k_tokens_batches": 1.25e-05,
"cache_read_input_token_cost_flex": 5e-07,
"cache_read_input_token_cost_priority": 2e-06,
"input_cost_per_token": 1e-05,
@ -30227,6 +30229,8 @@
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
"cache_read_input_token_cost_batches": 2e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
"cache_creation_input_token_cost_batches": 2.5e-06,
"cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06,
"cache_read_input_token_cost_flex": 2e-07,
"cache_read_input_token_cost_priority": 8e-07,
"input_cost_per_token": 4e-06,
@ -30298,6 +30302,8 @@
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
"cache_read_input_token_cost_batches": 2e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
"cache_creation_input_token_cost_batches": 2.5e-06,
"cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06,
"cache_read_input_token_cost_flex": 2e-07,
"cache_read_input_token_cost_priority": 8e-07,
"input_cost_per_token": 4e-06,
@ -30370,6 +30376,8 @@
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-07,
"cache_creation_input_token_cost_batches": 1.25e-06,
"cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-06,
"cache_read_input_token_cost_flex": 1e-07,
"cache_read_input_token_cost_priority": 4e-07,
"input_cost_per_token": 2e-06,
@ -30441,6 +30449,8 @@
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
"cache_read_input_token_cost_batches": 1e-08,
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-08,
"cache_creation_input_token_cost_batches": 1.25e-07,
"cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-07,
"cache_read_input_token_cost_flex": 1e-08,
"cache_read_input_token_cost_priority": 4e-08,
"input_cost_per_token": 2e-07,

View file

@ -258,6 +258,8 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
cache_read_input_token_cost_above_512k_tokens: float | None
cache_read_input_token_cost_batches: ReadOnly[float | None]
cache_read_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None]
cache_creation_input_token_cost_batches: ReadOnly[float | None]
cache_creation_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None]
# Smallest prefix this model will actually cache, whatever caching mechanism its provider uses.
# Absent means the provider-agnostic default applies; see MINIMUM_PROMPT_CACHE_TOKEN_COUNT.
prompt_cache_min_tokens: int | None
@ -3521,6 +3523,8 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
cache_read_input_token_cost_above_272k_tokens_flex: float | None = None
cache_read_input_token_cost_batches: float | None = None
cache_read_input_token_cost_above_272k_tokens_batches: float | None = None
cache_creation_input_token_cost_batches: float | None = None
cache_creation_input_token_cost_above_272k_tokens_batches: float | None = None
cache_read_input_audio_token_cost: float | None = None
input_cost_per_character_above_128k_tokens: float | None = None
input_cost_per_audio_token: float | None = None

View file

@ -5822,6 +5822,10 @@ def _get_model_info_helper(
cache_read_input_token_cost_above_272k_tokens_batches=_model_info.get(
"cache_read_input_token_cost_above_272k_tokens_batches"
),
cache_creation_input_token_cost_batches=_model_info.get("cache_creation_input_token_cost_batches"),
cache_creation_input_token_cost_above_272k_tokens_batches=_model_info.get(
"cache_creation_input_token_cost_above_272k_tokens_batches"
),
cache_creation_input_token_cost_above_1hr=_model_info.get(
"cache_creation_input_token_cost_above_1hr", None
),

View file

@ -30154,6 +30154,8 @@
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
"cache_read_input_token_cost_batches": 5e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 1e-06,
"cache_creation_input_token_cost_batches": 6.25e-06,
"cache_creation_input_token_cost_above_272k_tokens_batches": 1.25e-05,
"cache_read_input_token_cost_flex": 5e-07,
"cache_read_input_token_cost_priority": 2e-06,
"input_cost_per_token": 1e-05,
@ -30227,6 +30229,8 @@
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
"cache_read_input_token_cost_batches": 2e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
"cache_creation_input_token_cost_batches": 2.5e-06,
"cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06,
"cache_read_input_token_cost_flex": 2e-07,
"cache_read_input_token_cost_priority": 8e-07,
"input_cost_per_token": 4e-06,
@ -30298,6 +30302,8 @@
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
"cache_read_input_token_cost_batches": 2e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
"cache_creation_input_token_cost_batches": 2.5e-06,
"cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06,
"cache_read_input_token_cost_flex": 2e-07,
"cache_read_input_token_cost_priority": 8e-07,
"input_cost_per_token": 4e-06,
@ -30370,6 +30376,8 @@
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-07,
"cache_read_input_token_cost_batches": 1e-07,
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-07,
"cache_creation_input_token_cost_batches": 1.25e-06,
"cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-06,
"cache_read_input_token_cost_flex": 1e-07,
"cache_read_input_token_cost_priority": 4e-07,
"input_cost_per_token": 2e-06,
@ -30441,6 +30449,8 @@
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
"cache_read_input_token_cost_batches": 1e-08,
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-08,
"cache_creation_input_token_cost_batches": 1.25e-07,
"cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-07,
"cache_read_input_token_cost_flex": 1e-08,
"cache_read_input_token_cost_priority": 4e-08,
"input_cost_per_token": 2e-07,

View file

@ -109,6 +109,11 @@
"minimum": 0,
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_creation_input_token_cost_above_272k_tokens_batches": {
"type": "number",
"minimum": 0,
"description": "Batch API rate applied once the prompt exceeds the token threshold in the field name."
},
"cache_creation_input_token_cost_above_272k_tokens_flex": {
"type": "number",
"minimum": 0,
@ -119,6 +124,11 @@
"minimum": 0,
"description": "Priority service-tier rate for the same-named base field."
},
"cache_creation_input_token_cost_batches": {
"type": "number",
"minimum": 0,
"description": "USD per token written to the provider's prompt cache via its batch API."
},
"cache_creation_input_token_cost_flex": {
"type": "number",
"minimum": 0,

View file

@ -1777,7 +1777,7 @@ def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map):
sol = litellm.model_cost["gpt-5.6-sol"]
cost_fields = sorted(field for field in sol if "cost" in field)
assert len(cost_fields) == 31
assert len(cost_fields) == 33
for field in cost_fields:
assert alias.get(field) == sol.get(field), field
@ -4803,12 +4803,14 @@ def test_get_batch_cost_rates_falls_back_to_the_flat_rates_when_tier_rates_are_u
input_cost_per_token_above_272k_tokens_batches="two",
output_cost_per_token_above_272k_tokens_batches="six",
cache_read_input_token_cost_above_272k_tokens_batches="none",
cache_creation_input_token_cost_batches=1.25e-7,
cache_creation_input_token_cost_above_272k_tokens_batches="nope",
),
Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001),
"openai",
)
assert (rates.input, rates.output, rates.cache_read) == (1e-6, 4e-6, 1e-7)
assert (rates.input, rates.output, rates.cache_read, rates.cache_creation) == (1e-6, 4e-6, 1e-7, 1.25e-7)
def test_get_batch_cost_rates_has_no_cached_rate_without_a_cached_batch_key():
@ -4826,3 +4828,38 @@ def test_get_batch_cost_rates_has_no_cached_rate_without_a_cached_batch_key():
)
assert (rates.input, rates.output, rates.cache_read) == (2e-6, 4e-6, None)
@pytest.mark.parametrize(("prompt_tokens", "expected"), [(1_000, 1.25e-7), (272_000, 1.25e-7), (300_000, 2.5e-7)])
def test_get_batch_cost_rates_reads_the_batch_cache_write_rate_for_the_crossed_tier(prompt_tokens, expected):
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
rates = get_batch_cost_rates(
_batch_rates_model_info(
input_cost_per_token_batches=1e-7,
input_cost_per_token_above_272k_tokens_batches=2e-7,
cache_creation_input_token_cost_batches=1.25e-7,
cache_creation_input_token_cost_above_272k_tokens_batches=2.5e-7,
),
Usage(prompt_tokens=prompt_tokens, completion_tokens=1, total_tokens=prompt_tokens + 1),
"openai",
)
assert rates.cache_creation == expected
def test_get_batch_cost_rates_has_no_cache_write_rate_without_a_cache_write_batch_key():
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
rates = get_batch_cost_rates(
_batch_rates_model_info(
input_cost_per_token_batches=1e-7,
input_cost_per_token_above_272k_tokens_batches=2e-7,
cache_creation_input_token_cost=2.5e-7,
cache_creation_input_token_cost_above_272k_tokens=5e-7,
),
Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001),
"openai",
)
assert rates.cache_creation is None

View file

@ -6277,3 +6277,78 @@ def test_passthrough_embeddings_result_swapped_for_callbacks():
assert isinstance(swapped_result, EmbeddingResponse)
assert swapped_result.data[0]["embedding"] == [0.1, 0.2, 0.3]
@pytest.fixture
def _local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
def _luna_deployment_id(custom_pricing: dict[str, float]) -> str:
from litellm import Router
router = Router(
model_list=[
{
"model_name": "luna-batch",
"litellm_params": {"model": "openai/gpt-5.6-luna", "api_key": "sk-test", **custom_pricing},
}
]
)
return router.model_list[0]["model_info"]["id"]
def test_deployment_pricing_model_info_carries_every_published_input_batch_rate_when_only_output_is_declared(
_local_model_cost_map,
):
from litellm.litellm_core_utils.litellm_logging import deployment_pricing_model_info
info = deployment_pricing_model_info(
_luna_deployment_id({"output_cost_per_token_batches": 4e-6}), "openai/gpt-5.6-luna"
)
assert info is not None
assert info["input_cost_per_token_batches"] == 1e-7
assert info["input_cost_per_token_above_272k_tokens_batches"] == 2e-7
assert info["cache_read_input_token_cost_batches"] == 1e-8
assert info["cache_read_input_token_cost_above_272k_tokens_batches"] == 2e-8
assert info["cache_creation_input_token_cost_batches"] == 1.25e-7
assert info["cache_creation_input_token_cost_above_272k_tokens_batches"] == 2.5e-7
assert info["output_cost_per_token_batches"] == 4e-6
assert info["output_cost_per_token_above_272k_tokens_batches"] is None
def test_deployment_pricing_model_info_carries_the_published_output_batch_tier_when_only_input_is_declared(
_local_model_cost_map,
):
from litellm.litellm_core_utils.litellm_logging import deployment_pricing_model_info
info = deployment_pricing_model_info(
_luna_deployment_id({"input_cost_per_token_batches": 1e-6}), "openai/gpt-5.6-luna"
)
assert info is not None
assert info["output_cost_per_token_batches"] == 6e-7
assert info["output_cost_per_token_above_272k_tokens_batches"] == 9e-7
assert info["input_cost_per_token_batches"] == 1e-6
assert info["input_cost_per_token_above_272k_tokens_batches"] is None
assert info["cache_read_input_token_cost_batches"] is None
assert info["cache_creation_input_token_cost_batches"] is None
def test_deployment_pricing_model_info_honors_a_tier_only_batch_override_over_the_published_flat_rates(
_local_model_cost_map: None,
) -> None:
from litellm.litellm_core_utils.litellm_logging import deployment_pricing_model_info
info = deployment_pricing_model_info(
_luna_deployment_id({"input_cost_per_token_above_272k_tokens_batches": 1e-3}), "openai/gpt-5.6-luna"
)
assert info is not None
assert info["input_cost_per_token_above_272k_tokens_batches"] == 1e-3
assert info["input_cost_per_token_batches"] == 1e-7
assert info["cache_read_input_token_cost_batches"] == 1e-8
assert info["output_cost_per_token_batches"] == 6e-7
assert info["output_cost_per_token_above_272k_tokens_batches"] == 9e-7

View file

@ -4673,3 +4673,80 @@ def test_openai_cached_batch_rates_are_half_the_standard_cached_rates(_local_mod
entry["cache_read_input_token_cost_above_272k_tokens_batches"]
== entry["cache_read_input_token_cost_above_272k_tokens"] / 2
)
def _cache_write_usage(prompt_tokens: int, cache_write_tokens: int, completion_tokens: int) -> Usage:
return Usage(
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
total_tokens=prompt_tokens + completion_tokens,
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=0, cache_write_tokens=cache_write_tokens),
)
def test_batch_cost_calculator_bills_cache_write_tokens_at_the_long_context_batch_cache_write_rate(
_local_model_cost_map,
):
from litellm.cost_calculator import batch_cost_calculator
prompt_cost, completion_cost = batch_cost_calculator(
usage=_cache_write_usage(300_048, 300_045, 4), model="gpt-5.6-luna", custom_llm_provider="openai"
)
assert prompt_cost == pytest.approx(3 * 2e-7 + 300_045 * 2.5e-7)
assert completion_cost == pytest.approx(4 * 9e-7)
def test_batch_cost_calculator_bills_cache_write_tokens_at_the_flat_batch_cache_write_rate_at_or_below_272k(
_local_model_cost_map,
):
from litellm.cost_calculator import batch_cost_calculator
prompt_cost, _ = batch_cost_calculator(
usage=_cache_write_usage(1_000, 900, 4), model="gpt-5.6-luna", custom_llm_provider="openai"
)
assert prompt_cost == pytest.approx(100 * 1e-7 + 900 * 1.25e-7)
def test_batch_cost_calculator_bills_cache_write_tokens_at_the_batch_input_rate_without_a_cache_write_batch_rate(
_local_model_cost_map,
):
from litellm.cost_calculator import batch_cost_calculator
prompt_cost, _ = batch_cost_calculator(
usage=_cache_write_usage(300_048, 300_045, 4), model="gpt-5.5", custom_llm_provider="openai"
)
assert prompt_cost == pytest.approx(300_048 * 5e-6)
def test_get_model_info_exposes_the_batch_cache_write_rates(_local_model_cost_map):
info = litellm.get_model_info("gpt-5.6-luna", custom_llm_provider="openai")
assert info["cache_creation_input_token_cost_batches"] == 1.25e-7
assert info["cache_creation_input_token_cost_above_272k_tokens_batches"] == 2.5e-7
_OPENAI_ENTRIES_WITH_BATCH_CACHE_WRITE_RATES = frozenset(
{"gpt-6-astra", "gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"}
)
def test_openai_batch_cache_write_rates_are_half_the_standard_cache_write_rates(_local_model_cost_map):
carriers = {
name: entry
for name, entry in litellm.model_cost.items()
if isinstance(entry, dict)
and entry.get("litellm_provider") == "openai"
and entry.get("cache_creation_input_token_cost") is not None
and entry.get("input_cost_per_token_batches") is not None
}
assert _OPENAI_ENTRIES_WITH_BATCH_CACHE_WRITE_RATES <= set(carriers)
for entry in carriers.values():
assert entry["cache_creation_input_token_cost_batches"] == entry["cache_creation_input_token_cost"] / 2
assert (
entry["cache_creation_input_token_cost_above_272k_tokens_batches"]
== entry["cache_creation_input_token_cost_above_272k_tokens"] / 2
)

View file

@ -916,6 +916,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"cache_creation_input_token_cost_above_272k_tokens_priority": {
"type": "number"
},
"cache_creation_input_token_cost_above_272k_tokens_batches": {"type": "number"},
"cache_creation_input_token_cost_batches": {"type": "number"},
"cache_creation_input_token_cost_flex": {"type": "number"},
"cache_creation_input_token_cost_priority": {"type": "number"},
"cache_read_input_token_cost": {"type": "number"},

View file

@ -29316,10 +29316,14 @@ export interface components {
cache_creation_input_token_cost_above_200k_tokens?: number | null;
/** Cache Creation Input Token Cost Above 272K Tokens */
cache_creation_input_token_cost_above_272k_tokens?: number | null;
/** Cache Creation Input Token Cost Above 272K Tokens Batches */
cache_creation_input_token_cost_above_272k_tokens_batches?: number | null;
/** Cache Creation Input Token Cost Above 272K Tokens Flex */
cache_creation_input_token_cost_above_272k_tokens_flex?: number | null;
/** Cache Creation Input Token Cost Above 272K Tokens Priority */
cache_creation_input_token_cost_above_272k_tokens_priority?: number | null;
/** Cache Creation Input Token Cost Batches */
cache_creation_input_token_cost_batches?: number | null;
/** Cache Creation Input Token Cost Flex */
cache_creation_input_token_cost_flex?: number | null;
/** Cache Creation Input Token Cost Priority */
@ -39454,10 +39458,14 @@ export interface components {
cache_creation_input_token_cost_above_200k_tokens?: number | null;
/** Cache Creation Input Token Cost Above 272K Tokens */
cache_creation_input_token_cost_above_272k_tokens?: number | null;
/** Cache Creation Input Token Cost Above 272K Tokens Batches */
cache_creation_input_token_cost_above_272k_tokens_batches?: number | null;
/** Cache Creation Input Token Cost Above 272K Tokens Flex */
cache_creation_input_token_cost_above_272k_tokens_flex?: number | null;
/** Cache Creation Input Token Cost Above 272K Tokens Priority */
cache_creation_input_token_cost_above_272k_tokens_priority?: number | null;
/** Cache Creation Input Token Cost Batches */
cache_creation_input_token_cost_batches?: number | null;
/** Cache Creation Input Token Cost Flex */
cache_creation_input_token_cost_flex?: number | null;
/** Cache Creation Input Token Cost Priority */