diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index 21ed49afb54..f23166c5e7b 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -170,6 +170,9 @@ COST_DESCRIPTIONS: dict[str, str] = { "input_cost_per_token_batches": "USD per prompt token via the provider's batch API.", "output_cost_per_token_batches": "USD per generated token via the provider's batch API.", "cache_read_input_token_cost_batches": "USD per cached prompt token via the provider's batch API.", + "cache_creation_input_token_cost_batches": ( + "USD per token written to the provider's prompt cache via its batch API." + ), } diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index a43f1af0940..4f14385902d 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -2248,7 +2248,9 @@ def batch_cost_calculator( total_prompt_cost = 0.0 total_completion_cost = 0.0 if batch_rates.input is not None: - total_prompt_cost = _batch_prompt_cost(usage, batch_rates.input, batch_rates.cache_read) + total_prompt_cost = _batch_prompt_cost( + usage, batch_rates.input, batch_rates.cache_read, batch_rates.cache_creation + ) elif input_cost_per_token: details: Final = parse_prompt_tokens_details(usage) cache_read_tokens: Final = details["cache_hit_tokens"] @@ -2282,11 +2284,17 @@ def batch_cost_calculator( return total_prompt_cost, total_completion_cost -def _batch_prompt_cost(usage: Usage, input_rate: float, cache_read_rate: float | None) -> float: - if cache_read_rate is None: - return usage.prompt_tokens * input_rate - cached_tokens: Final = parse_prompt_tokens_details(usage)["cache_hit_tokens"] - return get_billable_input_tokens(usage) * input_rate + cached_tokens * cache_read_rate +def _batch_prompt_cost( + usage: Usage, input_rate: float, cache_read_rate: float | None, cache_creation_rate: float | None +) -> float: + details: Final = parse_prompt_tokens_details(usage) + cached_tokens: Final = details["cache_hit_tokens"] if cache_read_rate is not None else 0 + written_tokens: Final = details["cache_creation_tokens"] if cache_creation_rate is not None else 0 + return ( + (usage.prompt_tokens - cached_tokens - written_tokens) * input_rate + + cached_tokens * (cache_read_rate or 0.0) + + written_tokens * (cache_creation_rate or 0.0) + ) def _attribute_value(obj: object, name: str) -> object: diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index a175ca1c3f6..8a36ce609d3 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -330,7 +330,36 @@ _DEPLOYMENT_PRICING_KEYS: Final = ( "output_cost_per_token", "input_cost_per_token_batches", "output_cost_per_token_batches", + "input_cost_per_token_above_272k_tokens_batches", + "output_cost_per_token_above_272k_tokens_batches", + "cache_read_input_token_cost_batches", + "cache_read_input_token_cost_above_272k_tokens_batches", + "cache_creation_input_token_cost_batches", + "cache_creation_input_token_cost_above_272k_tokens_batches", ) +_INPUT_PRICING_KEY_PREFIXES: Final = ( + "input_cost_per_token", + "cache_read_input_token_cost", + "cache_creation_input_token_cost", +) +_OUTPUT_PRICING_KEY_PREFIXES: Final = ("output_cost_per_token",) +_BATCH_PRICING_KEY_SUFFIX: Final = "_batches" + + +_NO_CARRIED_RATES: Final[Mapping[str, object]] = MappingProxyType({}) + + +def _published_direction( + published: ModelInfo, registered: Mapping[str, object], flat_key: str, prefixes: tuple[str, ...] +) -> Mapping[str, object]: + return MappingProxyType( + { + key: value + for key, value in published.items() + if registered.get(key) is None + and (key == flat_key or (key.startswith(prefixes) and key.endswith(_BATCH_PRICING_KEY_SUFFIX))) + } + ) def deployment_pricing_model_info(model_id: str | None, deployment_model: str | None) -> ModelInfo | None: @@ -342,10 +371,11 @@ def deployment_pricing_model_info(model_id: str | None, deployment_model: str | get_model_info fills absent costs with 0, so asking it directly cannot tell "configured as free" apart from "no pricing configured". A deployment may declare only one side of its pricing, so the side it leaves out keeps - the model's published rates instead of billing as zero. Ownership is per - token direction: declaring either rate for a direction takes that whole - direction, so a published batch rate can never displace a standard rate - the deployment configured itself. + the model's published standard rate and every published batch rate for + that direction (flat, long-context tier, cached, cache write) instead of + billing as zero. Ownership is per token direction: declaring either rate + for a direction takes that whole direction, so a published batch rate can + never displace a standard rate the deployment configured itself. """ if model_id is None: return None @@ -366,13 +396,22 @@ def deployment_pricing_model_info(model_id: str | None, deployment_model: str | registered.get("output_cost_per_token") is not None or registered.get("output_cost_per_token_batches") is not None ) - if not declares_input: - merged["input_cost_per_token"] = published.get("input_cost_per_token") - merged["input_cost_per_token_batches"] = published.get("input_cost_per_token_batches") - if not declares_output: - merged["output_cost_per_token"] = published.get("output_cost_per_token") - merged["output_cost_per_token_batches"] = published.get("output_cost_per_token_batches") - return merged + carried_input: Final = ( + _NO_CARRIED_RATES + if declares_input + else _published_direction(published, registered, "input_cost_per_token", _INPUT_PRICING_KEY_PREFIXES) + ) + carried_output: Final = ( + _NO_CARRIED_RATES + if declares_output + else _published_direction(published, registered, "output_cost_per_token", _OUTPUT_PRICING_KEY_PREFIXES) + ) + priced: Final[ModelInfo] = { # pyright: ignore[reportAssignmentType] # carried keys are ModelInfo rates + **merged, + **carried_input, + **carried_output, + } + return priced def _published_pricing(deployment_model: str | None) -> ModelInfo | None: diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index 2b9fb004bb1..5eefe88a2e2 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -253,6 +253,7 @@ class BatchCostRates: input: float | None output: float | None cache_read: float | None + cache_creation: float | None def _batch_rate(model_info: ModelInfo, key: str) -> float | None: @@ -290,6 +291,7 @@ def get_batch_cost_rates(model_info: ModelInfo, usage: Usage, custom_llm_provide input=_batch_rate(model_info, "input_cost_per_token_batches"), output=_batch_rate(model_info, "output_cost_per_token_batches"), cache_read=_batch_rate(model_info, "cache_read_input_token_cost_batches"), + cache_creation=_batch_rate(model_info, "cache_creation_input_token_cost_batches"), ) return BatchCostRates( input=_batch_tier_rate(model_info, crossed_input_key, "input_cost_per_token_batches"), @@ -301,6 +303,11 @@ def get_batch_cost_rates(model_info: ModelInfo, usage: Usage, custom_llm_provide crossed_input_key.replace("input_cost_per_token", "cache_read_input_token_cost", 1), "cache_read_input_token_cost_batches", ), + cache_creation=_batch_tier_rate( + model_info, + crossed_input_key.replace("input_cost_per_token", "cache_creation_input_token_cost", 1), + "cache_creation_input_token_cost_batches", + ), ) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 8c862c6ead5..9c445f03776 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -30154,6 +30154,8 @@ "cache_read_input_token_cost_above_272k_tokens_priority": 4e-06, "cache_read_input_token_cost_batches": 5e-07, "cache_read_input_token_cost_above_272k_tokens_batches": 1e-06, + "cache_creation_input_token_cost_batches": 6.25e-06, + "cache_creation_input_token_cost_above_272k_tokens_batches": 1.25e-05, "cache_read_input_token_cost_flex": 5e-07, "cache_read_input_token_cost_priority": 2e-06, "input_cost_per_token": 1e-05, @@ -30227,6 +30229,8 @@ "cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06, "cache_read_input_token_cost_batches": 2e-07, "cache_read_input_token_cost_above_272k_tokens_batches": 4e-07, + "cache_creation_input_token_cost_batches": 2.5e-06, + "cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06, "cache_read_input_token_cost_flex": 2e-07, "cache_read_input_token_cost_priority": 8e-07, "input_cost_per_token": 4e-06, @@ -30298,6 +30302,8 @@ "cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06, "cache_read_input_token_cost_batches": 2e-07, "cache_read_input_token_cost_above_272k_tokens_batches": 4e-07, + "cache_creation_input_token_cost_batches": 2.5e-06, + "cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06, "cache_read_input_token_cost_flex": 2e-07, "cache_read_input_token_cost_priority": 8e-07, "input_cost_per_token": 4e-06, @@ -30370,6 +30376,8 @@ "cache_read_input_token_cost_above_272k_tokens_priority": 8e-07, "cache_read_input_token_cost_batches": 1e-07, "cache_read_input_token_cost_above_272k_tokens_batches": 2e-07, + "cache_creation_input_token_cost_batches": 1.25e-06, + "cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-06, "cache_read_input_token_cost_flex": 1e-07, "cache_read_input_token_cost_priority": 4e-07, "input_cost_per_token": 2e-06, @@ -30441,6 +30449,8 @@ "cache_read_input_token_cost_above_272k_tokens_priority": 8e-08, "cache_read_input_token_cost_batches": 1e-08, "cache_read_input_token_cost_above_272k_tokens_batches": 2e-08, + "cache_creation_input_token_cost_batches": 1.25e-07, + "cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-07, "cache_read_input_token_cost_flex": 1e-08, "cache_read_input_token_cost_priority": 4e-08, "input_cost_per_token": 2e-07, diff --git a/litellm/types/utils.py b/litellm/types/utils.py index c2e0179e103..98db6e5da44 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -258,6 +258,8 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False): cache_read_input_token_cost_above_512k_tokens: float | None cache_read_input_token_cost_batches: ReadOnly[float | None] cache_read_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None] + cache_creation_input_token_cost_batches: ReadOnly[float | None] + cache_creation_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None] # Smallest prefix this model will actually cache, whatever caching mechanism its provider uses. # Absent means the provider-agnostic default applies; see MINIMUM_PROMPT_CACHE_TOKEN_COUNT. prompt_cache_min_tokens: int | None @@ -3521,6 +3523,8 @@ class CustomPricingLiteLLMParams(MirroredPricingParams): cache_read_input_token_cost_above_272k_tokens_flex: float | None = None cache_read_input_token_cost_batches: float | None = None cache_read_input_token_cost_above_272k_tokens_batches: float | None = None + cache_creation_input_token_cost_batches: float | None = None + cache_creation_input_token_cost_above_272k_tokens_batches: float | None = None cache_read_input_audio_token_cost: float | None = None input_cost_per_character_above_128k_tokens: float | None = None input_cost_per_audio_token: float | None = None diff --git a/litellm/utils.py b/litellm/utils.py index b041d44b7d5..86393b1826b 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -5822,6 +5822,10 @@ def _get_model_info_helper( cache_read_input_token_cost_above_272k_tokens_batches=_model_info.get( "cache_read_input_token_cost_above_272k_tokens_batches" ), + cache_creation_input_token_cost_batches=_model_info.get("cache_creation_input_token_cost_batches"), + cache_creation_input_token_cost_above_272k_tokens_batches=_model_info.get( + "cache_creation_input_token_cost_above_272k_tokens_batches" + ), cache_creation_input_token_cost_above_1hr=_model_info.get( "cache_creation_input_token_cost_above_1hr", None ), diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 8c862c6ead5..9c445f03776 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -30154,6 +30154,8 @@ "cache_read_input_token_cost_above_272k_tokens_priority": 4e-06, "cache_read_input_token_cost_batches": 5e-07, "cache_read_input_token_cost_above_272k_tokens_batches": 1e-06, + "cache_creation_input_token_cost_batches": 6.25e-06, + "cache_creation_input_token_cost_above_272k_tokens_batches": 1.25e-05, "cache_read_input_token_cost_flex": 5e-07, "cache_read_input_token_cost_priority": 2e-06, "input_cost_per_token": 1e-05, @@ -30227,6 +30229,8 @@ "cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06, "cache_read_input_token_cost_batches": 2e-07, "cache_read_input_token_cost_above_272k_tokens_batches": 4e-07, + "cache_creation_input_token_cost_batches": 2.5e-06, + "cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06, "cache_read_input_token_cost_flex": 2e-07, "cache_read_input_token_cost_priority": 8e-07, "input_cost_per_token": 4e-06, @@ -30298,6 +30302,8 @@ "cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06, "cache_read_input_token_cost_batches": 2e-07, "cache_read_input_token_cost_above_272k_tokens_batches": 4e-07, + "cache_creation_input_token_cost_batches": 2.5e-06, + "cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06, "cache_read_input_token_cost_flex": 2e-07, "cache_read_input_token_cost_priority": 8e-07, "input_cost_per_token": 4e-06, @@ -30370,6 +30376,8 @@ "cache_read_input_token_cost_above_272k_tokens_priority": 8e-07, "cache_read_input_token_cost_batches": 1e-07, "cache_read_input_token_cost_above_272k_tokens_batches": 2e-07, + "cache_creation_input_token_cost_batches": 1.25e-06, + "cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-06, "cache_read_input_token_cost_flex": 1e-07, "cache_read_input_token_cost_priority": 4e-07, "input_cost_per_token": 2e-06, @@ -30441,6 +30449,8 @@ "cache_read_input_token_cost_above_272k_tokens_priority": 8e-08, "cache_read_input_token_cost_batches": 1e-08, "cache_read_input_token_cost_above_272k_tokens_batches": 2e-08, + "cache_creation_input_token_cost_batches": 1.25e-07, + "cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-07, "cache_read_input_token_cost_flex": 1e-08, "cache_read_input_token_cost_priority": 4e-08, "input_cost_per_token": 2e-07, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index 615836095c2..2f8fabc370a 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -109,6 +109,11 @@ "minimum": 0, "description": "Rate applied once the prompt exceeds the token threshold in the field name." }, + "cache_creation_input_token_cost_above_272k_tokens_batches": { + "type": "number", + "minimum": 0, + "description": "Batch API rate applied once the prompt exceeds the token threshold in the field name." + }, "cache_creation_input_token_cost_above_272k_tokens_flex": { "type": "number", "minimum": 0, @@ -119,6 +124,11 @@ "minimum": 0, "description": "Priority service-tier rate for the same-named base field." }, + "cache_creation_input_token_cost_batches": { + "type": "number", + "minimum": 0, + "description": "USD per token written to the provider's prompt cache via its batch API." + }, "cache_creation_input_token_cost_flex": { "type": "number", "minimum": 0, diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index a3d8815fbd6..acabc1e1291 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -1777,7 +1777,7 @@ def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map): sol = litellm.model_cost["gpt-5.6-sol"] cost_fields = sorted(field for field in sol if "cost" in field) - assert len(cost_fields) == 31 + assert len(cost_fields) == 33 for field in cost_fields: assert alias.get(field) == sol.get(field), field @@ -4803,12 +4803,14 @@ def test_get_batch_cost_rates_falls_back_to_the_flat_rates_when_tier_rates_are_u input_cost_per_token_above_272k_tokens_batches="two", output_cost_per_token_above_272k_tokens_batches="six", cache_read_input_token_cost_above_272k_tokens_batches="none", + cache_creation_input_token_cost_batches=1.25e-7, + cache_creation_input_token_cost_above_272k_tokens_batches="nope", ), Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001), "openai", ) - assert (rates.input, rates.output, rates.cache_read) == (1e-6, 4e-6, 1e-7) + assert (rates.input, rates.output, rates.cache_read, rates.cache_creation) == (1e-6, 4e-6, 1e-7, 1.25e-7) def test_get_batch_cost_rates_has_no_cached_rate_without_a_cached_batch_key(): @@ -4826,3 +4828,38 @@ def test_get_batch_cost_rates_has_no_cached_rate_without_a_cached_batch_key(): ) assert (rates.input, rates.output, rates.cache_read) == (2e-6, 4e-6, None) + + +@pytest.mark.parametrize(("prompt_tokens", "expected"), [(1_000, 1.25e-7), (272_000, 1.25e-7), (300_000, 2.5e-7)]) +def test_get_batch_cost_rates_reads_the_batch_cache_write_rate_for_the_crossed_tier(prompt_tokens, expected): + from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates + + rates = get_batch_cost_rates( + _batch_rates_model_info( + input_cost_per_token_batches=1e-7, + input_cost_per_token_above_272k_tokens_batches=2e-7, + cache_creation_input_token_cost_batches=1.25e-7, + cache_creation_input_token_cost_above_272k_tokens_batches=2.5e-7, + ), + Usage(prompt_tokens=prompt_tokens, completion_tokens=1, total_tokens=prompt_tokens + 1), + "openai", + ) + + assert rates.cache_creation == expected + + +def test_get_batch_cost_rates_has_no_cache_write_rate_without_a_cache_write_batch_key(): + from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates + + rates = get_batch_cost_rates( + _batch_rates_model_info( + input_cost_per_token_batches=1e-7, + input_cost_per_token_above_272k_tokens_batches=2e-7, + cache_creation_input_token_cost=2.5e-7, + cache_creation_input_token_cost_above_272k_tokens=5e-7, + ), + Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001), + "openai", + ) + + assert rates.cache_creation is None diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index f1de7390b5b..4103945a183 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -6277,3 +6277,78 @@ def test_passthrough_embeddings_result_swapped_for_callbacks(): assert isinstance(swapped_result, EmbeddingResponse) assert swapped_result.data[0]["embedding"] == [0.1, 0.2, 0.3] + + +@pytest.fixture +def _local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + + +def _luna_deployment_id(custom_pricing: dict[str, float]) -> str: + from litellm import Router + + router = Router( + model_list=[ + { + "model_name": "luna-batch", + "litellm_params": {"model": "openai/gpt-5.6-luna", "api_key": "sk-test", **custom_pricing}, + } + ] + ) + return router.model_list[0]["model_info"]["id"] + + +def test_deployment_pricing_model_info_carries_every_published_input_batch_rate_when_only_output_is_declared( + _local_model_cost_map, +): + from litellm.litellm_core_utils.litellm_logging import deployment_pricing_model_info + + info = deployment_pricing_model_info( + _luna_deployment_id({"output_cost_per_token_batches": 4e-6}), "openai/gpt-5.6-luna" + ) + + assert info is not None + assert info["input_cost_per_token_batches"] == 1e-7 + assert info["input_cost_per_token_above_272k_tokens_batches"] == 2e-7 + assert info["cache_read_input_token_cost_batches"] == 1e-8 + assert info["cache_read_input_token_cost_above_272k_tokens_batches"] == 2e-8 + assert info["cache_creation_input_token_cost_batches"] == 1.25e-7 + assert info["cache_creation_input_token_cost_above_272k_tokens_batches"] == 2.5e-7 + assert info["output_cost_per_token_batches"] == 4e-6 + assert info["output_cost_per_token_above_272k_tokens_batches"] is None + + +def test_deployment_pricing_model_info_carries_the_published_output_batch_tier_when_only_input_is_declared( + _local_model_cost_map, +): + from litellm.litellm_core_utils.litellm_logging import deployment_pricing_model_info + + info = deployment_pricing_model_info( + _luna_deployment_id({"input_cost_per_token_batches": 1e-6}), "openai/gpt-5.6-luna" + ) + + assert info is not None + assert info["output_cost_per_token_batches"] == 6e-7 + assert info["output_cost_per_token_above_272k_tokens_batches"] == 9e-7 + assert info["input_cost_per_token_batches"] == 1e-6 + assert info["input_cost_per_token_above_272k_tokens_batches"] is None + assert info["cache_read_input_token_cost_batches"] is None + assert info["cache_creation_input_token_cost_batches"] is None + + +def test_deployment_pricing_model_info_honors_a_tier_only_batch_override_over_the_published_flat_rates( + _local_model_cost_map: None, +) -> None: + from litellm.litellm_core_utils.litellm_logging import deployment_pricing_model_info + + info = deployment_pricing_model_info( + _luna_deployment_id({"input_cost_per_token_above_272k_tokens_batches": 1e-3}), "openai/gpt-5.6-luna" + ) + + assert info is not None + assert info["input_cost_per_token_above_272k_tokens_batches"] == 1e-3 + assert info["input_cost_per_token_batches"] == 1e-7 + assert info["cache_read_input_token_cost_batches"] == 1e-8 + assert info["output_cost_per_token_batches"] == 6e-7 + assert info["output_cost_per_token_above_272k_tokens_batches"] == 9e-7 diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 4d7d6a13949..7f27e03eeb2 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -4673,3 +4673,80 @@ def test_openai_cached_batch_rates_are_half_the_standard_cached_rates(_local_mod entry["cache_read_input_token_cost_above_272k_tokens_batches"] == entry["cache_read_input_token_cost_above_272k_tokens"] / 2 ) + + +def _cache_write_usage(prompt_tokens: int, cache_write_tokens: int, completion_tokens: int) -> Usage: + return Usage( + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + total_tokens=prompt_tokens + completion_tokens, + prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=0, cache_write_tokens=cache_write_tokens), + ) + + +def test_batch_cost_calculator_bills_cache_write_tokens_at_the_long_context_batch_cache_write_rate( + _local_model_cost_map, +): + from litellm.cost_calculator import batch_cost_calculator + + prompt_cost, completion_cost = batch_cost_calculator( + usage=_cache_write_usage(300_048, 300_045, 4), model="gpt-5.6-luna", custom_llm_provider="openai" + ) + + assert prompt_cost == pytest.approx(3 * 2e-7 + 300_045 * 2.5e-7) + assert completion_cost == pytest.approx(4 * 9e-7) + + +def test_batch_cost_calculator_bills_cache_write_tokens_at_the_flat_batch_cache_write_rate_at_or_below_272k( + _local_model_cost_map, +): + from litellm.cost_calculator import batch_cost_calculator + + prompt_cost, _ = batch_cost_calculator( + usage=_cache_write_usage(1_000, 900, 4), model="gpt-5.6-luna", custom_llm_provider="openai" + ) + + assert prompt_cost == pytest.approx(100 * 1e-7 + 900 * 1.25e-7) + + +def test_batch_cost_calculator_bills_cache_write_tokens_at_the_batch_input_rate_without_a_cache_write_batch_rate( + _local_model_cost_map, +): + from litellm.cost_calculator import batch_cost_calculator + + prompt_cost, _ = batch_cost_calculator( + usage=_cache_write_usage(300_048, 300_045, 4), model="gpt-5.5", custom_llm_provider="openai" + ) + + assert prompt_cost == pytest.approx(300_048 * 5e-6) + + +def test_get_model_info_exposes_the_batch_cache_write_rates(_local_model_cost_map): + info = litellm.get_model_info("gpt-5.6-luna", custom_llm_provider="openai") + + assert info["cache_creation_input_token_cost_batches"] == 1.25e-7 + assert info["cache_creation_input_token_cost_above_272k_tokens_batches"] == 2.5e-7 + + +_OPENAI_ENTRIES_WITH_BATCH_CACHE_WRITE_RATES = frozenset( + {"gpt-6-astra", "gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"} +) + + +def test_openai_batch_cache_write_rates_are_half_the_standard_cache_write_rates(_local_model_cost_map): + carriers = { + name: entry + for name, entry in litellm.model_cost.items() + if isinstance(entry, dict) + and entry.get("litellm_provider") == "openai" + and entry.get("cache_creation_input_token_cost") is not None + and entry.get("input_cost_per_token_batches") is not None + } + + assert _OPENAI_ENTRIES_WITH_BATCH_CACHE_WRITE_RATES <= set(carriers) + for entry in carriers.values(): + assert entry["cache_creation_input_token_cost_batches"] == entry["cache_creation_input_token_cost"] / 2 + assert ( + entry["cache_creation_input_token_cost_above_272k_tokens_batches"] + == entry["cache_creation_input_token_cost_above_272k_tokens"] / 2 + ) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 4c833b34a0c..2532a687e70 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -916,6 +916,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_creation_input_token_cost_above_272k_tokens_priority": { "type": "number" }, + "cache_creation_input_token_cost_above_272k_tokens_batches": {"type": "number"}, + "cache_creation_input_token_cost_batches": {"type": "number"}, "cache_creation_input_token_cost_flex": {"type": "number"}, "cache_creation_input_token_cost_priority": {"type": "number"}, "cache_read_input_token_cost": {"type": "number"}, diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 9bc513e38c6..1afd0424761 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -29316,10 +29316,14 @@ export interface components { cache_creation_input_token_cost_above_200k_tokens?: number | null; /** Cache Creation Input Token Cost Above 272K Tokens */ cache_creation_input_token_cost_above_272k_tokens?: number | null; + /** Cache Creation Input Token Cost Above 272K Tokens Batches */ + cache_creation_input_token_cost_above_272k_tokens_batches?: number | null; /** Cache Creation Input Token Cost Above 272K Tokens Flex */ cache_creation_input_token_cost_above_272k_tokens_flex?: number | null; /** Cache Creation Input Token Cost Above 272K Tokens Priority */ cache_creation_input_token_cost_above_272k_tokens_priority?: number | null; + /** Cache Creation Input Token Cost Batches */ + cache_creation_input_token_cost_batches?: number | null; /** Cache Creation Input Token Cost Flex */ cache_creation_input_token_cost_flex?: number | null; /** Cache Creation Input Token Cost Priority */ @@ -39454,10 +39458,14 @@ export interface components { cache_creation_input_token_cost_above_200k_tokens?: number | null; /** Cache Creation Input Token Cost Above 272K Tokens */ cache_creation_input_token_cost_above_272k_tokens?: number | null; + /** Cache Creation Input Token Cost Above 272K Tokens Batches */ + cache_creation_input_token_cost_above_272k_tokens_batches?: number | null; /** Cache Creation Input Token Cost Above 272K Tokens Flex */ cache_creation_input_token_cost_above_272k_tokens_flex?: number | null; /** Cache Creation Input Token Cost Above 272K Tokens Priority */ cache_creation_input_token_cost_above_272k_tokens_priority?: number | null; + /** Cache Creation Input Token Cost Batches */ + cache_creation_input_token_cost_batches?: number | null; /** Cache Creation Input Token Cost Flex */ cache_creation_input_token_cost_flex?: number | null; /** Cache Creation Input Token Cost Priority */