mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(cost): bill batch cache writes at the batch cache-write rate and carry published batch rates for one-sided deployments
OpenAI's Batch table prices cache writes for gpt-6-astra, gpt-5.6, gpt-5.6-sol, gpt-5.6-terra and gpt-5.6-luna at half the standard cache-write rate, so the cost map gains cache_creation_input_token_cost_batches and its above_272k tier for those entries and batch cost pulls written tokens out of the input bucket at that rate; models without the key keep billing writes at the batch input rate. A deployment declaring only one side of its batch pricing now carries every published batch rate of the other side (tier, cached, cache write), its own keys win, and a lone tier, cached or cache-write batch key counts as declared pricing instead of being ignored.
This commit is contained in:
parent
ab3c924174
commit
3f723bb918
14 changed files with 313 additions and 19 deletions
|
|
@ -170,6 +170,9 @@ COST_DESCRIPTIONS: dict[str, str] = {
|
|||
"input_cost_per_token_batches": "USD per prompt token via the provider's batch API.",
|
||||
"output_cost_per_token_batches": "USD per generated token via the provider's batch API.",
|
||||
"cache_read_input_token_cost_batches": "USD per cached prompt token via the provider's batch API.",
|
||||
"cache_creation_input_token_cost_batches": (
|
||||
"USD per token written to the provider's prompt cache via its batch API."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -2248,7 +2248,9 @@ def batch_cost_calculator(
|
|||
total_prompt_cost = 0.0
|
||||
total_completion_cost = 0.0
|
||||
if batch_rates.input is not None:
|
||||
total_prompt_cost = _batch_prompt_cost(usage, batch_rates.input, batch_rates.cache_read)
|
||||
total_prompt_cost = _batch_prompt_cost(
|
||||
usage, batch_rates.input, batch_rates.cache_read, batch_rates.cache_creation
|
||||
)
|
||||
elif input_cost_per_token:
|
||||
details: Final = parse_prompt_tokens_details(usage)
|
||||
cache_read_tokens: Final = details["cache_hit_tokens"]
|
||||
|
|
@ -2282,11 +2284,17 @@ def batch_cost_calculator(
|
|||
return total_prompt_cost, total_completion_cost
|
||||
|
||||
|
||||
def _batch_prompt_cost(usage: Usage, input_rate: float, cache_read_rate: float | None) -> float:
|
||||
if cache_read_rate is None:
|
||||
return usage.prompt_tokens * input_rate
|
||||
cached_tokens: Final = parse_prompt_tokens_details(usage)["cache_hit_tokens"]
|
||||
return get_billable_input_tokens(usage) * input_rate + cached_tokens * cache_read_rate
|
||||
def _batch_prompt_cost(
|
||||
usage: Usage, input_rate: float, cache_read_rate: float | None, cache_creation_rate: float | None
|
||||
) -> float:
|
||||
details: Final = parse_prompt_tokens_details(usage)
|
||||
cached_tokens: Final = details["cache_hit_tokens"] if cache_read_rate is not None else 0
|
||||
written_tokens: Final = details["cache_creation_tokens"] if cache_creation_rate is not None else 0
|
||||
return (
|
||||
(usage.prompt_tokens - cached_tokens - written_tokens) * input_rate
|
||||
+ cached_tokens * (cache_read_rate or 0.0)
|
||||
+ written_tokens * (cache_creation_rate or 0.0)
|
||||
)
|
||||
|
||||
|
||||
def _attribute_value(obj: object, name: str) -> object:
|
||||
|
|
|
|||
|
|
@ -330,7 +330,36 @@ _DEPLOYMENT_PRICING_KEYS: Final = (
|
|||
"output_cost_per_token",
|
||||
"input_cost_per_token_batches",
|
||||
"output_cost_per_token_batches",
|
||||
"input_cost_per_token_above_272k_tokens_batches",
|
||||
"output_cost_per_token_above_272k_tokens_batches",
|
||||
"cache_read_input_token_cost_batches",
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches",
|
||||
"cache_creation_input_token_cost_batches",
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches",
|
||||
)
|
||||
_INPUT_PRICING_KEY_PREFIXES: Final = (
|
||||
"input_cost_per_token",
|
||||
"cache_read_input_token_cost",
|
||||
"cache_creation_input_token_cost",
|
||||
)
|
||||
_OUTPUT_PRICING_KEY_PREFIXES: Final = ("output_cost_per_token",)
|
||||
_BATCH_PRICING_KEY_SUFFIX: Final = "_batches"
|
||||
|
||||
|
||||
_NO_CARRIED_RATES: Final[Mapping[str, object]] = MappingProxyType({})
|
||||
|
||||
|
||||
def _published_direction(
|
||||
published: ModelInfo, registered: Mapping[str, object], flat_key: str, prefixes: tuple[str, ...]
|
||||
) -> Mapping[str, object]:
|
||||
return MappingProxyType(
|
||||
{
|
||||
key: value
|
||||
for key, value in published.items()
|
||||
if registered.get(key) is None
|
||||
and (key == flat_key or (key.startswith(prefixes) and key.endswith(_BATCH_PRICING_KEY_SUFFIX)))
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def deployment_pricing_model_info(model_id: str | None, deployment_model: str | None) -> ModelInfo | None:
|
||||
|
|
@ -342,10 +371,11 @@ def deployment_pricing_model_info(model_id: str | None, deployment_model: str |
|
|||
get_model_info fills absent costs with 0, so asking it directly cannot
|
||||
tell "configured as free" apart from "no pricing configured". A deployment
|
||||
may declare only one side of its pricing, so the side it leaves out keeps
|
||||
the model's published rates instead of billing as zero. Ownership is per
|
||||
token direction: declaring either rate for a direction takes that whole
|
||||
direction, so a published batch rate can never displace a standard rate
|
||||
the deployment configured itself.
|
||||
the model's published standard rate and every published batch rate for
|
||||
that direction (flat, long-context tier, cached, cache write) instead of
|
||||
billing as zero. Ownership is per token direction: declaring either rate
|
||||
for a direction takes that whole direction, so a published batch rate can
|
||||
never displace a standard rate the deployment configured itself.
|
||||
"""
|
||||
if model_id is None:
|
||||
return None
|
||||
|
|
@ -366,13 +396,22 @@ def deployment_pricing_model_info(model_id: str | None, deployment_model: str |
|
|||
registered.get("output_cost_per_token") is not None
|
||||
or registered.get("output_cost_per_token_batches") is not None
|
||||
)
|
||||
if not declares_input:
|
||||
merged["input_cost_per_token"] = published.get("input_cost_per_token")
|
||||
merged["input_cost_per_token_batches"] = published.get("input_cost_per_token_batches")
|
||||
if not declares_output:
|
||||
merged["output_cost_per_token"] = published.get("output_cost_per_token")
|
||||
merged["output_cost_per_token_batches"] = published.get("output_cost_per_token_batches")
|
||||
return merged
|
||||
carried_input: Final = (
|
||||
_NO_CARRIED_RATES
|
||||
if declares_input
|
||||
else _published_direction(published, registered, "input_cost_per_token", _INPUT_PRICING_KEY_PREFIXES)
|
||||
)
|
||||
carried_output: Final = (
|
||||
_NO_CARRIED_RATES
|
||||
if declares_output
|
||||
else _published_direction(published, registered, "output_cost_per_token", _OUTPUT_PRICING_KEY_PREFIXES)
|
||||
)
|
||||
priced: Final[ModelInfo] = { # pyright: ignore[reportAssignmentType] # carried keys are ModelInfo rates
|
||||
**merged,
|
||||
**carried_input,
|
||||
**carried_output,
|
||||
}
|
||||
return priced
|
||||
|
||||
|
||||
def _published_pricing(deployment_model: str | None) -> ModelInfo | None:
|
||||
|
|
|
|||
|
|
@ -253,6 +253,7 @@ class BatchCostRates:
|
|||
input: float | None
|
||||
output: float | None
|
||||
cache_read: float | None
|
||||
cache_creation: float | None
|
||||
|
||||
|
||||
def _batch_rate(model_info: ModelInfo, key: str) -> float | None:
|
||||
|
|
@ -290,6 +291,7 @@ def get_batch_cost_rates(model_info: ModelInfo, usage: Usage, custom_llm_provide
|
|||
input=_batch_rate(model_info, "input_cost_per_token_batches"),
|
||||
output=_batch_rate(model_info, "output_cost_per_token_batches"),
|
||||
cache_read=_batch_rate(model_info, "cache_read_input_token_cost_batches"),
|
||||
cache_creation=_batch_rate(model_info, "cache_creation_input_token_cost_batches"),
|
||||
)
|
||||
return BatchCostRates(
|
||||
input=_batch_tier_rate(model_info, crossed_input_key, "input_cost_per_token_batches"),
|
||||
|
|
@ -301,6 +303,11 @@ def get_batch_cost_rates(model_info: ModelInfo, usage: Usage, custom_llm_provide
|
|||
crossed_input_key.replace("input_cost_per_token", "cache_read_input_token_cost", 1),
|
||||
"cache_read_input_token_cost_batches",
|
||||
),
|
||||
cache_creation=_batch_tier_rate(
|
||||
model_info,
|
||||
crossed_input_key.replace("input_cost_per_token", "cache_creation_input_token_cost", 1),
|
||||
"cache_creation_input_token_cost_batches",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -30154,6 +30154,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
|
||||
"cache_read_input_token_cost_batches": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 1e-06,
|
||||
"cache_creation_input_token_cost_batches": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": 1.25e-05,
|
||||
"cache_read_input_token_cost_flex": 5e-07,
|
||||
"cache_read_input_token_cost_priority": 2e-06,
|
||||
"input_cost_per_token": 1e-05,
|
||||
|
|
@ -30227,6 +30229,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
|
||||
"cache_read_input_token_cost_batches": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
|
||||
"cache_creation_input_token_cost_batches": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06,
|
||||
"cache_read_input_token_cost_flex": 2e-07,
|
||||
"cache_read_input_token_cost_priority": 8e-07,
|
||||
"input_cost_per_token": 4e-06,
|
||||
|
|
@ -30298,6 +30302,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
|
||||
"cache_read_input_token_cost_batches": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
|
||||
"cache_creation_input_token_cost_batches": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06,
|
||||
"cache_read_input_token_cost_flex": 2e-07,
|
||||
"cache_read_input_token_cost_priority": 8e-07,
|
||||
"input_cost_per_token": 4e-06,
|
||||
|
|
@ -30370,6 +30376,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-07,
|
||||
"cache_creation_input_token_cost_batches": 1.25e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-06,
|
||||
"cache_read_input_token_cost_flex": 1e-07,
|
||||
"cache_read_input_token_cost_priority": 4e-07,
|
||||
"input_cost_per_token": 2e-06,
|
||||
|
|
@ -30441,6 +30449,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
|
||||
"cache_read_input_token_cost_batches": 1e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-08,
|
||||
"cache_creation_input_token_cost_batches": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-07,
|
||||
"cache_read_input_token_cost_flex": 1e-08,
|
||||
"cache_read_input_token_cost_priority": 4e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
|
|
|
|||
|
|
@ -258,6 +258,8 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
cache_read_input_token_cost_above_512k_tokens: float | None
|
||||
cache_read_input_token_cost_batches: ReadOnly[float | None]
|
||||
cache_read_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None]
|
||||
cache_creation_input_token_cost_batches: ReadOnly[float | None]
|
||||
cache_creation_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None]
|
||||
# Smallest prefix this model will actually cache, whatever caching mechanism its provider uses.
|
||||
# Absent means the provider-agnostic default applies; see MINIMUM_PROMPT_CACHE_TOKEN_COUNT.
|
||||
prompt_cache_min_tokens: int | None
|
||||
|
|
@ -3521,6 +3523,8 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
cache_read_input_token_cost_above_272k_tokens_flex: float | None = None
|
||||
cache_read_input_token_cost_batches: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_batches: float | None = None
|
||||
cache_creation_input_token_cost_batches: float | None = None
|
||||
cache_creation_input_token_cost_above_272k_tokens_batches: float | None = None
|
||||
cache_read_input_audio_token_cost: float | None = None
|
||||
input_cost_per_character_above_128k_tokens: float | None = None
|
||||
input_cost_per_audio_token: float | None = None
|
||||
|
|
|
|||
|
|
@ -5822,6 +5822,10 @@ def _get_model_info_helper(
|
|||
cache_read_input_token_cost_above_272k_tokens_batches=_model_info.get(
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches"
|
||||
),
|
||||
cache_creation_input_token_cost_batches=_model_info.get("cache_creation_input_token_cost_batches"),
|
||||
cache_creation_input_token_cost_above_272k_tokens_batches=_model_info.get(
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches"
|
||||
),
|
||||
cache_creation_input_token_cost_above_1hr=_model_info.get(
|
||||
"cache_creation_input_token_cost_above_1hr", None
|
||||
),
|
||||
|
|
|
|||
|
|
@ -30154,6 +30154,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
|
||||
"cache_read_input_token_cost_batches": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 1e-06,
|
||||
"cache_creation_input_token_cost_batches": 6.25e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": 1.25e-05,
|
||||
"cache_read_input_token_cost_flex": 5e-07,
|
||||
"cache_read_input_token_cost_priority": 2e-06,
|
||||
"input_cost_per_token": 1e-05,
|
||||
|
|
@ -30227,6 +30229,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
|
||||
"cache_read_input_token_cost_batches": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
|
||||
"cache_creation_input_token_cost_batches": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06,
|
||||
"cache_read_input_token_cost_flex": 2e-07,
|
||||
"cache_read_input_token_cost_priority": 8e-07,
|
||||
"input_cost_per_token": 4e-06,
|
||||
|
|
@ -30298,6 +30302,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
|
||||
"cache_read_input_token_cost_batches": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
|
||||
"cache_creation_input_token_cost_batches": 2.5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": 5e-06,
|
||||
"cache_read_input_token_cost_flex": 2e-07,
|
||||
"cache_read_input_token_cost_priority": 8e-07,
|
||||
"input_cost_per_token": 4e-06,
|
||||
|
|
@ -30370,6 +30376,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-07,
|
||||
"cache_creation_input_token_cost_batches": 1.25e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-06,
|
||||
"cache_read_input_token_cost_flex": 1e-07,
|
||||
"cache_read_input_token_cost_priority": 4e-07,
|
||||
"input_cost_per_token": 2e-06,
|
||||
|
|
@ -30441,6 +30449,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
|
||||
"cache_read_input_token_cost_batches": 1e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-08,
|
||||
"cache_creation_input_token_cost_batches": 1.25e-07,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": 2.5e-07,
|
||||
"cache_read_input_token_cost_flex": 1e-08,
|
||||
"cache_read_input_token_cost_priority": 4e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
|
|
|
|||
|
|
@ -109,6 +109,11 @@
|
|||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"description": "Batch API rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"cache_creation_input_token_cost_above_272k_tokens_flex": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
@ -119,6 +124,11 @@
|
|||
"minimum": 0,
|
||||
"description": "Priority service-tier rate for the same-named base field."
|
||||
},
|
||||
"cache_creation_input_token_cost_batches": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"description": "USD per token written to the provider's prompt cache via its batch API."
|
||||
},
|
||||
"cache_creation_input_token_cost_flex": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
|
|||
|
|
@ -1777,7 +1777,7 @@ def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map):
|
|||
sol = litellm.model_cost["gpt-5.6-sol"]
|
||||
|
||||
cost_fields = sorted(field for field in sol if "cost" in field)
|
||||
assert len(cost_fields) == 31
|
||||
assert len(cost_fields) == 33
|
||||
|
||||
for field in cost_fields:
|
||||
assert alias.get(field) == sol.get(field), field
|
||||
|
|
@ -4803,12 +4803,14 @@ def test_get_batch_cost_rates_falls_back_to_the_flat_rates_when_tier_rates_are_u
|
|||
input_cost_per_token_above_272k_tokens_batches="two",
|
||||
output_cost_per_token_above_272k_tokens_batches="six",
|
||||
cache_read_input_token_cost_above_272k_tokens_batches="none",
|
||||
cache_creation_input_token_cost_batches=1.25e-7,
|
||||
cache_creation_input_token_cost_above_272k_tokens_batches="nope",
|
||||
),
|
||||
Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001),
|
||||
"openai",
|
||||
)
|
||||
|
||||
assert (rates.input, rates.output, rates.cache_read) == (1e-6, 4e-6, 1e-7)
|
||||
assert (rates.input, rates.output, rates.cache_read, rates.cache_creation) == (1e-6, 4e-6, 1e-7, 1.25e-7)
|
||||
|
||||
|
||||
def test_get_batch_cost_rates_has_no_cached_rate_without_a_cached_batch_key():
|
||||
|
|
@ -4826,3 +4828,38 @@ def test_get_batch_cost_rates_has_no_cached_rate_without_a_cached_batch_key():
|
|||
)
|
||||
|
||||
assert (rates.input, rates.output, rates.cache_read) == (2e-6, 4e-6, None)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(("prompt_tokens", "expected"), [(1_000, 1.25e-7), (272_000, 1.25e-7), (300_000, 2.5e-7)])
|
||||
def test_get_batch_cost_rates_reads_the_batch_cache_write_rate_for_the_crossed_tier(prompt_tokens, expected):
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
|
||||
|
||||
rates = get_batch_cost_rates(
|
||||
_batch_rates_model_info(
|
||||
input_cost_per_token_batches=1e-7,
|
||||
input_cost_per_token_above_272k_tokens_batches=2e-7,
|
||||
cache_creation_input_token_cost_batches=1.25e-7,
|
||||
cache_creation_input_token_cost_above_272k_tokens_batches=2.5e-7,
|
||||
),
|
||||
Usage(prompt_tokens=prompt_tokens, completion_tokens=1, total_tokens=prompt_tokens + 1),
|
||||
"openai",
|
||||
)
|
||||
|
||||
assert rates.cache_creation == expected
|
||||
|
||||
|
||||
def test_get_batch_cost_rates_has_no_cache_write_rate_without_a_cache_write_batch_key():
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
|
||||
|
||||
rates = get_batch_cost_rates(
|
||||
_batch_rates_model_info(
|
||||
input_cost_per_token_batches=1e-7,
|
||||
input_cost_per_token_above_272k_tokens_batches=2e-7,
|
||||
cache_creation_input_token_cost=2.5e-7,
|
||||
cache_creation_input_token_cost_above_272k_tokens=5e-7,
|
||||
),
|
||||
Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001),
|
||||
"openai",
|
||||
)
|
||||
|
||||
assert rates.cache_creation is None
|
||||
|
|
|
|||
|
|
@ -6277,3 +6277,78 @@ def test_passthrough_embeddings_result_swapped_for_callbacks():
|
|||
|
||||
assert isinstance(swapped_result, EmbeddingResponse)
|
||||
assert swapped_result.data[0]["embedding"] == [0.1, 0.2, 0.3]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def _local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
|
||||
def _luna_deployment_id(custom_pricing: dict[str, float]) -> str:
|
||||
from litellm import Router
|
||||
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "luna-batch",
|
||||
"litellm_params": {"model": "openai/gpt-5.6-luna", "api_key": "sk-test", **custom_pricing},
|
||||
}
|
||||
]
|
||||
)
|
||||
return router.model_list[0]["model_info"]["id"]
|
||||
|
||||
|
||||
def test_deployment_pricing_model_info_carries_every_published_input_batch_rate_when_only_output_is_declared(
|
||||
_local_model_cost_map,
|
||||
):
|
||||
from litellm.litellm_core_utils.litellm_logging import deployment_pricing_model_info
|
||||
|
||||
info = deployment_pricing_model_info(
|
||||
_luna_deployment_id({"output_cost_per_token_batches": 4e-6}), "openai/gpt-5.6-luna"
|
||||
)
|
||||
|
||||
assert info is not None
|
||||
assert info["input_cost_per_token_batches"] == 1e-7
|
||||
assert info["input_cost_per_token_above_272k_tokens_batches"] == 2e-7
|
||||
assert info["cache_read_input_token_cost_batches"] == 1e-8
|
||||
assert info["cache_read_input_token_cost_above_272k_tokens_batches"] == 2e-8
|
||||
assert info["cache_creation_input_token_cost_batches"] == 1.25e-7
|
||||
assert info["cache_creation_input_token_cost_above_272k_tokens_batches"] == 2.5e-7
|
||||
assert info["output_cost_per_token_batches"] == 4e-6
|
||||
assert info["output_cost_per_token_above_272k_tokens_batches"] is None
|
||||
|
||||
|
||||
def test_deployment_pricing_model_info_carries_the_published_output_batch_tier_when_only_input_is_declared(
|
||||
_local_model_cost_map,
|
||||
):
|
||||
from litellm.litellm_core_utils.litellm_logging import deployment_pricing_model_info
|
||||
|
||||
info = deployment_pricing_model_info(
|
||||
_luna_deployment_id({"input_cost_per_token_batches": 1e-6}), "openai/gpt-5.6-luna"
|
||||
)
|
||||
|
||||
assert info is not None
|
||||
assert info["output_cost_per_token_batches"] == 6e-7
|
||||
assert info["output_cost_per_token_above_272k_tokens_batches"] == 9e-7
|
||||
assert info["input_cost_per_token_batches"] == 1e-6
|
||||
assert info["input_cost_per_token_above_272k_tokens_batches"] is None
|
||||
assert info["cache_read_input_token_cost_batches"] is None
|
||||
assert info["cache_creation_input_token_cost_batches"] is None
|
||||
|
||||
|
||||
def test_deployment_pricing_model_info_honors_a_tier_only_batch_override_over_the_published_flat_rates(
|
||||
_local_model_cost_map: None,
|
||||
) -> None:
|
||||
from litellm.litellm_core_utils.litellm_logging import deployment_pricing_model_info
|
||||
|
||||
info = deployment_pricing_model_info(
|
||||
_luna_deployment_id({"input_cost_per_token_above_272k_tokens_batches": 1e-3}), "openai/gpt-5.6-luna"
|
||||
)
|
||||
|
||||
assert info is not None
|
||||
assert info["input_cost_per_token_above_272k_tokens_batches"] == 1e-3
|
||||
assert info["input_cost_per_token_batches"] == 1e-7
|
||||
assert info["cache_read_input_token_cost_batches"] == 1e-8
|
||||
assert info["output_cost_per_token_batches"] == 6e-7
|
||||
assert info["output_cost_per_token_above_272k_tokens_batches"] == 9e-7
|
||||
|
|
|
|||
|
|
@ -4673,3 +4673,80 @@ def test_openai_cached_batch_rates_are_half_the_standard_cached_rates(_local_mod
|
|||
entry["cache_read_input_token_cost_above_272k_tokens_batches"]
|
||||
== entry["cache_read_input_token_cost_above_272k_tokens"] / 2
|
||||
)
|
||||
|
||||
|
||||
def _cache_write_usage(prompt_tokens: int, cache_write_tokens: int, completion_tokens: int) -> Usage:
|
||||
return Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=0, cache_write_tokens=cache_write_tokens),
|
||||
)
|
||||
|
||||
|
||||
def test_batch_cost_calculator_bills_cache_write_tokens_at_the_long_context_batch_cache_write_rate(
|
||||
_local_model_cost_map,
|
||||
):
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
prompt_cost, completion_cost = batch_cost_calculator(
|
||||
usage=_cache_write_usage(300_048, 300_045, 4), model="gpt-5.6-luna", custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(3 * 2e-7 + 300_045 * 2.5e-7)
|
||||
assert completion_cost == pytest.approx(4 * 9e-7)
|
||||
|
||||
|
||||
def test_batch_cost_calculator_bills_cache_write_tokens_at_the_flat_batch_cache_write_rate_at_or_below_272k(
|
||||
_local_model_cost_map,
|
||||
):
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
prompt_cost, _ = batch_cost_calculator(
|
||||
usage=_cache_write_usage(1_000, 900, 4), model="gpt-5.6-luna", custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(100 * 1e-7 + 900 * 1.25e-7)
|
||||
|
||||
|
||||
def test_batch_cost_calculator_bills_cache_write_tokens_at_the_batch_input_rate_without_a_cache_write_batch_rate(
|
||||
_local_model_cost_map,
|
||||
):
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
prompt_cost, _ = batch_cost_calculator(
|
||||
usage=_cache_write_usage(300_048, 300_045, 4), model="gpt-5.5", custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(300_048 * 5e-6)
|
||||
|
||||
|
||||
def test_get_model_info_exposes_the_batch_cache_write_rates(_local_model_cost_map):
|
||||
info = litellm.get_model_info("gpt-5.6-luna", custom_llm_provider="openai")
|
||||
|
||||
assert info["cache_creation_input_token_cost_batches"] == 1.25e-7
|
||||
assert info["cache_creation_input_token_cost_above_272k_tokens_batches"] == 2.5e-7
|
||||
|
||||
|
||||
_OPENAI_ENTRIES_WITH_BATCH_CACHE_WRITE_RATES = frozenset(
|
||||
{"gpt-6-astra", "gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"}
|
||||
)
|
||||
|
||||
|
||||
def test_openai_batch_cache_write_rates_are_half_the_standard_cache_write_rates(_local_model_cost_map):
|
||||
carriers = {
|
||||
name: entry
|
||||
for name, entry in litellm.model_cost.items()
|
||||
if isinstance(entry, dict)
|
||||
and entry.get("litellm_provider") == "openai"
|
||||
and entry.get("cache_creation_input_token_cost") is not None
|
||||
and entry.get("input_cost_per_token_batches") is not None
|
||||
}
|
||||
|
||||
assert _OPENAI_ENTRIES_WITH_BATCH_CACHE_WRITE_RATES <= set(carriers)
|
||||
for entry in carriers.values():
|
||||
assert entry["cache_creation_input_token_cost_batches"] == entry["cache_creation_input_token_cost"] / 2
|
||||
assert (
|
||||
entry["cache_creation_input_token_cost_above_272k_tokens_batches"]
|
||||
== entry["cache_creation_input_token_cost_above_272k_tokens"] / 2
|
||||
)
|
||||
|
|
|
|||
|
|
@ -916,6 +916,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_creation_input_token_cost_above_272k_tokens_priority": {
|
||||
"type": "number"
|
||||
},
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": {"type": "number"},
|
||||
"cache_creation_input_token_cost_batches": {"type": "number"},
|
||||
"cache_creation_input_token_cost_flex": {"type": "number"},
|
||||
"cache_creation_input_token_cost_priority": {"type": "number"},
|
||||
"cache_read_input_token_cost": {"type": "number"},
|
||||
|
|
|
|||
8
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
8
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -29316,10 +29316,14 @@ export interface components {
|
|||
cache_creation_input_token_cost_above_200k_tokens?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens */
|
||||
cache_creation_input_token_cost_above_272k_tokens?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens Batches */
|
||||
cache_creation_input_token_cost_above_272k_tokens_batches?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens Flex */
|
||||
cache_creation_input_token_cost_above_272k_tokens_flex?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens Priority */
|
||||
cache_creation_input_token_cost_above_272k_tokens_priority?: number | null;
|
||||
/** Cache Creation Input Token Cost Batches */
|
||||
cache_creation_input_token_cost_batches?: number | null;
|
||||
/** Cache Creation Input Token Cost Flex */
|
||||
cache_creation_input_token_cost_flex?: number | null;
|
||||
/** Cache Creation Input Token Cost Priority */
|
||||
|
|
@ -39454,10 +39458,14 @@ export interface components {
|
|||
cache_creation_input_token_cost_above_200k_tokens?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens */
|
||||
cache_creation_input_token_cost_above_272k_tokens?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens Batches */
|
||||
cache_creation_input_token_cost_above_272k_tokens_batches?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens Flex */
|
||||
cache_creation_input_token_cost_above_272k_tokens_flex?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens Priority */
|
||||
cache_creation_input_token_cost_above_272k_tokens_priority?: number | null;
|
||||
/** Cache Creation Input Token Cost Batches */
|
||||
cache_creation_input_token_cost_batches?: number | null;
|
||||
/** Cache Creation Input Token Cost Flex */
|
||||
cache_creation_input_token_cost_flex?: number | null;
|
||||
/** Cache Creation Input Token Cost Priority */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue