mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(cost): bill cached batch tokens at OpenAI's cached batch rate
Adds cache_read_input_token_cost_batches and cache_read_input_token_cost_above_272k_tokens_batches for the tiered OpenAI entries at half the standard cached rate, bills cached batch tokens at that rate per output line, and parses string-valued batch rates in deployment-level model_info.
This commit is contained in:
parent
9819e4e0f8
commit
ab3c924174
13 changed files with 293 additions and 17 deletions
|
|
@ -169,6 +169,7 @@ COST_DESCRIPTIONS: dict[str, str] = {
|
|||
"cache_read_input_token_cost": "USD per prompt token served from the provider's prompt cache.",
|
||||
"input_cost_per_token_batches": "USD per prompt token via the provider's batch API.",
|
||||
"output_cost_per_token_batches": "USD per generated token via the provider's batch API.",
|
||||
"cache_read_input_token_cost_batches": "USD per cached prompt token via the provider's batch API.",
|
||||
}
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -2242,15 +2242,13 @@ def batch_cost_calculator(
|
|||
if not model_info:
|
||||
return 0.0, 0.0
|
||||
|
||||
input_cost_per_token_batches, output_cost_per_token_batches = get_batch_cost_rates(
|
||||
model_info, usage, custom_llm_provider
|
||||
)
|
||||
batch_rates: Final = get_batch_cost_rates(model_info, usage, custom_llm_provider)
|
||||
input_cost_per_token: Final = model_info.get("input_cost_per_token")
|
||||
output_cost_per_token: Final = model_info.get("output_cost_per_token")
|
||||
total_prompt_cost = 0.0
|
||||
total_completion_cost = 0.0
|
||||
if input_cost_per_token_batches is not None:
|
||||
total_prompt_cost = usage.prompt_tokens * input_cost_per_token_batches
|
||||
if batch_rates.input is not None:
|
||||
total_prompt_cost = _batch_prompt_cost(usage, batch_rates.input, batch_rates.cache_read)
|
||||
elif input_cost_per_token:
|
||||
details: Final = parse_prompt_tokens_details(usage)
|
||||
cache_read_tokens: Final = details["cache_hit_tokens"]
|
||||
|
|
@ -2269,8 +2267,8 @@ def batch_cost_calculator(
|
|||
|
||||
cache_creation_cost: Final = model_info.get("cache_creation_input_token_cost") or input_cost_per_token
|
||||
total_prompt_cost += cache_creation_tokens * cache_creation_cost / 2
|
||||
if output_cost_per_token_batches is not None:
|
||||
total_completion_cost = usage.completion_tokens * output_cost_per_token_batches
|
||||
if batch_rates.output is not None:
|
||||
total_completion_cost = usage.completion_tokens * batch_rates.output
|
||||
elif output_cost_per_token:
|
||||
total_completion_cost = (
|
||||
usage.completion_tokens * (output_cost_per_token) / 2
|
||||
|
|
@ -2284,6 +2282,13 @@ def batch_cost_calculator(
|
|||
return total_prompt_cost, total_completion_cost
|
||||
|
||||
|
||||
def _batch_prompt_cost(usage: Usage, input_rate: float, cache_read_rate: float | None) -> float:
|
||||
if cache_read_rate is None:
|
||||
return usage.prompt_tokens * input_rate
|
||||
cached_tokens: Final = parse_prompt_tokens_details(usage)["cache_hit_tokens"]
|
||||
return get_billable_input_tokens(usage) * input_rate + cached_tokens * cache_read_rate
|
||||
|
||||
|
||||
def _attribute_value(obj: object, name: str) -> object:
|
||||
return getattr(obj, name)
|
||||
|
||||
|
|
|
|||
|
|
@ -248,14 +248,31 @@ def _prompt_exceeds_threshold(prompt_tokens: int, threshold: float, inclusive: b
|
|||
return prompt_tokens > threshold or (inclusive and prompt_tokens == threshold)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class BatchCostRates:
|
||||
input: float | None
|
||||
output: float | None
|
||||
cache_read: float | None
|
||||
|
||||
|
||||
def _batch_rate(model_info: ModelInfo, key: str) -> float | None:
|
||||
value: Final = model_info.get(key)
|
||||
return value if isinstance(value, (int, float)) else None
|
||||
if isinstance(value, (int, float)):
|
||||
return float(value)
|
||||
if not isinstance(value, str):
|
||||
return None
|
||||
try:
|
||||
return float(value)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def get_batch_cost_rates(
|
||||
model_info: ModelInfo, usage: Usage, custom_llm_provider: str | None
|
||||
) -> tuple[float | None, float | None]:
|
||||
def _batch_tier_rate(model_info: ModelInfo, tier_key: str, flat_key: str) -> float | None:
|
||||
tier_rate: Final = _batch_rate(model_info, tier_key)
|
||||
return _batch_rate(model_info, flat_key) if tier_rate is None else tier_rate
|
||||
|
||||
|
||||
def get_batch_cost_rates(model_info: ModelInfo, usage: Usage, custom_llm_provider: str | None) -> BatchCostRates:
|
||||
inclusive: Final = _uses_inclusive_token_thresholds(custom_llm_provider)
|
||||
tier_input_keys: Final = tuple(
|
||||
key for key, value in model_info.items() if _BATCH_TIER_INPUT_KEY.match(key) and value is not None
|
||||
|
|
@ -268,12 +285,23 @@ def get_batch_cost_rates(
|
|||
),
|
||||
None,
|
||||
)
|
||||
flat_output_rate: Final = _batch_rate(model_info, "output_cost_per_token_batches")
|
||||
if crossed_input_key is None:
|
||||
return _batch_rate(model_info, "input_cost_per_token_batches"), flat_output_rate
|
||||
tier_input_rate: Final = _batch_rate(model_info, crossed_input_key)
|
||||
tier_output_rate: Final = _batch_rate(model_info, crossed_input_key.replace("input_", "output_", 1))
|
||||
return tier_input_rate, flat_output_rate if tier_output_rate is None else tier_output_rate
|
||||
return BatchCostRates(
|
||||
input=_batch_rate(model_info, "input_cost_per_token_batches"),
|
||||
output=_batch_rate(model_info, "output_cost_per_token_batches"),
|
||||
cache_read=_batch_rate(model_info, "cache_read_input_token_cost_batches"),
|
||||
)
|
||||
return BatchCostRates(
|
||||
input=_batch_tier_rate(model_info, crossed_input_key, "input_cost_per_token_batches"),
|
||||
output=_batch_tier_rate(
|
||||
model_info, crossed_input_key.replace("input_", "output_", 1), "output_cost_per_token_batches"
|
||||
),
|
||||
cache_read=_batch_tier_rate(
|
||||
model_info,
|
||||
crossed_input_key.replace("input_cost_per_token", "cache_read_input_token_cost", 1),
|
||||
"cache_read_input_token_cost_batches",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _select_priced_tier(model_info: ModelInfo, usage: Usage) -> dict | None:
|
||||
|
|
|
|||
|
|
@ -30152,6 +30152,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens": 2e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 1e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
|
||||
"cache_read_input_token_cost_batches": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 1e-06,
|
||||
"cache_read_input_token_cost_flex": 5e-07,
|
||||
"cache_read_input_token_cost_priority": 2e-06,
|
||||
"input_cost_per_token": 1e-05,
|
||||
|
|
@ -30223,6 +30225,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
|
||||
"cache_read_input_token_cost_batches": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
|
||||
"cache_read_input_token_cost_flex": 2e-07,
|
||||
"cache_read_input_token_cost_priority": 8e-07,
|
||||
"input_cost_per_token": 4e-06,
|
||||
|
|
@ -30292,6 +30296,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
|
||||
"cache_read_input_token_cost_batches": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
|
||||
"cache_read_input_token_cost_flex": 2e-07,
|
||||
"cache_read_input_token_cost_priority": 8e-07,
|
||||
"input_cost_per_token": 4e-06,
|
||||
|
|
@ -30362,6 +30368,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens": 4e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-07,
|
||||
"cache_read_input_token_cost_flex": 1e-07,
|
||||
"cache_read_input_token_cost_priority": 4e-07,
|
||||
"input_cost_per_token": 2e-06,
|
||||
|
|
@ -30431,6 +30439,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
|
||||
"cache_read_input_token_cost_batches": 1e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-08,
|
||||
"cache_read_input_token_cost_flex": 1e-08,
|
||||
"cache_read_input_token_cost_priority": 4e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
|
|
@ -30717,6 +30727,8 @@
|
|||
"gpt-5.5": {
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_batches": 2.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 5e-07,
|
||||
"cache_read_input_token_cost_flex": 2.5e-07,
|
||||
"cache_read_input_token_cost_priority": 1.25e-06,
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -30776,6 +30788,8 @@
|
|||
"gpt-5.5-2026-04-23": {
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_batches": 2.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 5e-07,
|
||||
"cache_read_input_token_cost_flex": 2.5e-07,
|
||||
"cache_read_input_token_cost_priority": 1.25e-06,
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -30939,6 +30953,8 @@
|
|||
"gpt-5.4": {
|
||||
"cache_read_input_token_cost": 2.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 5e-07,
|
||||
"cache_read_input_token_cost_batches": 1.25e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07,
|
||||
"cache_read_input_token_cost_flex": 1.3e-07,
|
||||
"cache_read_input_token_cost_priority": 5e-07,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
|
|
@ -30993,6 +31009,8 @@
|
|||
"gpt-5.4-2026-03-05": {
|
||||
"cache_read_input_token_cost": 2.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 5e-07,
|
||||
"cache_read_input_token_cost_batches": 1.25e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07,
|
||||
"cache_read_input_token_cost_flex": 1.3e-07,
|
||||
"cache_read_input_token_cost_priority": 5e-07,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
|
|
|
|||
|
|
@ -256,6 +256,8 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
cache_read_input_token_cost_above_272k_tokens_priority: float | None
|
||||
cache_read_input_token_cost_above_272k_tokens_flex: float | None
|
||||
cache_read_input_token_cost_above_512k_tokens: float | None
|
||||
cache_read_input_token_cost_batches: ReadOnly[float | None]
|
||||
cache_read_input_token_cost_above_272k_tokens_batches: ReadOnly[float | None]
|
||||
# Smallest prefix this model will actually cache, whatever caching mechanism its provider uses.
|
||||
# Absent means the provider-agnostic default applies; see MINIMUM_PROMPT_CACHE_TOKEN_COUNT.
|
||||
prompt_cache_min_tokens: int | None
|
||||
|
|
@ -3517,6 +3519,8 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
cache_read_input_token_cost_above_200k_tokens_priority: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_priority: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_flex: float | None = None
|
||||
cache_read_input_token_cost_batches: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_batches: float | None = None
|
||||
cache_read_input_audio_token_cost: float | None = None
|
||||
input_cost_per_character_above_128k_tokens: float | None = None
|
||||
input_cost_per_audio_token: float | None = None
|
||||
|
|
|
|||
|
|
@ -5818,6 +5818,10 @@ def _get_model_info_helper(
|
|||
cache_read_input_token_cost_flex=_model_info.get("cache_read_input_token_cost_flex", None),
|
||||
cache_read_input_token_cost_priority=_model_info.get("cache_read_input_token_cost_priority", None),
|
||||
cache_read_input_token_cost_ultrafast=_model_info.get("cache_read_input_token_cost_ultrafast", None),
|
||||
cache_read_input_token_cost_batches=_model_info.get("cache_read_input_token_cost_batches"),
|
||||
cache_read_input_token_cost_above_272k_tokens_batches=_model_info.get(
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches"
|
||||
),
|
||||
cache_creation_input_token_cost_above_1hr=_model_info.get(
|
||||
"cache_creation_input_token_cost_above_1hr", None
|
||||
),
|
||||
|
|
|
|||
|
|
@ -30152,6 +30152,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens": 2e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 1e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 4e-06,
|
||||
"cache_read_input_token_cost_batches": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 1e-06,
|
||||
"cache_read_input_token_cost_flex": 5e-07,
|
||||
"cache_read_input_token_cost_priority": 2e-06,
|
||||
"input_cost_per_token": 1e-05,
|
||||
|
|
@ -30223,6 +30225,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
|
||||
"cache_read_input_token_cost_batches": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
|
||||
"cache_read_input_token_cost_flex": 2e-07,
|
||||
"cache_read_input_token_cost_priority": 8e-07,
|
||||
"input_cost_per_token": 4e-06,
|
||||
|
|
@ -30292,6 +30296,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens": 8e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 4e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 1.6e-06,
|
||||
"cache_read_input_token_cost_batches": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 4e-07,
|
||||
"cache_read_input_token_cost_flex": 2e-07,
|
||||
"cache_read_input_token_cost_priority": 8e-07,
|
||||
"input_cost_per_token": 4e-06,
|
||||
|
|
@ -30362,6 +30368,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens": 4e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 2e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-07,
|
||||
"cache_read_input_token_cost_batches": 1e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-07,
|
||||
"cache_read_input_token_cost_flex": 1e-07,
|
||||
"cache_read_input_token_cost_priority": 4e-07,
|
||||
"input_cost_per_token": 2e-06,
|
||||
|
|
@ -30431,6 +30439,8 @@
|
|||
"cache_read_input_token_cost_above_272k_tokens": 4e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": 2e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": 8e-08,
|
||||
"cache_read_input_token_cost_batches": 1e-08,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2e-08,
|
||||
"cache_read_input_token_cost_flex": 1e-08,
|
||||
"cache_read_input_token_cost_priority": 4e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
|
|
@ -30717,6 +30727,8 @@
|
|||
"gpt-5.5": {
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_batches": 2.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 5e-07,
|
||||
"cache_read_input_token_cost_flex": 2.5e-07,
|
||||
"cache_read_input_token_cost_priority": 1.25e-06,
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -30776,6 +30788,8 @@
|
|||
"gpt-5.5-2026-04-23": {
|
||||
"cache_read_input_token_cost": 5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 1e-06,
|
||||
"cache_read_input_token_cost_batches": 2.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 5e-07,
|
||||
"cache_read_input_token_cost_flex": 2.5e-07,
|
||||
"cache_read_input_token_cost_priority": 1.25e-06,
|
||||
"input_cost_per_token": 5e-06,
|
||||
|
|
@ -30939,6 +30953,8 @@
|
|||
"gpt-5.4": {
|
||||
"cache_read_input_token_cost": 2.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 5e-07,
|
||||
"cache_read_input_token_cost_batches": 1.25e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07,
|
||||
"cache_read_input_token_cost_flex": 1.3e-07,
|
||||
"cache_read_input_token_cost_priority": 5e-07,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
|
|
@ -30993,6 +31009,8 @@
|
|||
"gpt-5.4-2026-03-05": {
|
||||
"cache_read_input_token_cost": 2.5e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 5e-07,
|
||||
"cache_read_input_token_cost_batches": 1.25e-07,
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": 2.5e-07,
|
||||
"cache_read_input_token_cost_flex": 1.3e-07,
|
||||
"cache_read_input_token_cost_priority": 5e-07,
|
||||
"input_cost_per_token": 2.5e-06,
|
||||
|
|
|
|||
|
|
@ -163,6 +163,11 @@
|
|||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"description": "Batch API rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
@ -178,6 +183,11 @@
|
|||
"minimum": 0,
|
||||
"description": "Rate applied once the prompt exceeds the token threshold in the field name."
|
||||
},
|
||||
"cache_read_input_token_cost_batches": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
"description": "USD per cached prompt token via the provider's batch API."
|
||||
},
|
||||
"cache_read_input_token_cost_flex": {
|
||||
"type": "number",
|
||||
"minimum": 0,
|
||||
|
|
|
|||
|
|
@ -1745,3 +1745,42 @@ def test_unparsable_bedrock_batch_usage_warns(caplog):
|
|||
assert usage.total_tokens == 0
|
||||
assert "does not understand" in caplog.text
|
||||
assert "inputTextTokenCount" in caplog.text
|
||||
|
||||
|
||||
def test_total_cost_bills_cached_tokens_per_line_at_the_batch_cached_rate():
|
||||
responses_row = _success_row(
|
||||
usage={
|
||||
"input_tokens": 300_000,
|
||||
"output_tokens": 10,
|
||||
"total_tokens": 300_010,
|
||||
"input_tokens_details": {"cached_tokens": 299_000},
|
||||
}
|
||||
)
|
||||
chat_row = _success_row(usage={**_usage(100, 10), "prompt_tokens_details": {"cached_tokens": 60}})
|
||||
|
||||
result = bu._aggregate_batch_cost_usage_models(
|
||||
entries=[responses_row, chat_row],
|
||||
custom_llm_provider="openai",
|
||||
model_info=ModelInfo(
|
||||
key="lit-batch-cached-tier",
|
||||
max_tokens=None,
|
||||
max_input_tokens=None,
|
||||
max_output_tokens=None,
|
||||
input_cost_per_token=2e-6,
|
||||
output_cost_per_token=8e-6,
|
||||
cache_read_input_token_cost=1e-6,
|
||||
litellm_provider="openai",
|
||||
mode="chat",
|
||||
supported_openai_params=None,
|
||||
input_cost_per_token_batches=1e-6,
|
||||
output_cost_per_token_batches=4e-6,
|
||||
cache_read_input_token_cost_batches=5e-7,
|
||||
input_cost_per_token_above_272k_tokens_batches=2e-6,
|
||||
output_cost_per_token_above_272k_tokens_batches=6e-6,
|
||||
cache_read_input_token_cost_above_272k_tokens_batches=1e-6,
|
||||
),
|
||||
)
|
||||
|
||||
long_line = 1_000 * 2e-6 + 299_000 * 1e-6 + 10 * 6e-6
|
||||
short_line = 40 * 1e-6 + 60 * 5e-7 + 10 * 4e-6
|
||||
assert result.cost == pytest.approx(long_line + short_line)
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import json
|
||||
from typing import cast
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
|
@ -1776,7 +1777,7 @@ def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map):
|
|||
sol = litellm.model_cost["gpt-5.6-sol"]
|
||||
|
||||
cost_fields = sorted(field for field in sol if "cost" in field)
|
||||
assert len(cost_fields) == 29
|
||||
assert len(cost_fields) == 31
|
||||
|
||||
for field in cost_fields:
|
||||
assert alias.get(field) == sol.get(field), field
|
||||
|
|
@ -4766,3 +4767,62 @@ def test_route_image_generation_cost_falls_back_to_requested_size(monkeypatch, r
|
|||
)
|
||||
|
||||
assert cost == expected_cost
|
||||
|
||||
|
||||
def _batch_rates_model_info(**rates: object) -> ModelInfo:
|
||||
return cast(ModelInfo, dict(rates))
|
||||
|
||||
|
||||
def test_get_batch_cost_rates_parses_string_rates():
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
|
||||
|
||||
rates = get_batch_cost_rates(
|
||||
_batch_rates_model_info(
|
||||
input_cost_per_token_batches="1e-06",
|
||||
output_cost_per_token_batches="4e-06",
|
||||
cache_read_input_token_cost_batches="1e-07",
|
||||
input_cost_per_token_above_272k_tokens_batches="2e-06",
|
||||
output_cost_per_token_above_272k_tokens_batches="6e-06",
|
||||
cache_read_input_token_cost_above_272k_tokens_batches="2e-07",
|
||||
),
|
||||
Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001),
|
||||
"openai",
|
||||
)
|
||||
|
||||
assert (rates.input, rates.output, rates.cache_read) == (2e-6, 6e-6, 2e-7)
|
||||
|
||||
|
||||
def test_get_batch_cost_rates_falls_back_to_the_flat_rates_when_tier_rates_are_unparsable():
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
|
||||
|
||||
rates = get_batch_cost_rates(
|
||||
_batch_rates_model_info(
|
||||
input_cost_per_token_batches=1e-6,
|
||||
output_cost_per_token_batches=4e-6,
|
||||
cache_read_input_token_cost_batches=1e-7,
|
||||
input_cost_per_token_above_272k_tokens_batches="two",
|
||||
output_cost_per_token_above_272k_tokens_batches="six",
|
||||
cache_read_input_token_cost_above_272k_tokens_batches="none",
|
||||
),
|
||||
Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001),
|
||||
"openai",
|
||||
)
|
||||
|
||||
assert (rates.input, rates.output, rates.cache_read) == (1e-6, 4e-6, 1e-7)
|
||||
|
||||
|
||||
def test_get_batch_cost_rates_has_no_cached_rate_without_a_cached_batch_key():
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import get_batch_cost_rates
|
||||
|
||||
rates = get_batch_cost_rates(
|
||||
_batch_rates_model_info(
|
||||
input_cost_per_token_batches=1e-6,
|
||||
output_cost_per_token_batches=4e-6,
|
||||
cache_read_input_token_cost=2e-7,
|
||||
input_cost_per_token_above_272k_tokens_batches=2e-6,
|
||||
),
|
||||
Usage(prompt_tokens=300_000, completion_tokens=1, total_tokens=300_001),
|
||||
"openai",
|
||||
)
|
||||
|
||||
assert (rates.input, rates.output, rates.cache_read) == (2e-6, 4e-6, None)
|
||||
|
|
|
|||
|
|
@ -4523,6 +4523,8 @@ def test_get_model_info_exposes_the_long_context_batch_tier(_local_model_cost_ma
|
|||
|
||||
assert info["input_cost_per_token_above_272k_tokens_batches"] == 2.5e-6
|
||||
assert info["output_cost_per_token_above_272k_tokens_batches"] == 1.125e-5
|
||||
assert info["cache_read_input_token_cost_batches"] == 1.25e-7
|
||||
assert info["cache_read_input_token_cost_above_272k_tokens_batches"] == 2.5e-7
|
||||
|
||||
|
||||
def test_regular_path_never_bills_the_batch_tier_keys(_local_model_cost_map, monkeypatch):
|
||||
|
|
@ -4594,3 +4596,80 @@ def test_batch_cost_calculator_ignores_malformed_batch_tier_keys():
|
|||
|
||||
assert prompt_cost == pytest.approx(300_035 * 2e-6)
|
||||
assert completion_cost == pytest.approx(64 * 6e-6)
|
||||
|
||||
|
||||
def _cached_usage(prompt_tokens: int, cached_tokens: int, completion_tokens: int) -> Usage:
|
||||
return Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=cached_tokens),
|
||||
)
|
||||
|
||||
|
||||
def test_batch_cost_calculator_bills_cached_tokens_at_the_long_context_batch_cached_rate(_local_model_cost_map):
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
prompt_cost, completion_cost = batch_cost_calculator(
|
||||
usage=_cached_usage(300_048, 300_045, 11), model="gpt-5.6-luna", custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(3 * 2e-7 + 300_045 * 2e-8)
|
||||
assert completion_cost == pytest.approx(11 * 9e-7)
|
||||
|
||||
|
||||
def test_batch_cost_calculator_bills_cached_tokens_at_the_flat_batch_cached_rate_at_or_below_272k(
|
||||
_local_model_cost_map,
|
||||
):
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
prompt_cost, _ = batch_cost_calculator(
|
||||
usage=_cached_usage(1_000, 900, 4), model="gpt-5.6-luna", custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(100 * 1e-7 + 900 * 1e-8)
|
||||
|
||||
|
||||
def test_batch_cost_calculator_bills_cached_tokens_at_the_batch_input_rate_without_a_cached_batch_rate(
|
||||
_local_model_cost_map,
|
||||
):
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
prompt_cost, _ = batch_cost_calculator(
|
||||
usage=_cached_usage(300_048, 300_045, 11), model="gpt-5.5-pro", custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(300_048 * 3e-5)
|
||||
|
||||
|
||||
_OPENAI_ENTRIES_WITH_CACHED_BATCH_RATES = frozenset(
|
||||
{
|
||||
"gpt-6-astra",
|
||||
"gpt-5.6",
|
||||
"gpt-5.6-sol",
|
||||
"gpt-5.6-terra",
|
||||
"gpt-5.6-luna",
|
||||
"gpt-5.5",
|
||||
"gpt-5.5-2026-04-23",
|
||||
"gpt-5.4",
|
||||
"gpt-5.4-2026-03-05",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def test_openai_cached_batch_rates_are_half_the_standard_cached_rates(_local_model_cost_map):
|
||||
carriers = {
|
||||
name: entry
|
||||
for name, entry in litellm.model_cost.items()
|
||||
if isinstance(entry, dict)
|
||||
and entry.get("litellm_provider") == "openai"
|
||||
and entry.get("cache_read_input_token_cost_batches") is not None
|
||||
}
|
||||
|
||||
assert _OPENAI_ENTRIES_WITH_CACHED_BATCH_RATES <= set(carriers)
|
||||
for entry in carriers.values():
|
||||
assert entry["cache_read_input_token_cost_batches"] == entry["cache_read_input_token_cost"] / 2
|
||||
assert (
|
||||
entry["cache_read_input_token_cost_above_272k_tokens_batches"]
|
||||
== entry["cache_read_input_token_cost_above_272k_tokens"] / 2
|
||||
)
|
||||
|
|
|
|||
|
|
@ -927,6 +927,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"type": "number"
|
||||
},
|
||||
"cache_read_input_token_cost_above_512k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_batches": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": {
|
||||
"type": "number"
|
||||
},
|
||||
|
|
|
|||
8
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
8
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -29336,12 +29336,16 @@ export interface components {
|
|||
cache_read_input_token_cost_above_200k_tokens_priority?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens */
|
||||
cache_read_input_token_cost_above_272k_tokens?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens Batches */
|
||||
cache_read_input_token_cost_above_272k_tokens_batches?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens Flex */
|
||||
cache_read_input_token_cost_above_272k_tokens_flex?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens Priority */
|
||||
cache_read_input_token_cost_above_272k_tokens_priority?: number | null;
|
||||
/** Cache Read Input Token Cost Above 512K Tokens */
|
||||
cache_read_input_token_cost_above_512k_tokens?: number | null;
|
||||
/** Cache Read Input Token Cost Batches */
|
||||
cache_read_input_token_cost_batches?: number | null;
|
||||
/** Cache Read Input Token Cost Flex */
|
||||
cache_read_input_token_cost_flex?: number | null;
|
||||
/** Cache Read Input Token Cost Priority */
|
||||
|
|
@ -39470,12 +39474,16 @@ export interface components {
|
|||
cache_read_input_token_cost_above_200k_tokens_priority?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens */
|
||||
cache_read_input_token_cost_above_272k_tokens?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens Batches */
|
||||
cache_read_input_token_cost_above_272k_tokens_batches?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens Flex */
|
||||
cache_read_input_token_cost_above_272k_tokens_flex?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens Priority */
|
||||
cache_read_input_token_cost_above_272k_tokens_priority?: number | null;
|
||||
/** Cache Read Input Token Cost Above 512K Tokens */
|
||||
cache_read_input_token_cost_above_512k_tokens?: number | null;
|
||||
/** Cache Read Input Token Cost Batches */
|
||||
cache_read_input_token_cost_batches?: number | null;
|
||||
/** Cache Read Input Token Cost Flex */
|
||||
cache_read_input_token_cost_flex?: number | null;
|
||||
/** Cache Read Input Token Cost Priority */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue