mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
fix(savings): an explicit zero cache rate is a price, not a missing one
_baseline_cache_rate_keys tested the rate with bool(), so an explicit cache_read_input_token_cost of 0.0 read as absent. The cache-read tokens then fell through to ordinary text input and were charged the full input rate, which overstates savings for deployments where reads are genuinely free. Tests for absence instead. A rate that is present and zero is now honoured, a rate that is missing still falls back to plain input, which is what the existing docstring describes and what OpenAI, Azure and Gemini entries rely on for cache writes. explicit free (0.0) before (False, False) after (True, True) absent before (False, False) after (False, False)
This commit is contained in:
parent
1af7a403c6
commit
ef716fdcc5
2 changed files with 23 additions and 2 deletions
|
|
@ -212,11 +212,15 @@ def _baseline_cache_rate_keys(baseline_info: ModelInfo | None) -> tuple[bool, bo
|
|||
Gemini entry for cache writes, would carry the whole prompt for nothing and turn a
|
||||
profitable route into a reported loss. Such a model pays its plain input rate for
|
||||
those tokens, so the buckets it cannot price become ordinary input below.
|
||||
|
||||
Absence is the test, not truthiness: an explicit rate of `0.0` is a real price,
|
||||
and treating it as missing charges free cache reads at the full input rate.
|
||||
"""
|
||||
if baseline_info is None:
|
||||
return True, True
|
||||
return bool(baseline_info.get("cache_read_input_token_cost")), bool(
|
||||
baseline_info.get("cache_creation_input_token_cost")
|
||||
return (
|
||||
baseline_info.get("cache_read_input_token_cost") is not None,
|
||||
baseline_info.get("cache_creation_input_token_cost") is not None,
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1570,3 +1570,20 @@ def test_marks_gateway_injection_credits_only_the_deployment_that_was_injected()
|
|||
assert marks_gateway_injection({"litellm_gateway_injected_cache": ""}, None) is True
|
||||
assert marks_gateway_injection({"litellm_call_id": "c1"}, "dep-a") is False
|
||||
assert marks_gateway_injection({"litellm_gateway_injected_cache": True}, "dep-a") is False
|
||||
|
||||
|
||||
def test_explicit_zero_cache_rate_is_a_price_not_an_absence():
|
||||
from litellm.proxy.spend_tracking.savings import _baseline_cache_rate_keys
|
||||
|
||||
free = {
|
||||
"cache_read_input_token_cost": 0.0,
|
||||
"cache_creation_input_token_cost": 0.0,
|
||||
}
|
||||
priced = {
|
||||
"cache_read_input_token_cost": 1e-07,
|
||||
"cache_creation_input_token_cost": 2e-07,
|
||||
}
|
||||
|
||||
assert _baseline_cache_rate_keys(free) == (True, True)
|
||||
assert _baseline_cache_rate_keys(priced) == (True, True)
|
||||
assert _baseline_cache_rate_keys({}) == (False, False)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue