fix(savings): an explicit zero cache rate is a price, not a missing one

_baseline_cache_rate_keys tested the rate with bool(), so an explicit
cache_read_input_token_cost of 0.0 read as absent. The cache-read tokens then
fell through to ordinary text input and were charged the full input rate, which
overstates savings for deployments where reads are genuinely free.

Tests for absence instead. A rate that is present and zero is now honoured, a
rate that is missing still falls back to plain input, which is what the existing
docstring describes and what OpenAI, Azure and Gemini entries rely on for cache
writes.

  explicit free (0.0)  before (False, False)  after (True, True)
  absent               before (False, False)  after (False, False)
This commit is contained in:
ayaangazali 2026-08-29 15:52:25 -07:00
parent 1af7a403c6
commit ef716fdcc5
2 changed files with 23 additions and 2 deletions

View file

@ -212,11 +212,15 @@ def _baseline_cache_rate_keys(baseline_info: ModelInfo | None) -> tuple[bool, bo
Gemini entry for cache writes, would carry the whole prompt for nothing and turn a
profitable route into a reported loss. Such a model pays its plain input rate for
those tokens, so the buckets it cannot price become ordinary input below.
Absence is the test, not truthiness: an explicit rate of `0.0` is a real price,
and treating it as missing charges free cache reads at the full input rate.
"""
if baseline_info is None:
return True, True
return bool(baseline_info.get("cache_read_input_token_cost")), bool(
baseline_info.get("cache_creation_input_token_cost")
return (
baseline_info.get("cache_read_input_token_cost") is not None,
baseline_info.get("cache_creation_input_token_cost") is not None,
)

View file

@ -1570,3 +1570,20 @@ def test_marks_gateway_injection_credits_only_the_deployment_that_was_injected()
assert marks_gateway_injection({"litellm_gateway_injected_cache": ""}, None) is True
assert marks_gateway_injection({"litellm_call_id": "c1"}, "dep-a") is False
assert marks_gateway_injection({"litellm_gateway_injected_cache": True}, "dep-a") is False
def test_explicit_zero_cache_rate_is_a_price_not_an_absence():
from litellm.proxy.spend_tracking.savings import _baseline_cache_rate_keys
free = {
"cache_read_input_token_cost": 0.0,
"cache_creation_input_token_cost": 0.0,
}
priced = {
"cache_read_input_token_cost": 1e-07,
"cache_creation_input_token_cost": 2e-07,
}
assert _baseline_cache_rate_keys(free) == (True, True)
assert _baseline_cache_rate_keys(priced) == (True, True)
assert _baseline_cache_rate_keys({}) == (False, False)