From ef716fdcc5e545360277dcb892fe3865c875cc57 Mon Sep 17 00:00:00 2001 From: ayaangazali Date: Sat, 29 Aug 2026 15:52:25 -0700 Subject: [PATCH] fix(savings): an explicit zero cache rate is a price, not a missing one _baseline_cache_rate_keys tested the rate with bool(), so an explicit cache_read_input_token_cost of 0.0 read as absent. The cache-read tokens then fell through to ordinary text input and were charged the full input rate, which overstates savings for deployments where reads are genuinely free. Tests for absence instead. A rate that is present and zero is now honoured, a rate that is missing still falls back to plain input, which is what the existing docstring describes and what OpenAI, Azure and Gemini entries rely on for cache writes. explicit free (0.0) before (False, False) after (True, True) absent before (False, False) after (False, False) --- litellm/proxy/spend_tracking/savings.py | 8 ++++++-- .../proxy/spend_tracking/test_savings.py | 17 +++++++++++++++++ 2 files changed, 23 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/spend_tracking/savings.py b/litellm/proxy/spend_tracking/savings.py index 950fcca2039..620a21d40f6 100644 --- a/litellm/proxy/spend_tracking/savings.py +++ b/litellm/proxy/spend_tracking/savings.py @@ -212,11 +212,15 @@ def _baseline_cache_rate_keys(baseline_info: ModelInfo | None) -> tuple[bool, bo Gemini entry for cache writes, would carry the whole prompt for nothing and turn a profitable route into a reported loss. Such a model pays its plain input rate for those tokens, so the buckets it cannot price become ordinary input below. + + Absence is the test, not truthiness: an explicit rate of `0.0` is a real price, + and treating it as missing charges free cache reads at the full input rate. """ if baseline_info is None: return True, True - return bool(baseline_info.get("cache_read_input_token_cost")), bool( - baseline_info.get("cache_creation_input_token_cost") + return ( + baseline_info.get("cache_read_input_token_cost") is not None, + baseline_info.get("cache_creation_input_token_cost") is not None, ) diff --git a/tests/test_litellm/proxy/spend_tracking/test_savings.py b/tests/test_litellm/proxy/spend_tracking/test_savings.py index e466edab131..760caded635 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_savings.py +++ b/tests/test_litellm/proxy/spend_tracking/test_savings.py @@ -1570,3 +1570,20 @@ def test_marks_gateway_injection_credits_only_the_deployment_that_was_injected() assert marks_gateway_injection({"litellm_gateway_injected_cache": ""}, None) is True assert marks_gateway_injection({"litellm_call_id": "c1"}, "dep-a") is False assert marks_gateway_injection({"litellm_gateway_injected_cache": True}, "dep-a") is False + + +def test_explicit_zero_cache_rate_is_a_price_not_an_absence(): + from litellm.proxy.spend_tracking.savings import _baseline_cache_rate_keys + + free = { + "cache_read_input_token_cost": 0.0, + "cache_creation_input_token_cost": 0.0, + } + priced = { + "cache_read_input_token_cost": 1e-07, + "cache_creation_input_token_cost": 2e-07, + } + + assert _baseline_cache_rate_keys(free) == (True, True) + assert _baseline_cache_rate_keys(priced) == (True, True) + assert _baseline_cache_rate_keys({}) == (False, False)