fix(savings): gate a zero cache-read rate on supports_prompt_caching

Review response. The previous commit tested both cache buckets for absence, which
was right for reads and wrong for writes, and it was wrong for reads on a model
that does not cache at all.

Reads now require an explicit rate AND supports_prompt_caching. Six cost-map
entries pair cache_read_input_token_cost of 0 with supports_prompt_caching false,
including gemini-robotics-er-1.5-preview and openrouter/z-ai/glm-4.7. On those the
zero is a placeholder for a model that has no cache, not a free one, so honouring
it priced 20,000 baseline tokens at $0.00 instead of $0.006.

Writes go back to truthiness. A zero cache-creation price does not mean free, it
means writes bill at the plain input rate, and 36 entries rely on that inheritance
including deepseek/deepseek-chat. Reading it as a real price dropped a 10,000 token
first-turn baseline from $0.0028 to $0.00.

The helper-only test is replaced with cases that go through _baseline_usage on real
cost-map entries, covering all three shapes: supported free reads stay in the read
bucket, unsupported reads move to text_tokens, and a zero write rate moves to
text_tokens. Both new tests fail if either leg is reverted.
This commit is contained in:
ayaangazali 2026-09-08 10:00:27 -07:00
parent ef716fdcc5
commit 080de70bd3
2 changed files with 88 additions and 17 deletions

View file

@ -213,15 +213,28 @@ def _baseline_cache_rate_keys(baseline_info: ModelInfo | None) -> tuple[bool, bo
profitable route into a reported loss. Such a model pays its plain input rate for
those tokens, so the buckets it cannot price become ordinary input below.
Absence is the test, not truthiness: an explicit rate of `0.0` is a real price,
and treating it as missing charges free cache reads at the full input rate.
The two buckets need different tests, because a `0.0` means something different in
each and the cost map proves it.
Reads: an explicit `0.0` is a real "cache reads are free" price, so absence rather
than truthiness is the right test. It only means that on a model that actually
caches, though. Six entries pair `cache_read_input_token_cost` of `0` with
`supports_prompt_caching` of `False`, `gemini-robotics-er-1.5-preview` and
`openrouter/z-ai/glm-4.7` among them, where the zero is a placeholder for a model
that has no cache rather than a free one. Honouring it priced 20,000 baseline tokens
at `$0.00` instead of `$0.006`.
Writes: truthiness stays. A `0.0` cache-creation price is not free, it means writes
bill at the plain input rate, and 36 entries rely on that inheritance including
`deepseek/deepseek-chat`. Reading it as a real price dropped a 10,000 token
first-turn baseline from `$0.0028` to `$0.00`.
"""
if baseline_info is None:
return True, True
return (
baseline_info.get("cache_read_input_token_cost") is not None,
baseline_info.get("cache_creation_input_token_cost") is not None,
prices_reads: Final = baseline_info.get("cache_read_input_token_cost") is not None and bool(
baseline_info.get("supports_prompt_caching")
)
return prices_reads, bool(baseline_info.get("cache_creation_input_token_cost"))
def _baseline_usage(usage: Usage, conversation_continuing: bool, baseline_info: ModelInfo | None = None) -> Usage:

View file

@ -1572,18 +1572,76 @@ def test_marks_gateway_injection_credits_only_the_deployment_that_was_injected()
assert marks_gateway_injection({"litellm_gateway_injected_cache": True}, "dep-a") is False
def test_explicit_zero_cache_rate_is_a_price_not_an_absence():
def _usage_with_cached(cached: int) -> Usage:
"""A continuing-conversation request whose whole prompt came from cache."""
return Usage(
prompt_tokens=cached,
completion_tokens=0,
total_tokens=cached,
prompt_tokens_details={"cached_tokens": cached, "cache_creation_tokens": 0, "text_tokens": 0},
cache_read_input_tokens=cached,
)
def _usage_with_written(written: int) -> Usage:
"""A first-turn request whose whole prompt was written to cache."""
return Usage(
prompt_tokens=written,
completion_tokens=0,
total_tokens=written,
prompt_tokens_details={"cached_tokens": 0, "cache_creation_tokens": written, "text_tokens": 0},
cache_creation_input_tokens=written,
)
def test_zero_cache_read_rate_is_free_only_when_the_model_actually_caches():
"""A `0.0` read rate is a real price on a caching model and a placeholder otherwise.
`tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8` prices cache reads at `0` and
sets `supports_prompt_caching`, so its reads really are free and stay in the read
bucket. `gemini/gemini-robotics-er-1.5-preview` carries the same `0` with
`supports_prompt_caching` false: it has no cache at all, so those tokens are its
plain input and must move to `text_tokens`. Reading the zero as a price either way
billed 20,000 baseline tokens at `$0.00` instead of `$0.006`.
"""
caches = litellm.get_model_info(model="tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8")
does_not = litellm.get_model_info(model="gemini/gemini-robotics-er-1.5-preview")
assert caches["cache_read_input_token_cost"] == 0 and caches["supports_prompt_caching"]
assert does_not["cache_read_input_token_cost"] == 0 and not does_not["supports_prompt_caching"]
free = _baseline_usage(_usage_with_cached(20_000), conversation_continuing=True, baseline_info=caches)
assert free.prompt_tokens_details.cached_tokens == 20_000
assert (free.prompt_tokens_details.text_tokens or 0) == 0
unpriced = _baseline_usage(_usage_with_cached(20_000), conversation_continuing=True, baseline_info=does_not)
assert unpriced.prompt_tokens_details.cached_tokens == 0
assert unpriced.prompt_tokens_details.text_tokens == 20_000
# 20,000 at the plain input rate is the $0.006 the buggy read dropped to $0.00.
assert 20_000 * (does_not["input_cost_per_token"] or 0.0) == pytest.approx(0.006)
def test_zero_cache_write_rate_still_inherits_input_pricing():
"""A `0.0` cache-creation price means writes bill at input, not that they are free.
36 cost-map entries rely on that inheritance, `deepseek/deepseek-chat` among them.
Treating the zero as a real price moved a 10,000 token first-turn baseline from
`$0.0028` to `$0.00`.
"""
info = litellm.get_model_info(model="deepseek/deepseek-chat")
assert info["cache_creation_input_token_cost"] == 0.0
baseline = _baseline_usage(_usage_with_written(10_000), conversation_continuing=False, baseline_info=info)
assert (baseline.prompt_tokens_details.cache_creation_tokens or 0) == 0
assert baseline.prompt_tokens_details.text_tokens == 10_000
assert 10_000 * (info["input_cost_per_token"] or 0.0) == pytest.approx(0.0028)
def test_a_missing_cache_rate_is_still_not_a_free_bucket():
"""The original guard: absent rates fall back to plain input, and `None` info prices both."""
from litellm.proxy.spend_tracking.savings import _baseline_cache_rate_keys
free = {
"cache_read_input_token_cost": 0.0,
"cache_creation_input_token_cost": 0.0,
}
priced = {
"cache_read_input_token_cost": 1e-07,
"cache_creation_input_token_cost": 2e-07,
}
assert _baseline_cache_rate_keys(free) == (True, True)
assert _baseline_cache_rate_keys(priced) == (True, True)
assert _baseline_cache_rate_keys({}) == (False, False)
assert _baseline_cache_rate_keys(None) == (True, True)
assert _baseline_cache_rate_keys(
{"cache_read_input_token_cost": 1e-07, "supports_prompt_caching": True, "cache_creation_input_token_cost": 2e-07}
) == (True, True)