fix(cost): stop double-billing overlapping cached and modality input tokens

generic_cost_per_token() billed cache_hit_tokens and image/audio/video
tokens as independent buckets, even though a cached image's tokens are
counted in both cached_tokens and image_tokens by some OpenAI-compatible
providers. When a modality has no distinct per-token rate, fold its
count into text_tokens before the cache-overlap detection instead of
billing it a second time on top of the cache-hit tokens that already
cover it.
This commit is contained in:
Anuj7411 2026-08-26 19:28:40 +05:30
parent e17988f4fe
commit b395bd8c8d
2 changed files with 91 additions and 0 deletions

View file

@ -818,6 +818,47 @@ def _resolve_reasoning_token_cost(
return standard_reasoning_cost if standard_reasoning_cost is not None else completion_base_cost
def _fold_undistinguished_input_modality_tokens(
prompt_tokens_details: PromptTokensDetailsResult,
model_info: ModelInfo,
) -> PromptTokensDetailsResult:
"""
Some providers report per-modality input token counts (image/audio/video) that are
not guaranteed to be disjoint from cache_hit_tokens -- e.g. a cached image's tokens
are counted in both ``cached_tokens`` and ``image_tokens``. A modality with no
distinct per-token rate bills at the same rate as plain text either way, so folding
its tokens into ``text_tokens`` up front removes the overlap entirely instead of
risking it being billed once as cache_hit_tokens and again as that modality's
tokens. Modalities with an explicit distinct rate are left untouched: resolving
their overlap with cache_hit_tokens needs a per-modality cache breakdown the
aggregate usage fields don't provide.
"""
text_tokens = prompt_tokens_details["text_tokens"]
image_tokens = prompt_tokens_details["image_tokens"]
audio_tokens = prompt_tokens_details["audio_tokens"]
video_tokens = prompt_tokens_details["video_tokens"]
if image_tokens and model_info.get("input_cost_per_image_token") is None:
text_tokens += image_tokens
image_tokens = 0
if audio_tokens and model_info.get("input_cost_per_audio_token") is None:
text_tokens += audio_tokens
audio_tokens = 0
if video_tokens and model_info.get("input_cost_per_video_token") is None:
text_tokens += video_tokens
video_tokens = 0
return PromptTokensDetailsResult(
**{
**prompt_tokens_details,
"text_tokens": text_tokens,
"image_tokens": image_tokens,
"audio_tokens": audio_tokens,
"video_tokens": video_tokens,
}
)
def generic_cost_per_token(
model: str,
usage: Usage,
@ -872,6 +913,7 @@ def generic_cost_per_token(
)
if usage.prompt_tokens_details:
prompt_tokens_details = parse_prompt_tokens_details(usage)
prompt_tokens_details = _fold_undistinguished_input_modality_tokens(prompt_tokens_details, model_info)
## EDGE CASE - text tokens not set or includes cached tokens (double-counting)
## Some providers (like xAI) report text_tokens = prompt_tokens (including cached)

View file

@ -270,6 +270,55 @@ def test_image_tokens_fallback_to_base_cost():
assert round(completion_cost, 12) == round(expected_completion_cost, 12)
def test_generic_cost_per_token_does_not_double_bill_overlapping_cached_image_tokens():
"""
Regression test for https://github.com/BerriAI/litellm/issues/37281.
Some OpenAI-compatible providers report `image_tokens` as a subset of
`prompt_tokens` that is not guaranteed to be disjoint from `cached_tokens` (a
cached image's tokens are counted in both). When the model has no distinct
`input_cost_per_image_token`, image tokens must bill at the same rate as text
either way, so they should never be billed a second time on top of the
cache-hit tokens that already cover them.
"""
from unittest.mock import patch
mock_model_info = {
"input_cost_per_token": 1e-6,
"cache_read_input_token_cost": 1e-7,
"output_cost_per_token": 2e-6,
# No input_cost_per_image_token defined - image tokens share the text rate.
}
usage = Usage(
prompt_tokens=100,
completion_tokens=10,
total_tokens=110,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=None,
cached_tokens=90,
image_tokens=80,
),
)
with patch(
"litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info",
return_value=mock_model_info,
):
prompt_cost, completion_cost = generic_cost_per_token(
model="test-model", usage=usage, custom_llm_provider="openai"
)
# 10 uncached tokens at the base rate + 90 cache-hit tokens at the cache rate.
# Before the fix, the 70 overlapping tokens were billed once via cache_hit_tokens
# and again via image_tokens, producing 89e-6 instead of 19e-6.
expected_prompt_cost = 10 * 1e-6 + 90 * 1e-7
expected_completion_cost = 10 * 2e-6
assert round(prompt_cost, 12) == round(expected_prompt_cost, 12)
assert round(completion_cost, 12) == round(expected_completion_cost, 12)
def test_video_output_tokens_gemini_omni_flash_preview():
"""Video output tokens are billed at output_cost_per_video_token, not the text rate and not zero."""
model = "gemini-omni-flash-preview"