From dd3bcd34213ed7a3731be251e1ee9403781123c6 Mon Sep 17 00:00:00 2001 From: Praveen11558 <44603409+Praveen11558@users.noreply.github.com> Date: Thu, 5 Mar 2026 12:28:06 +0530 Subject: [PATCH] token_count calculation during inline data sent during multimodal inputs --- litellm/litellm_core_utils/llm_cost_calc/utils.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index e7c09f6ec0c..5809b8c4d8c 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -663,7 +663,7 @@ def generic_cost_per_token( # noqa: PLR0915 accounted_tokens = text_tokens + cache_hit + audio_tokens + cache_creation + image_tokens has_double_counting = cache_hit > 0 and accounted_tokens > usage.prompt_tokens # Double-counting fix (xAI etc.): recalculate text_tokens from scratch - if has_double_counting: + if (text_tokens == 0 and prompt_tokens_details["image_count"] == 0) or has_double_counting: text_tokens = ( usage.prompt_tokens - cache_hit