From 144fc2035836a25b4d442535dec26925e581b3a7 Mon Sep 17 00:00:00 2001 From: Praveen11558 <44603409+Praveen11558@users.noreply.github.com> Date: Mon, 23 Mar 2026 23:30:22 +0530 Subject: [PATCH] Refactor token accounting logic in utils.py --- .../litellm_core_utils/llm_cost_calc/utils.py | 19 ++++++++++++------- 1 file changed, 12 insertions(+), 7 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index 486621ddb7b..ac90f8902e2 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -668,9 +668,11 @@ def generic_cost_per_token( # noqa: PLR0915 image_tokens = prompt_tokens_details["image_tokens"] # Check for double-counting: sum of details > prompt_tokens means overlap - accounted_tokens = text_tokens + cache_hit + audio_tokens + cache_creation + image_tokens + accounted_tokens = ( + text_tokens + cache_hit + audio_tokens + cache_creation + image_tokens + ) has_double_counting = cache_hit > 0 and accounted_tokens > usage.prompt_tokens - + # Some models use alternative billing dimensions (character_count, video_length_seconds) # that account for prompt_tokens without being included in the per-token detail fields. # When these are active, a gap between accounted_tokens and prompt_tokens is expected @@ -679,7 +681,7 @@ def generic_cost_per_token( # noqa: PLR0915 prompt_tokens_details["character_count"] > 0 or prompt_tokens_details["video_length_seconds"] > 0 ) - + if has_double_counting: # Double-counting fix (xAI etc.): recalculate text_tokens from scratch # Clamp to 0 to prevent negative cost when cache_hit exceeds prompt_tokens @@ -692,7 +694,11 @@ def generic_cost_per_token( # noqa: PLR0915 - image_tokens, ) prompt_tokens_details["text_tokens"] = text_tokens - elif accounted_tokens < usage.prompt_tokens and not has_non_token_alternative_billing and prompt_tokens_details["image_count"] == 0: + elif ( + accounted_tokens < usage.prompt_tokens + and not has_non_token_alternative_billing + and prompt_tokens_details["image_count"] == 0 + ): # Unaccounted tokens fix: inline documents (PDF, DOCX, etc.) are counted # in prompt_tokens by the provider but not broken out into any detail field. # Add the gap to text_tokens so they are costed at input_cost_per_token. @@ -707,8 +713,7 @@ def generic_cost_per_token( # noqa: PLR0915 # billed separately via input_cost_per_image and shouldn't be double-charged unaccounted_tokens = usage.prompt_tokens - accounted_tokens prompt_tokens_details["text_tokens"] += unaccounted_tokens - - + ( prompt_base_cost, completion_base_cost, @@ -718,7 +723,7 @@ def generic_cost_per_token( # noqa: PLR0915 ) = _get_token_base_cost( model_info=model_info, usage=usage, service_tier=service_tier ) - + prompt_cost = _calculate_input_cost( prompt_tokens_details=prompt_tokens_details, model_info=model_info,