From b0edaebfd75d9d9e9f5b7eb19228e09f438dc7f0 Mon Sep 17 00:00:00 2001 From: Praveen11558 <44603409+Praveen11558@users.noreply.github.com> Date: Thu, 12 Mar 2026 16:01:35 +0530 Subject: [PATCH] Update utils.py --- .../litellm_core_utils/llm_cost_calc/utils.py | 33 ++++++++++++++++--- 1 file changed, 28 insertions(+), 5 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index a0c52445b4f..63f090b22b7 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -673,32 +673,45 @@ def generic_cost_per_token( # noqa: PLR0915 or prompt_tokens_details["character_count"] > 0 or prompt_tokens_details["video_length_seconds"] > 0 ) + should_fill_unaccounted_tokens = ( + accounted_tokens < usage.prompt_tokens + and not (accounted_tokens == 0 and has_alternative_billing) + ) if has_double_counting: # Double-counting fix (xAI etc.): recalculate text_tokens from scratch - text_tokens = ( + # Clamp to 0 to prevent negative cost when cache_hit exceeds prompt_tokens + text_tokens = max( + 0, usage.prompt_tokens - cache_hit - audio_tokens - cache_creation - - image_tokens + - image_tokens, ) prompt_tokens_details["text_tokens"] = text_tokens elif (text_tokens == 0 and not has_alternative_billing): # text_tokens not set by provider and no alternative billing dimensions: # calculate text_tokens as the remainder of prompt_tokens - text_tokens = ( + text_tokens = max( + 0, usage.prompt_tokens - cache_hit - audio_tokens - cache_creation - - image_tokens + - image_tokens, ) prompt_tokens_details["text_tokens"] = text_tokens - elif accounted_tokens < usage.prompt_tokens and not has_alternative_billing: + elif should_fill_unaccounted_tokens: # Unaccounted tokens fix: inline documents (PDF, DOCX, etc.) are counted # in prompt_tokens by the provider but not broken out into any detail field. # Add the gap to text_tokens so they are costed at input_cost_per_token. + # + # The guard above only skips when there is ZERO token-level accounting + # AND alternative billing is active (e.g. Bedrock Nova image-only + # embeddings where prompt_tokens are entirely explained by image_count). + # Mixed requests (PDF + image_count) where the provider reports + # text_tokens > 0 still have their gap filled correctly. unaccounted_tokens = usage.prompt_tokens - accounted_tokens prompt_tokens_details["text_tokens"] += unaccounted_tokens @@ -713,6 +726,16 @@ def generic_cost_per_token( # noqa: PLR0915 model_info=model_info, usage=usage, service_tier=service_tier ) + prompt_cost = _calculate_input_cost( + prompt_tokens_details=prompt_tokens_details, + model_info=model_info, + prompt_base_cost=prompt_base_cost, + cache_read_cost=cache_read_cost, + cache_creation_cost=cache_creation_cost, + cache_creation_cost_above_1hr=cache_creation_cost_above_1hr, + service_tier=service_tier, + ) + prompt_cost = _calculate_input_cost( prompt_tokens_details=prompt_tokens_details, model_info=model_info,