From 4a024957ecf6826d35dc7416f8ddb9dcdd6a86f9 Mon Sep 17 00:00:00 2001 From: Monesh Ram <31161039+WhoisMonesh@users.noreply.github.com> Date: Tue, 24 Feb 2026 09:37:38 +0530 Subject: [PATCH] fix: only recalculate total_tokens when cache tokens present --- litellm/llms/bedrock/chat/converse_transformation.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 8d9a9ef5aa3..abfa98afebf 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -1644,9 +1644,10 @@ class AmazonConverseConfig(BaseConfig): if "cacheWriteInputTokens" in usage: cache_creation_input_tokens = usage["cacheWriteInputTokens"] input_tokens += cache_creation_input_tokens - # Recalculate total_tokens to include cache tokens added to input_tokens - # The API's totalTokens doesn't include cache hits/writes, but prompt_tokens does - total_tokens = input_tokens + output_tokens + # Only recalculate total_tokens if we added cache tokens to input_tokens + # Otherwise, trust the API's totalTokens (which may include other token categories) + if cache_read_input_tokens > 0 or cache_creation_input_tokens > 0: + total_tokens = input_tokens + output_tokens prompt_tokens_details = PromptTokensDetailsWrapper( cached_tokens=cache_read_input_tokens )