From e76b25816fc14ddfcada478a7e4823d09dcf8460 Mon Sep 17 00:00:00 2001 From: Tanvir Alam Date: Wed, 23 Sep 2026 00:36:27 -0400 Subject: [PATCH] fix(cost): bill undetailed cache-write tokens at the 5m rate Streamed Anthropic server-tool requests keep the message_start 5m/1h breakdown while cache_creation_input_tokens grows across iterations. calculate_cache_writing_cost was pricing only the stale breakdown and dropping the remainder. Bill max(total - 5m - 1h, 0) at the 5m rate, matching the non-streaming aggregate path. Fixes #42663 --- litellm/litellm_core_utils/llm_cost_calc/utils.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index 46bf2ec2960..eb500187462 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -891,6 +891,14 @@ def calculate_cache_writing_cost( ) -> float: """ Adjust cost of cache creation tokens based on the cache creation token details. + + When a 5m/1h breakdown is present but covers fewer tokens than + ``cache_creation_tokens`` (common on streamed Anthropic server-tool + requests, where the stream merger keeps the ``message_start`` breakdown + while the top-level total grows across iterations), bill the undetailed + remainder at the default 5m rate — same rule + ``AnthropicConfig._aggregate_cache_creation_token_details`` uses for + non-streaming ``usage.iterations``. """ total_cost: float = 0.0 if cache_creation_token_details is not None: @@ -902,6 +910,9 @@ def calculate_cache_writing_cost( total_cost += ( cache_creation_tokens_1h * cache_creation_cost_above_1hr if cache_creation_tokens_1h is not None else 0.0 ) + detailed: Final = (cache_creation_tokens_5m or 0) + (cache_creation_tokens_1h or 0) + undetailed: Final = max(cache_creation_tokens - detailed, 0) + total_cost += undetailed * cache_creation_cost else: total_cost += cache_creation_tokens * cache_creation_cost return total_cost