From e76b25816fc14ddfcada478a7e4823d09dcf8460 Mon Sep 17 00:00:00 2001 From: Tanvir Alam Date: Wed, 23 Sep 2026 00:36:27 -0400 Subject: [PATCH 1/2] fix(cost): bill undetailed cache-write tokens at the 5m rate Streamed Anthropic server-tool requests keep the message_start 5m/1h breakdown while cache_creation_input_tokens grows across iterations. calculate_cache_writing_cost was pricing only the stale breakdown and dropping the remainder. Bill max(total - 5m - 1h, 0) at the 5m rate, matching the non-streaming aggregate path. Fixes #42663 --- litellm/litellm_core_utils/llm_cost_calc/utils.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index 46bf2ec2960..eb500187462 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -891,6 +891,14 @@ def calculate_cache_writing_cost( ) -> float: """ Adjust cost of cache creation tokens based on the cache creation token details. + + When a 5m/1h breakdown is present but covers fewer tokens than + ``cache_creation_tokens`` (common on streamed Anthropic server-tool + requests, where the stream merger keeps the ``message_start`` breakdown + while the top-level total grows across iterations), bill the undetailed + remainder at the default 5m rate — same rule + ``AnthropicConfig._aggregate_cache_creation_token_details`` uses for + non-streaming ``usage.iterations``. """ total_cost: float = 0.0 if cache_creation_token_details is not None: @@ -902,6 +910,9 @@ def calculate_cache_writing_cost( total_cost += ( cache_creation_tokens_1h * cache_creation_cost_above_1hr if cache_creation_tokens_1h is not None else 0.0 ) + detailed: Final = (cache_creation_tokens_5m or 0) + (cache_creation_tokens_1h or 0) + undetailed: Final = max(cache_creation_tokens - detailed, 0) + total_cost += undetailed * cache_creation_cost else: total_cost += cache_creation_tokens * cache_creation_cost return total_cost From 3b239626c01791d37979558ee5f79dc95028f5bd Mon Sep 17 00:00:00 2001 From: Tanvir Alam Date: Wed, 23 Sep 2026 00:36:55 -0400 Subject: [PATCH 2/2] test(cost): cover undetailed cache-write remainder billing Regression for #42663: stale 5m/1h breakdown with a larger total, already-reconciled details (no double-bill), and mixed 5m+1h remainder. --- .../test_calculate_cache_writing_cost.py | 66 +++++++++++++++++++ 1 file changed, 66 insertions(+) create mode 100644 tests/test_litellm/litellm_core_utils/test_calculate_cache_writing_cost.py diff --git a/tests/test_litellm/litellm_core_utils/test_calculate_cache_writing_cost.py b/tests/test_litellm/litellm_core_utils/test_calculate_cache_writing_cost.py new file mode 100644 index 00000000000..266b6ce795d --- /dev/null +++ b/tests/test_litellm/litellm_core_utils/test_calculate_cache_writing_cost.py @@ -0,0 +1,66 @@ +"""Regression tests for calculate_cache_writing_cost undetailed remainder (#42663).""" + +import pytest + +from litellm.litellm_core_utils.llm_cost_calc.utils import calculate_cache_writing_cost +from litellm.types.utils import CacheCreationTokenDetails + + +def test_calculate_cache_writing_cost_bills_undetailed_remainder_at_5m_rate(): + """ + Streamed Anthropic server-tool runs keep the message_start 5m/1h breakdown + while cache_creation_input_tokens grows across iterations. Pricing must + still bill the undetailed remainder at the 5m rate instead of dropping it. + """ + rate_5m = 3.75e-6 + rate_1h = 6.0e-6 + details = CacheCreationTokenDetails( + ephemeral_5m_input_tokens=31490, + ephemeral_1h_input_tokens=0, + ) + + # Stale breakdown (message_start only) vs full streamed total. + cost = calculate_cache_writing_cost( + cache_creation_tokens=184457, + cache_creation_token_details=details, + cache_creation_cost_above_1hr=rate_1h, + cache_creation_cost=rate_5m, + ) + assert cost == pytest.approx(184457 * rate_5m) + + # Already-reconciled details (non-streaming aggregate path) must not double-bill. + reconciled = CacheCreationTokenDetails( + ephemeral_5m_input_tokens=184457, + ephemeral_1h_input_tokens=0, + ) + cost_reconciled = calculate_cache_writing_cost( + cache_creation_tokens=184457, + cache_creation_token_details=reconciled, + cache_creation_cost_above_1hr=rate_1h, + cache_creation_cost=rate_5m, + ) + assert cost_reconciled == pytest.approx(184457 * rate_5m) + + # Mixed 5m + 1h with an undetailed remainder. + mixed = CacheCreationTokenDetails( + ephemeral_5m_input_tokens=1000, + ephemeral_1h_input_tokens=2000, + ) + cost_mixed = calculate_cache_writing_cost( + cache_creation_tokens=5000, + cache_creation_token_details=mixed, + cache_creation_cost_above_1hr=rate_1h, + cache_creation_cost=rate_5m, + ) + assert cost_mixed == pytest.approx(1000 * rate_5m + 2000 * rate_1h + 2000 * rate_5m) + + +def test_calculate_cache_writing_cost_without_details_uses_total(): + rate_5m = 3.75e-6 + cost = calculate_cache_writing_cost( + cache_creation_tokens=1000, + cache_creation_token_details=None, + cache_creation_cost_above_1hr=6.0e-6, + cache_creation_cost=rate_5m, + ) + assert cost == pytest.approx(1000 * rate_5m)