diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 9ecd0df0cb8..9e1599fd0f6 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -2214,18 +2214,33 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): if "inference_geo" in _usage and _usage["inference_geo"] is not None: inference_geo = _usage["inference_geo"] - if ( - "cache_creation_input_tokens" in _usage - and _usage["cache_creation_input_tokens"] is not None - ): - cache_creation_input_tokens = _usage["cache_creation_input_tokens"] - prompt_tokens += cache_creation_input_tokens - if ( - "cache_read_input_tokens" in _usage - and _usage["cache_read_input_tokens"] is not None - ): - cache_read_input_tokens = _usage["cache_read_input_tokens"] - prompt_tokens += cache_read_input_tokens + iterations: Optional[List[Any]] = _usage.get("iterations") + if iterations: + prompt_tokens = sum(it.get("input_tokens", 0) or 0 for it in iterations) + completion_tokens = sum( + it.get("output_tokens", 0) or 0 for it in iterations + ) + cache_creation_input_tokens = sum( + it.get("cache_creation_input_tokens", 0) or 0 for it in iterations + ) + cache_read_input_tokens = sum( + it.get("cache_read_input_tokens", 0) or 0 for it in iterations + ) + prompt_tokens += cache_creation_input_tokens + cache_read_input_tokens + + if not iterations: + if ( + "cache_creation_input_tokens" in _usage + and _usage["cache_creation_input_tokens"] is not None + ): + cache_creation_input_tokens = _usage["cache_creation_input_tokens"] + prompt_tokens += cache_creation_input_tokens + if ( + "cache_read_input_tokens" in _usage + and _usage["cache_read_input_tokens"] is not None + ): + cache_read_input_tokens = _usage["cache_read_input_tokens"] + prompt_tokens += cache_read_input_tokens if "server_tool_use" in _usage and _usage["server_tool_use"] is not None: if ( "web_search_requests" in _usage["server_tool_use"] @@ -2264,7 +2279,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): ), ) - raw_input_tokens = usage_object.get("input_tokens", 0) or 0 + raw_input_tokens = prompt_tokens - cache_read_input_tokens - cache_creation_input_tokens prompt_tokens_details = PromptTokensDetailsWrapper( cached_tokens=cache_read_input_tokens, cache_creation_tokens=cache_creation_input_tokens, @@ -2296,6 +2311,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): cache_creation_input_tokens=cache_creation_input_tokens, cache_read_input_tokens=cache_read_input_tokens, completion_tokens_details=completion_token_details, + iterations=iterations, server_tool_use=( ServerToolUse( web_search_requests=web_search_requests, diff --git a/tests/test_anthropic_compaction_usage.py b/tests/test_anthropic_compaction_usage.py new file mode 100644 index 00000000000..a6158d2fc3b --- /dev/null +++ b/tests/test_anthropic_compaction_usage.py @@ -0,0 +1,95 @@ +from litellm.llms.anthropic.chat.transformation import AnthropicConfig + +def test_anthropic_compaction_usage_calculation(): + """ + Test that calculate_usage correctly sums tokens from the iterations array + as requested in Issue #27060. + """ + anthropic_config = AnthropicConfig() + + # Mock usage object with compaction iterations + usage_object = { + "input_tokens": 100, # Top-level (excludes compaction) + "output_tokens": 50, # Top-level (excludes compaction) + "iterations": [ + { + "iteration": 1, + "type": "compaction", + "input_tokens": 1000, + "output_tokens": 500 + }, + { + "iteration": 2, + "type": "message", + "input_tokens": 100, + "output_tokens": 50 + } + ] + } + + usage = anthropic_config.calculate_usage( + usage_object=usage_object, + reasoning_content=None + ) + + # Assertions + # Total prompt tokens should be 1000 + 100 = 1100 + assert usage.prompt_tokens == 1100 + # Total completion tokens should be 500 + 50 = 550 + assert usage.completion_tokens == 550 + # Total tokens should be 1650 + assert usage.total_tokens == 1650 + + # Assert details + assert usage.prompt_tokens_details.text_tokens == 1100 + + # Assert iterations passthrough + assert usage.iterations is not None + assert len(usage.iterations) == 2 + assert usage.iterations[0]["type"] == "compaction" + +def test_anthropic_compaction_usage_with_iteration_cache(): + """ + Test that calculate_usage correctly sums caching tokens FROM iterations. + This covers the specific case mentioned by JasonPan. + """ + anthropic_config = AnthropicConfig() + + usage_object = { + "input_tokens": 100, + "output_tokens": 50, + "iterations": [ + { + "type": "compaction", + "input_tokens": 500, + "output_tokens": 200, + "cache_creation_input_tokens": 50, + "cache_read_input_tokens": 17000 + }, + { + "type": "message", + "input_tokens": 100, + "output_tokens": 50, + "cache_creation_input_tokens": 10, + "cache_read_input_tokens": 20 + } + ] + } + + usage = anthropic_config.calculate_usage( + usage_object=usage_object, + reasoning_content=None + ) + + # input_tokens sum = 500 + 100 = 600 + # cache_creation sum = 50 + 10 = 60 + # cache_read sum = 17000 + 20 = 17020 + # Total prompt tokens = 600 + 60 + 17020 = 17680 + assert usage.prompt_tokens == 17680 + assert usage.completion_tokens == 250 + assert usage.prompt_tokens_details.cache_creation_tokens == 60 + assert usage.prompt_tokens_details.cached_tokens == 17020 + +if __name__ == "__main__": + test_anthropic_compaction_usage_calculation() + test_anthropic_compaction_usage_with_iteration_cache() diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py index 3a1d15ef79c..9e77c6ecc9b 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py @@ -359,6 +359,7 @@ ignored_keys = [ "metadata.additional_usage_values.cache_read_input_tokens", "metadata.additional_usage_values.inference_geo", "metadata.additional_usage_values.speed", + "metadata.additional_usage_values.iterations", "metadata.litellm_overhead_time_ms", "metadata.cost_breakdown", "metadata.user_api_key",