From 7e98843d97c3b72af087d629b59bd1eb2902065e Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Thu, 8 Jan 2026 10:57:07 +0530 Subject: [PATCH] Fix: Incomplete usage in response object passed --- litellm/llms/anthropic/chat/transformation.py | 15 ++--- .../test_anthropic_chat_transformation.py | 58 +++++++++++++++++++ 2 files changed, 66 insertions(+), 7 deletions(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index c71edcdc2d1..57391c152cb 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1265,14 +1265,15 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): cache_creation_tokens=cache_creation_input_tokens, cache_creation_token_details=cache_creation_token_details, ) - completion_token_details = ( - CompletionTokensDetailsWrapper( - reasoning_tokens=token_counter( - text=reasoning_content, count_response_tokens=True - ) - ) + # Always populate completion_token_details, not just when there's reasoning_content + reasoning_tokens = ( + token_counter(text=reasoning_content, count_response_tokens=True) if reasoning_content - else None + else 0 + ) + completion_token_details = CompletionTokensDetailsWrapper( + reasoning_tokens=reasoning_tokens if reasoning_tokens > 0 else None, + text_tokens=completion_tokens - reasoning_tokens if reasoning_tokens > 0 else completion_tokens, ) total_tokens = prompt_tokens + completion_tokens diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py index 9b6d1c6e178..5e58601d589 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -1754,3 +1754,61 @@ def test_transform_request_respects_user_max_tokens(): ) assert result["max_tokens"] == 1000 + + +def test_calculate_usage_completion_tokens_details_always_populated(): + """ + Test that completion_tokens_details is always populated in Usage object, + not just when there's reasoning_content. + + Fixes: https://github.com/BerriAI/litellm/issues/18772 + Bug: completion_tokens_details was None for regular Claude responses without reasoning + """ + config = AnthropicConfig() + + # Test without reasoning_content - completion_tokens_details should still be populated + usage_object = { + "input_tokens": 37, + "output_tokens": 248, + } + usage = config.calculate_usage(usage_object=usage_object, reasoning_content=None) + + # completion_tokens_details should NOT be None + assert usage.completion_tokens_details is not None + assert usage.completion_tokens_details.reasoning_tokens is None + assert usage.completion_tokens_details.text_tokens == 248 + assert usage.completion_tokens == 248 + assert usage.prompt_tokens == 37 + assert usage.total_tokens == 285 + + +def test_calculate_usage_completion_tokens_details_with_reasoning(): + """ + Test that completion_tokens_details correctly splits text_tokens and reasoning_tokens + when reasoning_content is present. + + Fixes: https://github.com/BerriAI/litellm/issues/18772 + """ + config = AnthropicConfig() + + # Test with reasoning_content - should split tokens correctly + usage_object = { + "input_tokens": 100, + "output_tokens": 500, + } + # Simulating reasoning content that would count as ~50 tokens + reasoning_content = "Let me think about this step by step. " * 10 # Roughly 50 tokens + + usage = config.calculate_usage( + usage_object=usage_object, + reasoning_content=reasoning_content + ) + + # completion_tokens_details should be populated with both reasoning and text tokens + assert usage.completion_tokens_details is not None + assert usage.completion_tokens_details.reasoning_tokens is not None + assert usage.completion_tokens_details.reasoning_tokens > 0 + # text_tokens should be total minus reasoning + expected_text_tokens = 500 - usage.completion_tokens_details.reasoning_tokens + assert usage.completion_tokens_details.text_tokens == expected_text_tokens + assert usage.completion_tokens == 500