diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index e99f356f8f2..8620d237a68 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -1,6 +1,7 @@ import json import re import time +from collections.abc import Mapping, Sequence from typing import ( TYPE_CHECKING, Any, @@ -139,6 +140,28 @@ _ANTHROPIC_TOOL_NAME_MAX_LEN = 128 ANTHROPIC_TOOL_NAME_REVERSE_MAP_KEY = "_anthropic_tool_name_map" +def _reported_thinking_tokens(usage_object: Mapping[str, Any]) -> Optional[int]: + """Read the provider-reported reasoning token count from an Anthropic usage object. + + Anthropic (and Bedrock's Anthropic-compatible surface) reports extended-thinking + tokens as ``usage.output_tokens_details.thinking_tokens``; it is authoritative and + is present even when the thinking text is empty or redacted. + """ + details = usage_object.get("output_tokens_details") + if not isinstance(details, Mapping): + return None + thinking_tokens = details.get("thinking_tokens") + if isinstance(thinking_tokens, bool) or not isinstance(thinking_tokens, int): + return None + return thinking_tokens + + +def _sum_reported_thinking_tokens(iterations: Sequence[Any]) -> Optional[int]: + reported = tuple(_reported_thinking_tokens(iteration) for iteration in iterations if isinstance(iteration, Mapping)) + present = tuple(tokens for tokens in reported if tokens is not None) + return sum(present) if present else None + + def _basic_sanitize_anthropic_tool_name(name: str) -> str: """Lossy: replace [^a-zA-Z0-9_-] with '_' and truncate to 128. @@ -2201,10 +2224,15 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): text_tokens=raw_input_tokens, ) # Always populate completion_token_details, not just when there's reasoning_content - estimated_reasoning_tokens = ( - token_counter(text=reasoning_content, count_response_tokens=True) if reasoning_content else 0 + reported_thinking_tokens = ( + _sum_reported_thinking_tokens(iterations) if iterations else _reported_thinking_tokens(_usage) ) - reasoning_tokens = min(estimated_reasoning_tokens, completion_tokens) + resolved_reasoning_tokens = ( + reported_thinking_tokens + if reported_thinking_tokens is not None + else (token_counter(text=reasoning_content, count_response_tokens=True) if reasoning_content else 0) + ) + reasoning_tokens = min(resolved_reasoning_tokens, completion_tokens) completion_token_details = CompletionTokensDetailsWrapper( reasoning_tokens=reasoning_tokens if reasoning_tokens > 0 else 0, text_tokens=(completion_tokens - reasoning_tokens if reasoning_tokens > 0 else completion_tokens), diff --git a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py index 50e90699194..edb124d7cd7 100644 --- a/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py +++ b/litellm/proxy/pass_through_endpoints/llm_provider_handlers/anthropic_passthrough_logging_handler.py @@ -669,6 +669,7 @@ class AnthropicPassthroughLoggingHandler: tool_search_requests: Optional[int] = None inference_geo: Optional[str] = None stop_reason: Optional[str] = None + thinking_tokens: Optional[int] = None found_usage = False resolved_model = model for _chunk_str in all_chunks: @@ -693,6 +694,9 @@ class AnthropicPassthroughLoggingHandler: inference_geo = usage.get("inference_geo") if usage.get("output_tokens") is not None: output_tokens = usage.get("output_tokens") + _otd = usage.get("output_tokens_details") + if isinstance(_otd, dict) and _otd.get("thinking_tokens") is not None: + thinking_tokens = _otd.get("thinking_tokens") found_usage = True elif event_type == "message_delta": _delta_stop = (data.get("delta") or {}).get("stop_reason") @@ -711,6 +715,9 @@ class AnthropicPassthroughLoggingHandler: cache_read = usage.get("cache_read_input_tokens") if usage.get("inference_geo") is not None: inference_geo = usage.get("inference_geo") + _otd = usage.get("output_tokens_details") + if isinstance(_otd, dict) and _otd.get("thinking_tokens") is not None: + thinking_tokens = _otd.get("thinking_tokens") found_usage = True if not found_usage: return None @@ -742,6 +749,8 @@ class AnthropicPassthroughLoggingHandler: usage_object["server_tool_use"] = _server_tool_use if inference_geo is not None: usage_object["inference_geo"] = inference_geo + if thinking_tokens is not None: + usage_object["output_tokens_details"] = {"thinking_tokens": thinking_tokens} usage_obj = AnthropicConfig().calculate_usage(usage_object=usage_object, reasoning_content=None) return ModelResponse( model=resolved_model, diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py index 94a4a3fc945..8f50b3a1af1 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -119,6 +119,97 @@ def test_calculate_usage_clamps_text_tokens_when_reasoning_estimate_exceeds_outp assert usage.completion_tokens_details.text_tokens == 0 +def test_calculate_usage_prefers_provider_reported_thinking_tokens(): + """ + Anthropic / Bedrock report extended-thinking usage as + ``output_tokens_details.thinking_tokens``; it must win over the + token_counter estimate of the thinking text. + """ + config = AnthropicConfig() + + usage = config.calculate_usage( + usage_object={ + "input_tokens": 70, + "output_tokens": 548, + "output_tokens_details": {"thinking_tokens": 275}, + }, + reasoning_content="short thinking text", + ) + + assert usage.completion_tokens_details is not None + assert usage.completion_tokens_details.reasoning_tokens == 275 + assert usage.completion_tokens_details.text_tokens == 548 - 275 + + +def test_calculate_usage_reports_thinking_tokens_when_thinking_text_is_empty(): + """ + Bedrock returns thinking blocks whose text is empty (redacted / truncated), + which used to log reasoning_tokens=0 even though the model reasoned. + """ + config = AnthropicConfig() + + usage = config.calculate_usage( + usage_object={ + "input_tokens": 78, + "output_tokens": 64, + "output_tokens_details": {"thinking_tokens": 64}, + }, + reasoning_content="", + ) + + assert usage.completion_tokens_details is not None + assert usage.completion_tokens_details.reasoning_tokens == 64 + assert usage.completion_tokens_details.text_tokens == 0 + + +def test_calculate_usage_falls_back_to_estimate_without_thinking_tokens(): + config = AnthropicConfig() + + usage = config.calculate_usage( + usage_object={"input_tokens": 10, "output_tokens": 500}, + reasoning_content="some reasoning text that tokenizes to a few tokens", + ) + + assert usage.completion_tokens_details is not None + assert usage.completion_tokens_details.reasoning_tokens > 0 + + +def test_calculate_usage_ignores_non_int_thinking_tokens(): + config = AnthropicConfig() + + usage = config.calculate_usage( + usage_object={ + "input_tokens": 10, + "output_tokens": 500, + "output_tokens_details": {"thinking_tokens": None}, + }, + reasoning_content=None, + ) + + assert usage.completion_tokens_details is not None + assert usage.completion_tokens_details.reasoning_tokens == 0 + assert usage.completion_tokens_details.text_tokens == 500 + + +def test_calculate_usage_sums_thinking_tokens_across_iterations(): + config = AnthropicConfig() + + usage = config.calculate_usage( + usage_object={ + "iterations": [ + {"input_tokens": 10, "output_tokens": 100, "output_tokens_details": {"thinking_tokens": 40}}, + {"input_tokens": 10, "output_tokens": 200, "output_tokens_details": {"thinking_tokens": 60}}, + ] + }, + reasoning_content=None, + ) + + assert usage.completion_tokens == 300 + assert usage.completion_tokens_details is not None + assert usage.completion_tokens_details.reasoning_tokens == 100 + assert usage.completion_tokens_details.text_tokens == 200 + + def test_calculate_usage_handles_mocked_output_tokens_with_reasoning_content(): config = AnthropicConfig() diff --git a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py index 947a7a64beb..ac413305299 100644 --- a/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py +++ b/tests/test_litellm/proxy/pass_through_endpoints/llm_provider_handlers/test_anthropic_passthrough_logging_handler.py @@ -1055,6 +1055,54 @@ class TestBuildCompleteStreamingResponseRobustness: assert result.choices[0].message.content == "The stream ends with [DONE]" +class TestStreamingThinkingTokens: + """ + A streamed thinking response must be logged with the reasoning token count + Anthropic/Bedrock reported in message_delta, not a token_counter estimate of + the thinking text (which is 0 when the thinking text is empty/redacted). + """ + + @staticmethod + def _chunks(thinking_text: str) -> List[str]: + return [ + 'event: message_start\ndata: {"type":"message_start","message":{"id":"msg_1","type":"message","role":"assistant","content":[],"model":"claude-sonnet-4-5-20250929","stop_reason":null,"stop_sequence":null,"usage":{"input_tokens":70,"output_tokens":1}}}', + 'event: content_block_start\ndata: {"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":"","signature":""}}', + "event: content_block_delta\ndata: " + + json.dumps( + { + "type": "content_block_delta", + "index": 0, + "delta": {"type": "thinking_delta", "thinking": thinking_text}, + } + ), + 'event: content_block_stop\ndata: {"type":"content_block_stop","index":0}', + 'event: content_block_start\ndata: {"type":"content_block_start","index":1,"content_block":{"type":"text","text":""}}', + 'event: content_block_delta\ndata: {"type":"content_block_delta","index":1,"delta":{"type":"text_delta","text":"33 liters"}}', + 'event: content_block_stop\ndata: {"type":"content_block_stop","index":1}', + 'event: message_delta\ndata: {"type":"message_delta","delta":{"stop_reason":"end_turn","stop_sequence":null},"usage":{"input_tokens":70,"output_tokens":548,"output_tokens_details":{"thinking_tokens":275}}}', + 'event: message_stop\ndata: {"type":"message_stop"}', + ] + + def _build(self, thinking_text: str): + logging_obj = MagicMock() + logging_obj.model_call_details = {} + return AnthropicPassthroughLoggingHandler._build_complete_streaming_response( + all_chunks=self._chunks(thinking_text), + litellm_logging_obj=logging_obj, + model="claude-sonnet-4-5-20250929", + ) + + def test_reported_thinking_tokens_used_for_streamed_thinking_text(self): + result = self._build("Let me work through the fuel calculation step by step.") + assert result is not None + assert result.usage.completion_tokens_details.reasoning_tokens == 275 + + def test_reported_thinking_tokens_used_when_thinking_text_is_empty(self): + result = self._build("") + assert result is not None + assert result.usage.completion_tokens_details.reasoning_tokens == 275 + + class TestPureTextFastPathParity: """ The pure-text fast path in _build_complete_streaming_response must produce @@ -1939,6 +1987,36 @@ class TestAnthropicUsageOnlyFallback: assert usage.prompt_tokens_details.cached_tokens == 40 assert usage.server_tool_use.web_search_requests == 2 + def test_build_usage_only_recovers_provider_reported_thinking_tokens(self): + chunks = [ + _sse_bytes( + { + "type": "message_start", + "message": { + "model": "claude-sonnet-4-5-20250929", + "usage": {"input_tokens": 70, "output_tokens": 1}, + }, + } + ), + _sse_bytes( + { + "type": "message_delta", + "usage": { + "output_tokens": 548, + "output_tokens_details": {"thinking_tokens": 275}, + }, + } + ), + ] + response = ( + AnthropicPassthroughLoggingHandler._build_usage_only_response_from_chunks( + all_chunks=chunks, model="claude-sonnet-4-5-20250929" + ) + ) + assert response is not None + assert response.usage.completion_tokens_details.reasoning_tokens == 275 + assert response.usage.completion_tokens_details.text_tokens == 273 + def test_build_usage_only_returns_none_without_usage_events(self): chunks = [_sse_bytes({"type": "content_block_delta", "delta": {"text": "hi"}})] assert (