fix(anthropic): log provider-reported thinking tokens as reasoning_tokens

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
milan 2026-07-31 03:55:56 +00:00
parent 81ff7cb38f
commit 7a2735e028
4 changed files with 209 additions and 3 deletions

View file

@ -1,6 +1,7 @@
import json
import re
import time
from collections.abc import Mapping, Sequence
from typing import (
TYPE_CHECKING,
Any,
@ -139,6 +140,28 @@ _ANTHROPIC_TOOL_NAME_MAX_LEN = 128
ANTHROPIC_TOOL_NAME_REVERSE_MAP_KEY = "_anthropic_tool_name_map"
def _reported_thinking_tokens(usage_object: Mapping[str, Any]) -> Optional[int]:
"""Read the provider-reported reasoning token count from an Anthropic usage object.
Anthropic (and Bedrock's Anthropic-compatible surface) reports extended-thinking
tokens as ``usage.output_tokens_details.thinking_tokens``; it is authoritative and
is present even when the thinking text is empty or redacted.
"""
details = usage_object.get("output_tokens_details")
if not isinstance(details, Mapping):
return None
thinking_tokens = details.get("thinking_tokens")
if isinstance(thinking_tokens, bool) or not isinstance(thinking_tokens, int):
return None
return thinking_tokens
def _sum_reported_thinking_tokens(iterations: Sequence[Any]) -> Optional[int]:
reported = tuple(_reported_thinking_tokens(iteration) for iteration in iterations if isinstance(iteration, Mapping))
present = tuple(tokens for tokens in reported if tokens is not None)
return sum(present) if present else None
def _basic_sanitize_anthropic_tool_name(name: str) -> str:
"""Lossy: replace [^a-zA-Z0-9_-] with '_' and truncate to 128.
@ -2201,10 +2224,15 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
text_tokens=raw_input_tokens,
)
# Always populate completion_token_details, not just when there's reasoning_content
estimated_reasoning_tokens = (
token_counter(text=reasoning_content, count_response_tokens=True) if reasoning_content else 0
reported_thinking_tokens = (
_sum_reported_thinking_tokens(iterations) if iterations else _reported_thinking_tokens(_usage)
)
reasoning_tokens = min(estimated_reasoning_tokens, completion_tokens)
resolved_reasoning_tokens = (
reported_thinking_tokens
if reported_thinking_tokens is not None
else (token_counter(text=reasoning_content, count_response_tokens=True) if reasoning_content else 0)
)
reasoning_tokens = min(resolved_reasoning_tokens, completion_tokens)
completion_token_details = CompletionTokensDetailsWrapper(
reasoning_tokens=reasoning_tokens if reasoning_tokens > 0 else 0,
text_tokens=(completion_tokens - reasoning_tokens if reasoning_tokens > 0 else completion_tokens),

View file

@ -669,6 +669,7 @@ class AnthropicPassthroughLoggingHandler:
tool_search_requests: Optional[int] = None
inference_geo: Optional[str] = None
stop_reason: Optional[str] = None
thinking_tokens: Optional[int] = None
found_usage = False
resolved_model = model
for _chunk_str in all_chunks:
@ -693,6 +694,9 @@ class AnthropicPassthroughLoggingHandler:
inference_geo = usage.get("inference_geo")
if usage.get("output_tokens") is not None:
output_tokens = usage.get("output_tokens")
_otd = usage.get("output_tokens_details")
if isinstance(_otd, dict) and _otd.get("thinking_tokens") is not None:
thinking_tokens = _otd.get("thinking_tokens")
found_usage = True
elif event_type == "message_delta":
_delta_stop = (data.get("delta") or {}).get("stop_reason")
@ -711,6 +715,9 @@ class AnthropicPassthroughLoggingHandler:
cache_read = usage.get("cache_read_input_tokens")
if usage.get("inference_geo") is not None:
inference_geo = usage.get("inference_geo")
_otd = usage.get("output_tokens_details")
if isinstance(_otd, dict) and _otd.get("thinking_tokens") is not None:
thinking_tokens = _otd.get("thinking_tokens")
found_usage = True
if not found_usage:
return None
@ -742,6 +749,8 @@ class AnthropicPassthroughLoggingHandler:
usage_object["server_tool_use"] = _server_tool_use
if inference_geo is not None:
usage_object["inference_geo"] = inference_geo
if thinking_tokens is not None:
usage_object["output_tokens_details"] = {"thinking_tokens": thinking_tokens}
usage_obj = AnthropicConfig().calculate_usage(usage_object=usage_object, reasoning_content=None)
return ModelResponse(
model=resolved_model,

View file

@ -119,6 +119,97 @@ def test_calculate_usage_clamps_text_tokens_when_reasoning_estimate_exceeds_outp
assert usage.completion_tokens_details.text_tokens == 0
def test_calculate_usage_prefers_provider_reported_thinking_tokens():
"""
Anthropic / Bedrock report extended-thinking usage as
``output_tokens_details.thinking_tokens``; it must win over the
token_counter estimate of the thinking text.
"""
config = AnthropicConfig()
usage = config.calculate_usage(
usage_object={
"input_tokens": 70,
"output_tokens": 548,
"output_tokens_details": {"thinking_tokens": 275},
},
reasoning_content="short thinking text",
)
assert usage.completion_tokens_details is not None
assert usage.completion_tokens_details.reasoning_tokens == 275
assert usage.completion_tokens_details.text_tokens == 548 - 275
def test_calculate_usage_reports_thinking_tokens_when_thinking_text_is_empty():
"""
Bedrock returns thinking blocks whose text is empty (redacted / truncated),
which used to log reasoning_tokens=0 even though the model reasoned.
"""
config = AnthropicConfig()
usage = config.calculate_usage(
usage_object={
"input_tokens": 78,
"output_tokens": 64,
"output_tokens_details": {"thinking_tokens": 64},
},
reasoning_content="",
)
assert usage.completion_tokens_details is not None
assert usage.completion_tokens_details.reasoning_tokens == 64
assert usage.completion_tokens_details.text_tokens == 0
def test_calculate_usage_falls_back_to_estimate_without_thinking_tokens():
config = AnthropicConfig()
usage = config.calculate_usage(
usage_object={"input_tokens": 10, "output_tokens": 500},
reasoning_content="some reasoning text that tokenizes to a few tokens",
)
assert usage.completion_tokens_details is not None
assert usage.completion_tokens_details.reasoning_tokens > 0
def test_calculate_usage_ignores_non_int_thinking_tokens():
config = AnthropicConfig()
usage = config.calculate_usage(
usage_object={
"input_tokens": 10,
"output_tokens": 500,
"output_tokens_details": {"thinking_tokens": None},
},
reasoning_content=None,
)
assert usage.completion_tokens_details is not None
assert usage.completion_tokens_details.reasoning_tokens == 0
assert usage.completion_tokens_details.text_tokens == 500
def test_calculate_usage_sums_thinking_tokens_across_iterations():
config = AnthropicConfig()
usage = config.calculate_usage(
usage_object={
"iterations": [
{"input_tokens": 10, "output_tokens": 100, "output_tokens_details": {"thinking_tokens": 40}},
{"input_tokens": 10, "output_tokens": 200, "output_tokens_details": {"thinking_tokens": 60}},
]
},
reasoning_content=None,
)
assert usage.completion_tokens == 300
assert usage.completion_tokens_details is not None
assert usage.completion_tokens_details.reasoning_tokens == 100
assert usage.completion_tokens_details.text_tokens == 200
def test_calculate_usage_handles_mocked_output_tokens_with_reasoning_content():
config = AnthropicConfig()

View file

@ -1055,6 +1055,54 @@ class TestBuildCompleteStreamingResponseRobustness:
assert result.choices[0].message.content == "The stream ends with [DONE]"
class TestStreamingThinkingTokens:
"""
A streamed thinking response must be logged with the reasoning token count
Anthropic/Bedrock reported in message_delta, not a token_counter estimate of
the thinking text (which is 0 when the thinking text is empty/redacted).
"""
@staticmethod
def _chunks(thinking_text: str) -> List[str]:
return [
'event: message_start\ndata: {"type":"message_start","message":{"id":"msg_1","type":"message","role":"assistant","content":[],"model":"claude-sonnet-4-5-20250929","stop_reason":null,"stop_sequence":null,"usage":{"input_tokens":70,"output_tokens":1}}}',
'event: content_block_start\ndata: {"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":"","signature":""}}',
"event: content_block_delta\ndata: "
+ json.dumps(
{
"type": "content_block_delta",
"index": 0,
"delta": {"type": "thinking_delta", "thinking": thinking_text},
}
),
'event: content_block_stop\ndata: {"type":"content_block_stop","index":0}',
'event: content_block_start\ndata: {"type":"content_block_start","index":1,"content_block":{"type":"text","text":""}}',
'event: content_block_delta\ndata: {"type":"content_block_delta","index":1,"delta":{"type":"text_delta","text":"33 liters"}}',
'event: content_block_stop\ndata: {"type":"content_block_stop","index":1}',
'event: message_delta\ndata: {"type":"message_delta","delta":{"stop_reason":"end_turn","stop_sequence":null},"usage":{"input_tokens":70,"output_tokens":548,"output_tokens_details":{"thinking_tokens":275}}}',
'event: message_stop\ndata: {"type":"message_stop"}',
]
def _build(self, thinking_text: str):
logging_obj = MagicMock()
logging_obj.model_call_details = {}
return AnthropicPassthroughLoggingHandler._build_complete_streaming_response(
all_chunks=self._chunks(thinking_text),
litellm_logging_obj=logging_obj,
model="claude-sonnet-4-5-20250929",
)
def test_reported_thinking_tokens_used_for_streamed_thinking_text(self):
result = self._build("Let me work through the fuel calculation step by step.")
assert result is not None
assert result.usage.completion_tokens_details.reasoning_tokens == 275
def test_reported_thinking_tokens_used_when_thinking_text_is_empty(self):
result = self._build("")
assert result is not None
assert result.usage.completion_tokens_details.reasoning_tokens == 275
class TestPureTextFastPathParity:
"""
The pure-text fast path in _build_complete_streaming_response must produce
@ -1939,6 +1987,36 @@ class TestAnthropicUsageOnlyFallback:
assert usage.prompt_tokens_details.cached_tokens == 40
assert usage.server_tool_use.web_search_requests == 2
def test_build_usage_only_recovers_provider_reported_thinking_tokens(self):
chunks = [
_sse_bytes(
{
"type": "message_start",
"message": {
"model": "claude-sonnet-4-5-20250929",
"usage": {"input_tokens": 70, "output_tokens": 1},
},
}
),
_sse_bytes(
{
"type": "message_delta",
"usage": {
"output_tokens": 548,
"output_tokens_details": {"thinking_tokens": 275},
},
}
),
]
response = (
AnthropicPassthroughLoggingHandler._build_usage_only_response_from_chunks(
all_chunks=chunks, model="claude-sonnet-4-5-20250929"
)
)
assert response is not None
assert response.usage.completion_tokens_details.reasoning_tokens == 275
assert response.usage.completion_tokens_details.text_tokens == 273
def test_build_usage_only_returns_none_without_usage_events(self):
chunks = [_sse_bytes({"type": "content_block_delta", "delta": {"text": "hi"}})]
assert (