mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(anthropic): log provider-reported thinking tokens as reasoning_tokens
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
81ff7cb38f
commit
7a2735e028
4 changed files with 209 additions and 3 deletions
|
|
@ -1,6 +1,7 @@
|
|||
import json
|
||||
import re
|
||||
import time
|
||||
from collections.abc import Mapping, Sequence
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
|
|
@ -139,6 +140,28 @@ _ANTHROPIC_TOOL_NAME_MAX_LEN = 128
|
|||
ANTHROPIC_TOOL_NAME_REVERSE_MAP_KEY = "_anthropic_tool_name_map"
|
||||
|
||||
|
||||
def _reported_thinking_tokens(usage_object: Mapping[str, Any]) -> Optional[int]:
|
||||
"""Read the provider-reported reasoning token count from an Anthropic usage object.
|
||||
|
||||
Anthropic (and Bedrock's Anthropic-compatible surface) reports extended-thinking
|
||||
tokens as ``usage.output_tokens_details.thinking_tokens``; it is authoritative and
|
||||
is present even when the thinking text is empty or redacted.
|
||||
"""
|
||||
details = usage_object.get("output_tokens_details")
|
||||
if not isinstance(details, Mapping):
|
||||
return None
|
||||
thinking_tokens = details.get("thinking_tokens")
|
||||
if isinstance(thinking_tokens, bool) or not isinstance(thinking_tokens, int):
|
||||
return None
|
||||
return thinking_tokens
|
||||
|
||||
|
||||
def _sum_reported_thinking_tokens(iterations: Sequence[Any]) -> Optional[int]:
|
||||
reported = tuple(_reported_thinking_tokens(iteration) for iteration in iterations if isinstance(iteration, Mapping))
|
||||
present = tuple(tokens for tokens in reported if tokens is not None)
|
||||
return sum(present) if present else None
|
||||
|
||||
|
||||
def _basic_sanitize_anthropic_tool_name(name: str) -> str:
|
||||
"""Lossy: replace [^a-zA-Z0-9_-] with '_' and truncate to 128.
|
||||
|
||||
|
|
@ -2201,10 +2224,15 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
text_tokens=raw_input_tokens,
|
||||
)
|
||||
# Always populate completion_token_details, not just when there's reasoning_content
|
||||
estimated_reasoning_tokens = (
|
||||
token_counter(text=reasoning_content, count_response_tokens=True) if reasoning_content else 0
|
||||
reported_thinking_tokens = (
|
||||
_sum_reported_thinking_tokens(iterations) if iterations else _reported_thinking_tokens(_usage)
|
||||
)
|
||||
reasoning_tokens = min(estimated_reasoning_tokens, completion_tokens)
|
||||
resolved_reasoning_tokens = (
|
||||
reported_thinking_tokens
|
||||
if reported_thinking_tokens is not None
|
||||
else (token_counter(text=reasoning_content, count_response_tokens=True) if reasoning_content else 0)
|
||||
)
|
||||
reasoning_tokens = min(resolved_reasoning_tokens, completion_tokens)
|
||||
completion_token_details = CompletionTokensDetailsWrapper(
|
||||
reasoning_tokens=reasoning_tokens if reasoning_tokens > 0 else 0,
|
||||
text_tokens=(completion_tokens - reasoning_tokens if reasoning_tokens > 0 else completion_tokens),
|
||||
|
|
|
|||
|
|
@ -669,6 +669,7 @@ class AnthropicPassthroughLoggingHandler:
|
|||
tool_search_requests: Optional[int] = None
|
||||
inference_geo: Optional[str] = None
|
||||
stop_reason: Optional[str] = None
|
||||
thinking_tokens: Optional[int] = None
|
||||
found_usage = False
|
||||
resolved_model = model
|
||||
for _chunk_str in all_chunks:
|
||||
|
|
@ -693,6 +694,9 @@ class AnthropicPassthroughLoggingHandler:
|
|||
inference_geo = usage.get("inference_geo")
|
||||
if usage.get("output_tokens") is not None:
|
||||
output_tokens = usage.get("output_tokens")
|
||||
_otd = usage.get("output_tokens_details")
|
||||
if isinstance(_otd, dict) and _otd.get("thinking_tokens") is not None:
|
||||
thinking_tokens = _otd.get("thinking_tokens")
|
||||
found_usage = True
|
||||
elif event_type == "message_delta":
|
||||
_delta_stop = (data.get("delta") or {}).get("stop_reason")
|
||||
|
|
@ -711,6 +715,9 @@ class AnthropicPassthroughLoggingHandler:
|
|||
cache_read = usage.get("cache_read_input_tokens")
|
||||
if usage.get("inference_geo") is not None:
|
||||
inference_geo = usage.get("inference_geo")
|
||||
_otd = usage.get("output_tokens_details")
|
||||
if isinstance(_otd, dict) and _otd.get("thinking_tokens") is not None:
|
||||
thinking_tokens = _otd.get("thinking_tokens")
|
||||
found_usage = True
|
||||
if not found_usage:
|
||||
return None
|
||||
|
|
@ -742,6 +749,8 @@ class AnthropicPassthroughLoggingHandler:
|
|||
usage_object["server_tool_use"] = _server_tool_use
|
||||
if inference_geo is not None:
|
||||
usage_object["inference_geo"] = inference_geo
|
||||
if thinking_tokens is not None:
|
||||
usage_object["output_tokens_details"] = {"thinking_tokens": thinking_tokens}
|
||||
usage_obj = AnthropicConfig().calculate_usage(usage_object=usage_object, reasoning_content=None)
|
||||
return ModelResponse(
|
||||
model=resolved_model,
|
||||
|
|
|
|||
|
|
@ -119,6 +119,97 @@ def test_calculate_usage_clamps_text_tokens_when_reasoning_estimate_exceeds_outp
|
|||
assert usage.completion_tokens_details.text_tokens == 0
|
||||
|
||||
|
||||
def test_calculate_usage_prefers_provider_reported_thinking_tokens():
|
||||
"""
|
||||
Anthropic / Bedrock report extended-thinking usage as
|
||||
``output_tokens_details.thinking_tokens``; it must win over the
|
||||
token_counter estimate of the thinking text.
|
||||
"""
|
||||
config = AnthropicConfig()
|
||||
|
||||
usage = config.calculate_usage(
|
||||
usage_object={
|
||||
"input_tokens": 70,
|
||||
"output_tokens": 548,
|
||||
"output_tokens_details": {"thinking_tokens": 275},
|
||||
},
|
||||
reasoning_content="short thinking text",
|
||||
)
|
||||
|
||||
assert usage.completion_tokens_details is not None
|
||||
assert usage.completion_tokens_details.reasoning_tokens == 275
|
||||
assert usage.completion_tokens_details.text_tokens == 548 - 275
|
||||
|
||||
|
||||
def test_calculate_usage_reports_thinking_tokens_when_thinking_text_is_empty():
|
||||
"""
|
||||
Bedrock returns thinking blocks whose text is empty (redacted / truncated),
|
||||
which used to log reasoning_tokens=0 even though the model reasoned.
|
||||
"""
|
||||
config = AnthropicConfig()
|
||||
|
||||
usage = config.calculate_usage(
|
||||
usage_object={
|
||||
"input_tokens": 78,
|
||||
"output_tokens": 64,
|
||||
"output_tokens_details": {"thinking_tokens": 64},
|
||||
},
|
||||
reasoning_content="",
|
||||
)
|
||||
|
||||
assert usage.completion_tokens_details is not None
|
||||
assert usage.completion_tokens_details.reasoning_tokens == 64
|
||||
assert usage.completion_tokens_details.text_tokens == 0
|
||||
|
||||
|
||||
def test_calculate_usage_falls_back_to_estimate_without_thinking_tokens():
|
||||
config = AnthropicConfig()
|
||||
|
||||
usage = config.calculate_usage(
|
||||
usage_object={"input_tokens": 10, "output_tokens": 500},
|
||||
reasoning_content="some reasoning text that tokenizes to a few tokens",
|
||||
)
|
||||
|
||||
assert usage.completion_tokens_details is not None
|
||||
assert usage.completion_tokens_details.reasoning_tokens > 0
|
||||
|
||||
|
||||
def test_calculate_usage_ignores_non_int_thinking_tokens():
|
||||
config = AnthropicConfig()
|
||||
|
||||
usage = config.calculate_usage(
|
||||
usage_object={
|
||||
"input_tokens": 10,
|
||||
"output_tokens": 500,
|
||||
"output_tokens_details": {"thinking_tokens": None},
|
||||
},
|
||||
reasoning_content=None,
|
||||
)
|
||||
|
||||
assert usage.completion_tokens_details is not None
|
||||
assert usage.completion_tokens_details.reasoning_tokens == 0
|
||||
assert usage.completion_tokens_details.text_tokens == 500
|
||||
|
||||
|
||||
def test_calculate_usage_sums_thinking_tokens_across_iterations():
|
||||
config = AnthropicConfig()
|
||||
|
||||
usage = config.calculate_usage(
|
||||
usage_object={
|
||||
"iterations": [
|
||||
{"input_tokens": 10, "output_tokens": 100, "output_tokens_details": {"thinking_tokens": 40}},
|
||||
{"input_tokens": 10, "output_tokens": 200, "output_tokens_details": {"thinking_tokens": 60}},
|
||||
]
|
||||
},
|
||||
reasoning_content=None,
|
||||
)
|
||||
|
||||
assert usage.completion_tokens == 300
|
||||
assert usage.completion_tokens_details is not None
|
||||
assert usage.completion_tokens_details.reasoning_tokens == 100
|
||||
assert usage.completion_tokens_details.text_tokens == 200
|
||||
|
||||
|
||||
def test_calculate_usage_handles_mocked_output_tokens_with_reasoning_content():
|
||||
config = AnthropicConfig()
|
||||
|
||||
|
|
|
|||
|
|
@ -1055,6 +1055,54 @@ class TestBuildCompleteStreamingResponseRobustness:
|
|||
assert result.choices[0].message.content == "The stream ends with [DONE]"
|
||||
|
||||
|
||||
class TestStreamingThinkingTokens:
|
||||
"""
|
||||
A streamed thinking response must be logged with the reasoning token count
|
||||
Anthropic/Bedrock reported in message_delta, not a token_counter estimate of
|
||||
the thinking text (which is 0 when the thinking text is empty/redacted).
|
||||
"""
|
||||
|
||||
@staticmethod
|
||||
def _chunks(thinking_text: str) -> List[str]:
|
||||
return [
|
||||
'event: message_start\ndata: {"type":"message_start","message":{"id":"msg_1","type":"message","role":"assistant","content":[],"model":"claude-sonnet-4-5-20250929","stop_reason":null,"stop_sequence":null,"usage":{"input_tokens":70,"output_tokens":1}}}',
|
||||
'event: content_block_start\ndata: {"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":"","signature":""}}',
|
||||
"event: content_block_delta\ndata: "
|
||||
+ json.dumps(
|
||||
{
|
||||
"type": "content_block_delta",
|
||||
"index": 0,
|
||||
"delta": {"type": "thinking_delta", "thinking": thinking_text},
|
||||
}
|
||||
),
|
||||
'event: content_block_stop\ndata: {"type":"content_block_stop","index":0}',
|
||||
'event: content_block_start\ndata: {"type":"content_block_start","index":1,"content_block":{"type":"text","text":""}}',
|
||||
'event: content_block_delta\ndata: {"type":"content_block_delta","index":1,"delta":{"type":"text_delta","text":"33 liters"}}',
|
||||
'event: content_block_stop\ndata: {"type":"content_block_stop","index":1}',
|
||||
'event: message_delta\ndata: {"type":"message_delta","delta":{"stop_reason":"end_turn","stop_sequence":null},"usage":{"input_tokens":70,"output_tokens":548,"output_tokens_details":{"thinking_tokens":275}}}',
|
||||
'event: message_stop\ndata: {"type":"message_stop"}',
|
||||
]
|
||||
|
||||
def _build(self, thinking_text: str):
|
||||
logging_obj = MagicMock()
|
||||
logging_obj.model_call_details = {}
|
||||
return AnthropicPassthroughLoggingHandler._build_complete_streaming_response(
|
||||
all_chunks=self._chunks(thinking_text),
|
||||
litellm_logging_obj=logging_obj,
|
||||
model="claude-sonnet-4-5-20250929",
|
||||
)
|
||||
|
||||
def test_reported_thinking_tokens_used_for_streamed_thinking_text(self):
|
||||
result = self._build("Let me work through the fuel calculation step by step.")
|
||||
assert result is not None
|
||||
assert result.usage.completion_tokens_details.reasoning_tokens == 275
|
||||
|
||||
def test_reported_thinking_tokens_used_when_thinking_text_is_empty(self):
|
||||
result = self._build("")
|
||||
assert result is not None
|
||||
assert result.usage.completion_tokens_details.reasoning_tokens == 275
|
||||
|
||||
|
||||
class TestPureTextFastPathParity:
|
||||
"""
|
||||
The pure-text fast path in _build_complete_streaming_response must produce
|
||||
|
|
@ -1939,6 +1987,36 @@ class TestAnthropicUsageOnlyFallback:
|
|||
assert usage.prompt_tokens_details.cached_tokens == 40
|
||||
assert usage.server_tool_use.web_search_requests == 2
|
||||
|
||||
def test_build_usage_only_recovers_provider_reported_thinking_tokens(self):
|
||||
chunks = [
|
||||
_sse_bytes(
|
||||
{
|
||||
"type": "message_start",
|
||||
"message": {
|
||||
"model": "claude-sonnet-4-5-20250929",
|
||||
"usage": {"input_tokens": 70, "output_tokens": 1},
|
||||
},
|
||||
}
|
||||
),
|
||||
_sse_bytes(
|
||||
{
|
||||
"type": "message_delta",
|
||||
"usage": {
|
||||
"output_tokens": 548,
|
||||
"output_tokens_details": {"thinking_tokens": 275},
|
||||
},
|
||||
}
|
||||
),
|
||||
]
|
||||
response = (
|
||||
AnthropicPassthroughLoggingHandler._build_usage_only_response_from_chunks(
|
||||
all_chunks=chunks, model="claude-sonnet-4-5-20250929"
|
||||
)
|
||||
)
|
||||
assert response is not None
|
||||
assert response.usage.completion_tokens_details.reasoning_tokens == 275
|
||||
assert response.usage.completion_tokens_details.text_tokens == 273
|
||||
|
||||
def test_build_usage_only_returns_none_without_usage_events(self):
|
||||
chunks = [_sse_bytes({"type": "content_block_delta", "delta": {"text": "hi"}})]
|
||||
assert (
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue