fix: read cached tokens from input_tokens_details and exclude them from input_tokens

Mirrors the sibling Anthropic adapter: cache reads fall back to the
OpenAI Responses usage shape (input_tokens_details.cached_tokens), and
Anthropic input_tokens excludes cache read/creation tokens.
This commit is contained in:
David-Wu1119 2026-07-09 14:13:56 +08:00
parent 78111d8947
commit aca69354f2
2 changed files with 28 additions and 10 deletions

View file

@ -246,6 +246,11 @@ class AnthropicResponsesStreamWrapper:
output_tokens = _get_field(usage, "output_tokens", 0) or 0
cache_creation_tokens = int(_get_field(usage, "cache_creation_input_tokens", 0) or 0)
cache_read_tokens = int(_get_field(usage, "cache_read_input_tokens", 0) or 0)
if cache_read_tokens == 0:
input_tokens_details = _get_field(usage, "input_tokens_details")
if input_tokens_details is not None:
cache_read_tokens = int(_get_field(input_tokens_details, "cached_tokens", 0) or 0)
input_tokens = max(input_tokens - cache_read_tokens - cache_creation_tokens, 0)
# Check if tool_use was in the output to override stop_reason
if response_obj is not None:

View file

@ -79,15 +79,6 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded:
class TestDictShapedCompletedEvents:
"""Usage, status, and output must be read from dict-shaped
`response.completed` payloads, not only attribute-shaped ones.
Previously this branch used getattr-only access, so dict-shaped events
always produced usage 0/0 (disabling spend tracking and TPM enforcement,
https://github.com/BerriAI/litellm/issues/32086) and mapped
`response.incomplete` to end_turn instead of max_tokens.
"""
def test_dict_usage_is_extracted(self):
chunks = _process_all(
[
@ -106,11 +97,33 @@ class TestDictShapedCompletedEvents:
)
assert chunks[0]["type"] == "message_delta"
assert chunks[0]["usage"] == {
"input_tokens": 11,
"input_tokens": 4,
"output_tokens": 42,
"cache_read_input_tokens": 7,
}
def test_openai_responses_cached_tokens_details_extracted(self):
chunks = _process_all(
[
{
"type": "response.completed",
"response": {
"status": "completed",
"usage": {
"input_tokens": 100,
"output_tokens": 20,
"input_tokens_details": {"cached_tokens": 30},
},
},
}
]
)
assert chunks[0]["usage"] == {
"input_tokens": 70,
"output_tokens": 20,
"cache_read_input_tokens": 30,
}
def test_dict_incomplete_maps_to_max_tokens(self):
chunks = _process_all([{"type": "response.incomplete", "response": {"status": "incomplete"}}])
assert chunks[0]["delta"]["stop_reason"] == "max_tokens"