From aca69354f297a3342dd5fc71366fbeebd7378e50 Mon Sep 17 00:00:00 2001 From: David-Wu1119 <133224895+David-Wu1119@users.noreply.github.com> Date: Thu, 9 Jul 2026 14:13:56 +0800 Subject: [PATCH] fix: read cached tokens from input_tokens_details and exclude them from input_tokens Mirrors the sibling Anthropic adapter: cache reads fall back to the OpenAI Responses usage shape (input_tokens_details.cached_tokens), and Anthropic input_tokens excludes cache read/creation tokens. --- .../responses_adapters/streaming_iterator.py | 5 +++ ...t_responses_adapters_streaming_iterator.py | 33 +++++++++++++------ 2 files changed, 28 insertions(+), 10 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index 48499810d17..1aa2f074cb8 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -246,6 +246,11 @@ class AnthropicResponsesStreamWrapper: output_tokens = _get_field(usage, "output_tokens", 0) or 0 cache_creation_tokens = int(_get_field(usage, "cache_creation_input_tokens", 0) or 0) cache_read_tokens = int(_get_field(usage, "cache_read_input_tokens", 0) or 0) + if cache_read_tokens == 0: + input_tokens_details = _get_field(usage, "input_tokens_details") + if input_tokens_details is not None: + cache_read_tokens = int(_get_field(input_tokens_details, "cached_tokens", 0) or 0) + input_tokens = max(input_tokens - cache_read_tokens - cache_creation_tokens, 0) # Check if tool_use was in the output to override stop_reason if response_obj is not None: diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py index 58dbed20462..bd9cd92c3b2 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py @@ -79,15 +79,6 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: class TestDictShapedCompletedEvents: - """Usage, status, and output must be read from dict-shaped - `response.completed` payloads, not only attribute-shaped ones. - - Previously this branch used getattr-only access, so dict-shaped events - always produced usage 0/0 (disabling spend tracking and TPM enforcement, - https://github.com/BerriAI/litellm/issues/32086) and mapped - `response.incomplete` to end_turn instead of max_tokens. - """ - def test_dict_usage_is_extracted(self): chunks = _process_all( [ @@ -106,11 +97,33 @@ class TestDictShapedCompletedEvents: ) assert chunks[0]["type"] == "message_delta" assert chunks[0]["usage"] == { - "input_tokens": 11, + "input_tokens": 4, "output_tokens": 42, "cache_read_input_tokens": 7, } + def test_openai_responses_cached_tokens_details_extracted(self): + chunks = _process_all( + [ + { + "type": "response.completed", + "response": { + "status": "completed", + "usage": { + "input_tokens": 100, + "output_tokens": 20, + "input_tokens_details": {"cached_tokens": 30}, + }, + }, + } + ] + ) + assert chunks[0]["usage"] == { + "input_tokens": 70, + "output_tokens": 20, + "cache_read_input_tokens": 30, + } + def test_dict_incomplete_maps_to_max_tokens(self): chunks = _process_all([{"type": "response.incomplete", "response": {"status": "incomplete"}}]) assert chunks[0]["delta"]["stop_reason"] == "max_tokens"