diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index 48499810d17..1aa2f074cb8 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -246,6 +246,11 @@ class AnthropicResponsesStreamWrapper: output_tokens = _get_field(usage, "output_tokens", 0) or 0 cache_creation_tokens = int(_get_field(usage, "cache_creation_input_tokens", 0) or 0) cache_read_tokens = int(_get_field(usage, "cache_read_input_tokens", 0) or 0) + if cache_read_tokens == 0: + input_tokens_details = _get_field(usage, "input_tokens_details") + if input_tokens_details is not None: + cache_read_tokens = int(_get_field(input_tokens_details, "cached_tokens", 0) or 0) + input_tokens = max(input_tokens - cache_read_tokens - cache_creation_tokens, 0) # Check if tool_use was in the output to override stop_reason if response_obj is not None: diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py index 58dbed20462..bd9cd92c3b2 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py @@ -79,15 +79,6 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: class TestDictShapedCompletedEvents: - """Usage, status, and output must be read from dict-shaped - `response.completed` payloads, not only attribute-shaped ones. - - Previously this branch used getattr-only access, so dict-shaped events - always produced usage 0/0 (disabling spend tracking and TPM enforcement, - https://github.com/BerriAI/litellm/issues/32086) and mapped - `response.incomplete` to end_turn instead of max_tokens. - """ - def test_dict_usage_is_extracted(self): chunks = _process_all( [ @@ -106,11 +97,33 @@ class TestDictShapedCompletedEvents: ) assert chunks[0]["type"] == "message_delta" assert chunks[0]["usage"] == { - "input_tokens": 11, + "input_tokens": 4, "output_tokens": 42, "cache_read_input_tokens": 7, } + def test_openai_responses_cached_tokens_details_extracted(self): + chunks = _process_all( + [ + { + "type": "response.completed", + "response": { + "status": "completed", + "usage": { + "input_tokens": 100, + "output_tokens": 20, + "input_tokens_details": {"cached_tokens": 30}, + }, + }, + } + ] + ) + assert chunks[0]["usage"] == { + "input_tokens": 70, + "output_tokens": 20, + "cache_read_input_tokens": 30, + } + def test_dict_incomplete_maps_to_max_tokens(self): chunks = _process_all([{"type": "response.incomplete", "response": {"status": "incomplete"}}]) assert chunks[0]["delta"]["stop_reason"] == "max_tokens"