From 78111d8947a169168fb896bcb1139f502fa78847 Mon Sep 17 00:00:00 2001 From: David-Wu1119 <133224895+David-Wu1119@users.noreply.github.com> Date: Thu, 9 Jul 2026 14:13:56 +0800 Subject: [PATCH 1/4] fix(anthropic): read usage and status from dict-shaped Responses completed events The response.completed/incomplete branch used getattr-only access for status, usage, token fields, and output, while the rest of the wrapper is dict-aware. Dict-shaped payloads therefore always produced message_delta.usage of 0/0 (disabling spend tracking and TPM enforcement), mapped response.incomplete to end_turn instead of max_tokens, and missed function_call outputs for the tool_use stop_reason. Also removes two dead *_tokens_details assignments that were immediately overwritten. Addresses the usage-extraction finding in #32086 Co-Authored-By: Claude Fable 5 --- .../responses_adapters/streaming_iterator.py | 23 ++++--- ...t_responses_adapters_streaming_iterator.py | 68 ++++++++++++++++++- 2 files changed, 78 insertions(+), 13 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index 0d02b4fa969..48499810d17 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -9,6 +9,12 @@ from litellm import verbose_logger from litellm._uuid import uuid +def _get_field(obj: Any, key: str, default: Any = None) -> Any: + if isinstance(obj, dict): + return obj.get(key, default) + return getattr(obj, key, default) + + class AnthropicResponsesStreamWrapper: """ Wraps a Responses API streaming iterator and re-emits events in Anthropic SSE format. @@ -231,22 +237,19 @@ class AnthropicResponsesStreamWrapper: cache_read_tokens = 0 if response_obj is not None: - status = getattr(response_obj, "status", None) + status = _get_field(response_obj, "status") if status == "incomplete": stop_reason = "max_tokens" - usage = getattr(response_obj, "usage", None) + usage = _get_field(response_obj, "usage") if usage is not None: - input_tokens = getattr(usage, "input_tokens", 0) or 0 - output_tokens = getattr(usage, "output_tokens", 0) or 0 - cache_creation_tokens = getattr(usage, "input_tokens_details", None) # type: ignore[assignment] - cache_read_tokens = getattr(usage, "output_tokens_details", None) # type: ignore[assignment] - # Prefer direct cache fields if present - cache_creation_tokens = int(getattr(usage, "cache_creation_input_tokens", 0) or 0) - cache_read_tokens = int(getattr(usage, "cache_read_input_tokens", 0) or 0) + input_tokens = _get_field(usage, "input_tokens", 0) or 0 + output_tokens = _get_field(usage, "output_tokens", 0) or 0 + cache_creation_tokens = int(_get_field(usage, "cache_creation_input_tokens", 0) or 0) + cache_read_tokens = int(_get_field(usage, "cache_read_input_tokens", 0) or 0) # Check if tool_use was in the output to override stop_reason if response_obj is not None: - output = getattr(response_obj, "output", []) or [] + output = _get_field(response_obj, "output", []) or [] for out_item in output: out_type = getattr(out_item, "type", None) or ( out_item.get("type") if isinstance(out_item, dict) else None diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py index 450f69fb87c..58dbed20462 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py @@ -5,10 +5,9 @@ Tests for AnthropicResponsesStreamWrapper import os import sys +from types import SimpleNamespace -sys.path.insert( - 0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../..")) -) +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../.."))) from litellm.llms.anthropic.experimental_pass_through.responses_adapters.streaming_iterator import ( AnthropicResponsesStreamWrapper, @@ -77,3 +76,66 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: ("content_block_start", 0), ("content_block_delta", 0), ] + + +class TestDictShapedCompletedEvents: + """Usage, status, and output must be read from dict-shaped + `response.completed` payloads, not only attribute-shaped ones. + + Previously this branch used getattr-only access, so dict-shaped events + always produced usage 0/0 (disabling spend tracking and TPM enforcement, + https://github.com/BerriAI/litellm/issues/32086) and mapped + `response.incomplete` to end_turn instead of max_tokens. + """ + + def test_dict_usage_is_extracted(self): + chunks = _process_all( + [ + { + "type": "response.completed", + "response": { + "status": "completed", + "usage": { + "input_tokens": 11, + "output_tokens": 42, + "cache_read_input_tokens": 7, + }, + }, + } + ] + ) + assert chunks[0]["type"] == "message_delta" + assert chunks[0]["usage"] == { + "input_tokens": 11, + "output_tokens": 42, + "cache_read_input_tokens": 7, + } + + def test_dict_incomplete_maps_to_max_tokens(self): + chunks = _process_all([{"type": "response.incomplete", "response": {"status": "incomplete"}}]) + assert chunks[0]["delta"]["stop_reason"] == "max_tokens" + + def test_dict_function_call_output_maps_to_tool_use(self): + chunks = _process_all( + [ + { + "type": "response.completed", + "response": { + "status": "completed", + "output": [{"type": "function_call", "name": "get_weather"}], + }, + } + ] + ) + assert chunks[0]["delta"]["stop_reason"] == "tool_use" + + def test_object_shaped_usage_still_extracted(self): + usage = SimpleNamespace( + input_tokens=3, + output_tokens=5, + cache_creation_input_tokens=0, + cache_read_input_tokens=0, + ) + response = SimpleNamespace(status="completed", usage=usage, output=[]) + chunks = _process_all([{"type": "response.completed", "response": response}]) + assert chunks[0]["usage"] == {"input_tokens": 3, "output_tokens": 5} From aca69354f297a3342dd5fc71366fbeebd7378e50 Mon Sep 17 00:00:00 2001 From: David-Wu1119 <133224895+David-Wu1119@users.noreply.github.com> Date: Thu, 9 Jul 2026 14:13:56 +0800 Subject: [PATCH 2/4] fix: read cached tokens from input_tokens_details and exclude them from input_tokens Mirrors the sibling Anthropic adapter: cache reads fall back to the OpenAI Responses usage shape (input_tokens_details.cached_tokens), and Anthropic input_tokens excludes cache read/creation tokens. --- .../responses_adapters/streaming_iterator.py | 5 +++ ...t_responses_adapters_streaming_iterator.py | 33 +++++++++++++------ 2 files changed, 28 insertions(+), 10 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index 48499810d17..1aa2f074cb8 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -246,6 +246,11 @@ class AnthropicResponsesStreamWrapper: output_tokens = _get_field(usage, "output_tokens", 0) or 0 cache_creation_tokens = int(_get_field(usage, "cache_creation_input_tokens", 0) or 0) cache_read_tokens = int(_get_field(usage, "cache_read_input_tokens", 0) or 0) + if cache_read_tokens == 0: + input_tokens_details = _get_field(usage, "input_tokens_details") + if input_tokens_details is not None: + cache_read_tokens = int(_get_field(input_tokens_details, "cached_tokens", 0) or 0) + input_tokens = max(input_tokens - cache_read_tokens - cache_creation_tokens, 0) # Check if tool_use was in the output to override stop_reason if response_obj is not None: diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py index 58dbed20462..bd9cd92c3b2 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py @@ -79,15 +79,6 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: class TestDictShapedCompletedEvents: - """Usage, status, and output must be read from dict-shaped - `response.completed` payloads, not only attribute-shaped ones. - - Previously this branch used getattr-only access, so dict-shaped events - always produced usage 0/0 (disabling spend tracking and TPM enforcement, - https://github.com/BerriAI/litellm/issues/32086) and mapped - `response.incomplete` to end_turn instead of max_tokens. - """ - def test_dict_usage_is_extracted(self): chunks = _process_all( [ @@ -106,11 +97,33 @@ class TestDictShapedCompletedEvents: ) assert chunks[0]["type"] == "message_delta" assert chunks[0]["usage"] == { - "input_tokens": 11, + "input_tokens": 4, "output_tokens": 42, "cache_read_input_tokens": 7, } + def test_openai_responses_cached_tokens_details_extracted(self): + chunks = _process_all( + [ + { + "type": "response.completed", + "response": { + "status": "completed", + "usage": { + "input_tokens": 100, + "output_tokens": 20, + "input_tokens_details": {"cached_tokens": 30}, + }, + }, + } + ] + ) + assert chunks[0]["usage"] == { + "input_tokens": 70, + "output_tokens": 20, + "cache_read_input_tokens": 30, + } + def test_dict_incomplete_maps_to_max_tokens(self): chunks = _process_all([{"type": "response.incomplete", "response": {"status": "incomplete"}}]) assert chunks[0]["delta"]["stop_reason"] == "max_tokens" From f476158f7a5abcfd64c4bf3e122e70b329f5d642 Mon Sep 17 00:00:00 2001 From: David Wu Date: Sat, 8 Aug 2026 14:47:42 +0800 Subject: [PATCH 3/4] refactor: preserve Anthropic usage type discipline --- .../responses_adapters/streaming_iterator.py | 64 +++++++++++-------- 1 file changed, 39 insertions(+), 25 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index ca23e1e85fa..cc150fca1a0 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -20,6 +20,43 @@ def _get_field(obj: Any, key: str, default: Any = None) -> Any: return getattr(obj, key, default) +def _translate_usage(raw_usage: Any) -> AnthropicUsage: + if raw_usage is None or isinstance(raw_usage, ResponseAPIUsage): + return LiteLLMAnthropicToResponsesAPIAdapter.translate_responses_api_usage_to_anthropic_usage(raw_usage) + + input_tokens: Final = int(_get_field(raw_usage, "input_tokens", 0) or 0) + output_tokens: Final = int(_get_field(raw_usage, "output_tokens", 0) or 0) + input_tokens_details: Final = _get_field(raw_usage, "input_tokens_details") + cache_creation_tokens: Final = int(_get_field(raw_usage, "cache_creation_input_tokens", 0) or 0) or int( + _get_field(input_tokens_details, "cache_write_tokens", 0) or 0 + ) + cache_read_tokens: Final = int(_get_field(raw_usage, "cache_read_input_tokens", 0) or 0) or int( + _get_field(input_tokens_details, "cached_tokens", 0) or 0 + ) + uncached_input_tokens: Final = max(input_tokens - cache_read_tokens - cache_creation_tokens, 0) + + if cache_creation_tokens and cache_read_tokens: + return AnthropicUsage( + input_tokens=uncached_input_tokens, + output_tokens=output_tokens, + cache_creation_input_tokens=cache_creation_tokens, + cache_read_input_tokens=cache_read_tokens, + ) + if cache_creation_tokens: + return AnthropicUsage( + input_tokens=uncached_input_tokens, + output_tokens=output_tokens, + cache_creation_input_tokens=cache_creation_tokens, + ) + if cache_read_tokens: + return AnthropicUsage( + input_tokens=uncached_input_tokens, + output_tokens=output_tokens, + cache_read_input_tokens=cache_read_tokens, + ) + return AnthropicUsage(input_tokens=uncached_input_tokens, output_tokens=output_tokens) + + class AnthropicResponsesStreamWrapper: """ Wraps a Responses API streaming iterator and re-emits events in Anthropic SSE format. @@ -237,36 +274,13 @@ class AnthropicResponsesStreamWrapper: event.get("response") if isinstance(event, dict) else None ) stop_reason = "end_turn" - anthropic_usage: AnthropicUsage = AnthropicUsage(input_tokens=0, output_tokens=0) + raw_usage: Final = _get_field(response_obj, "usage") if response_obj is not None else None + anthropic_usage: Final = _translate_usage(raw_usage) if response_obj is not None: status: Final = _get_field(response_obj, "status") if status == "incomplete": stop_reason = "max_tokens" - raw_usage: Final = _get_field(response_obj, "usage") - if raw_usage is not None and not isinstance(raw_usage, ResponseAPIUsage): - input_tokens = int(_get_field(raw_usage, "input_tokens", 0) or 0) - output_tokens = int(_get_field(raw_usage, "output_tokens", 0) or 0) - cache_creation_tokens = int(_get_field(raw_usage, "cache_creation_input_tokens", 0) or 0) - cache_read_tokens = int(_get_field(raw_usage, "cache_read_input_tokens", 0) or 0) - input_tokens_details = _get_field(raw_usage, "input_tokens_details") - if input_tokens_details is not None: - if cache_creation_tokens == 0: - cache_creation_tokens = int(_get_field(input_tokens_details, "cache_write_tokens", 0) or 0) - if cache_read_tokens == 0: - cache_read_tokens = int(_get_field(input_tokens_details, "cached_tokens", 0) or 0) - anthropic_usage = AnthropicUsage( - input_tokens=max(input_tokens - cache_read_tokens - cache_creation_tokens, 0), - output_tokens=output_tokens, - ) - if cache_creation_tokens: - anthropic_usage["cache_creation_input_tokens"] = cache_creation_tokens - if cache_read_tokens: - anthropic_usage["cache_read_input_tokens"] = cache_read_tokens - else: - anthropic_usage = ( - LiteLLMAnthropicToResponsesAPIAdapter.translate_responses_api_usage_to_anthropic_usage(raw_usage) - ) # Check if tool_use was in the output to override stop_reason if response_obj is not None: From c8098164cef9e10e7d8cedf2bd2f935b5b39ad4d Mon Sep 17 00:00:00 2001 From: David Wu Date: Sat, 8 Aug 2026 15:01:17 +0800 Subject: [PATCH 4/4] refactor: narrow Anthropic usage helper types --- .../responses_adapters/streaming_iterator.py | 21 ++++++++++--------- 1 file changed, 11 insertions(+), 10 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index cc150fca1a0..0c13a718e36 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -14,13 +14,13 @@ from litellm.types.llms.openai import ResponseAPIUsage from .transformation import LiteLLMAnthropicToResponsesAPIAdapter -def _get_field(obj: Any, key: str, default: Any = None) -> Any: +def _get_field(obj: object, key: str, default: object = None) -> object: if isinstance(obj, dict): return obj.get(key, default) return getattr(obj, key, default) -def _translate_usage(raw_usage: Any) -> AnthropicUsage: +def _translate_usage(raw_usage: object) -> AnthropicUsage: if raw_usage is None or isinstance(raw_usage, ResponseAPIUsage): return LiteLLMAnthropicToResponsesAPIAdapter.translate_responses_api_usage_to_anthropic_usage(raw_usage) @@ -284,14 +284,15 @@ class AnthropicResponsesStreamWrapper: # Check if tool_use was in the output to override stop_reason if response_obj is not None: - output: Final = _get_field(response_obj, "output", []) or [] - for out_item in output: - out_type = getattr(out_item, "type", None) or ( - out_item.get("type") if isinstance(out_item, dict) else None - ) - if out_type == "function_call": - stop_reason = "tool_use" - break + output: Final = _get_field(response_obj, "output", []) + if isinstance(output, list): + for out_item in output: + out_type = getattr(out_item, "type", None) or ( + out_item.get("type") if isinstance(out_item, dict) else None + ) + if out_type == "function_call": + stop_reason = "tool_use" + break self._chunk_queue.append( {