From 78111d8947a169168fb896bcb1139f502fa78847 Mon Sep 17 00:00:00 2001 From: David-Wu1119 <133224895+David-Wu1119@users.noreply.github.com> Date: Thu, 9 Jul 2026 14:13:56 +0800 Subject: [PATCH] fix(anthropic): read usage and status from dict-shaped Responses completed events The response.completed/incomplete branch used getattr-only access for status, usage, token fields, and output, while the rest of the wrapper is dict-aware. Dict-shaped payloads therefore always produced message_delta.usage of 0/0 (disabling spend tracking and TPM enforcement), mapped response.incomplete to end_turn instead of max_tokens, and missed function_call outputs for the tool_use stop_reason. Also removes two dead *_tokens_details assignments that were immediately overwritten. Addresses the usage-extraction finding in #32086 Co-Authored-By: Claude Fable 5 --- .../responses_adapters/streaming_iterator.py | 23 ++++--- ...t_responses_adapters_streaming_iterator.py | 68 ++++++++++++++++++- 2 files changed, 78 insertions(+), 13 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py index 0d02b4fa969..48499810d17 100644 --- a/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py +++ b/litellm/llms/anthropic/experimental_pass_through/responses_adapters/streaming_iterator.py @@ -9,6 +9,12 @@ from litellm import verbose_logger from litellm._uuid import uuid +def _get_field(obj: Any, key: str, default: Any = None) -> Any: + if isinstance(obj, dict): + return obj.get(key, default) + return getattr(obj, key, default) + + class AnthropicResponsesStreamWrapper: """ Wraps a Responses API streaming iterator and re-emits events in Anthropic SSE format. @@ -231,22 +237,19 @@ class AnthropicResponsesStreamWrapper: cache_read_tokens = 0 if response_obj is not None: - status = getattr(response_obj, "status", None) + status = _get_field(response_obj, "status") if status == "incomplete": stop_reason = "max_tokens" - usage = getattr(response_obj, "usage", None) + usage = _get_field(response_obj, "usage") if usage is not None: - input_tokens = getattr(usage, "input_tokens", 0) or 0 - output_tokens = getattr(usage, "output_tokens", 0) or 0 - cache_creation_tokens = getattr(usage, "input_tokens_details", None) # type: ignore[assignment] - cache_read_tokens = getattr(usage, "output_tokens_details", None) # type: ignore[assignment] - # Prefer direct cache fields if present - cache_creation_tokens = int(getattr(usage, "cache_creation_input_tokens", 0) or 0) - cache_read_tokens = int(getattr(usage, "cache_read_input_tokens", 0) or 0) + input_tokens = _get_field(usage, "input_tokens", 0) or 0 + output_tokens = _get_field(usage, "output_tokens", 0) or 0 + cache_creation_tokens = int(_get_field(usage, "cache_creation_input_tokens", 0) or 0) + cache_read_tokens = int(_get_field(usage, "cache_read_input_tokens", 0) or 0) # Check if tool_use was in the output to override stop_reason if response_obj is not None: - output = getattr(response_obj, "output", []) or [] + output = _get_field(response_obj, "output", []) or [] for out_item in output: out_type = getattr(out_item, "type", None) or ( out_item.get("type") if isinstance(out_item, dict) else None diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py index 450f69fb87c..58dbed20462 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/responses_adapters/test_responses_adapters_streaming_iterator.py @@ -5,10 +5,9 @@ Tests for AnthropicResponsesStreamWrapper import os import sys +from types import SimpleNamespace -sys.path.insert( - 0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../..")) -) +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../.."))) from litellm.llms.anthropic.experimental_pass_through.responses_adapters.streaming_iterator import ( AnthropicResponsesStreamWrapper, @@ -77,3 +76,66 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded: ("content_block_start", 0), ("content_block_delta", 0), ] + + +class TestDictShapedCompletedEvents: + """Usage, status, and output must be read from dict-shaped + `response.completed` payloads, not only attribute-shaped ones. + + Previously this branch used getattr-only access, so dict-shaped events + always produced usage 0/0 (disabling spend tracking and TPM enforcement, + https://github.com/BerriAI/litellm/issues/32086) and mapped + `response.incomplete` to end_turn instead of max_tokens. + """ + + def test_dict_usage_is_extracted(self): + chunks = _process_all( + [ + { + "type": "response.completed", + "response": { + "status": "completed", + "usage": { + "input_tokens": 11, + "output_tokens": 42, + "cache_read_input_tokens": 7, + }, + }, + } + ] + ) + assert chunks[0]["type"] == "message_delta" + assert chunks[0]["usage"] == { + "input_tokens": 11, + "output_tokens": 42, + "cache_read_input_tokens": 7, + } + + def test_dict_incomplete_maps_to_max_tokens(self): + chunks = _process_all([{"type": "response.incomplete", "response": {"status": "incomplete"}}]) + assert chunks[0]["delta"]["stop_reason"] == "max_tokens" + + def test_dict_function_call_output_maps_to_tool_use(self): + chunks = _process_all( + [ + { + "type": "response.completed", + "response": { + "status": "completed", + "output": [{"type": "function_call", "name": "get_weather"}], + }, + } + ] + ) + assert chunks[0]["delta"]["stop_reason"] == "tool_use" + + def test_object_shaped_usage_still_extracted(self): + usage = SimpleNamespace( + input_tokens=3, + output_tokens=5, + cache_creation_input_tokens=0, + cache_read_input_tokens=0, + ) + response = SimpleNamespace(status="completed", usage=usage, output=[]) + chunks = _process_all([{"type": "response.completed", "response": response}]) + assert chunks[0]["usage"] == {"input_tokens": 3, "output_tokens": 5}