fix(anthropic): read usage and status from dict-shaped Responses completed events

The response.completed/incomplete branch used getattr-only access for
status, usage, token fields, and output, while the rest of the wrapper
is dict-aware. Dict-shaped payloads therefore always produced
message_delta.usage of 0/0 (disabling spend tracking and TPM
enforcement), mapped response.incomplete to end_turn instead of
max_tokens, and missed function_call outputs for the tool_use
stop_reason. Also removes two dead *_tokens_details assignments that
were immediately overwritten.

Addresses the usage-extraction finding in #32086

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
David-Wu1119 2026-07-09 14:13:56 +08:00
parent cd6e8cdf23
commit 78111d8947
2 changed files with 78 additions and 13 deletions

View file

@ -9,6 +9,12 @@ from litellm import verbose_logger
from litellm._uuid import uuid
def _get_field(obj: Any, key: str, default: Any = None) -> Any:
if isinstance(obj, dict):
return obj.get(key, default)
return getattr(obj, key, default)
class AnthropicResponsesStreamWrapper:
"""
Wraps a Responses API streaming iterator and re-emits events in Anthropic SSE format.
@ -231,22 +237,19 @@ class AnthropicResponsesStreamWrapper:
cache_read_tokens = 0
if response_obj is not None:
status = getattr(response_obj, "status", None)
status = _get_field(response_obj, "status")
if status == "incomplete":
stop_reason = "max_tokens"
usage = getattr(response_obj, "usage", None)
usage = _get_field(response_obj, "usage")
if usage is not None:
input_tokens = getattr(usage, "input_tokens", 0) or 0
output_tokens = getattr(usage, "output_tokens", 0) or 0
cache_creation_tokens = getattr(usage, "input_tokens_details", None) # type: ignore[assignment]
cache_read_tokens = getattr(usage, "output_tokens_details", None) # type: ignore[assignment]
# Prefer direct cache fields if present
cache_creation_tokens = int(getattr(usage, "cache_creation_input_tokens", 0) or 0)
cache_read_tokens = int(getattr(usage, "cache_read_input_tokens", 0) or 0)
input_tokens = _get_field(usage, "input_tokens", 0) or 0
output_tokens = _get_field(usage, "output_tokens", 0) or 0
cache_creation_tokens = int(_get_field(usage, "cache_creation_input_tokens", 0) or 0)
cache_read_tokens = int(_get_field(usage, "cache_read_input_tokens", 0) or 0)
# Check if tool_use was in the output to override stop_reason
if response_obj is not None:
output = getattr(response_obj, "output", []) or []
output = _get_field(response_obj, "output", []) or []
for out_item in output:
out_type = getattr(out_item, "type", None) or (
out_item.get("type") if isinstance(out_item, dict) else None

View file

@ -5,10 +5,9 @@ Tests for AnthropicResponsesStreamWrapper
import os
import sys
from types import SimpleNamespace
sys.path.insert(
0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../.."))
)
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../..")))
from litellm.llms.anthropic.experimental_pass_through.responses_adapters.streaming_iterator import (
AnthropicResponsesStreamWrapper,
@ -77,3 +76,66 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded:
("content_block_start", 0),
("content_block_delta", 0),
]
class TestDictShapedCompletedEvents:
"""Usage, status, and output must be read from dict-shaped
`response.completed` payloads, not only attribute-shaped ones.
Previously this branch used getattr-only access, so dict-shaped events
always produced usage 0/0 (disabling spend tracking and TPM enforcement,
https://github.com/BerriAI/litellm/issues/32086) and mapped
`response.incomplete` to end_turn instead of max_tokens.
"""
def test_dict_usage_is_extracted(self):
chunks = _process_all(
[
{
"type": "response.completed",
"response": {
"status": "completed",
"usage": {
"input_tokens": 11,
"output_tokens": 42,
"cache_read_input_tokens": 7,
},
},
}
]
)
assert chunks[0]["type"] == "message_delta"
assert chunks[0]["usage"] == {
"input_tokens": 11,
"output_tokens": 42,
"cache_read_input_tokens": 7,
}
def test_dict_incomplete_maps_to_max_tokens(self):
chunks = _process_all([{"type": "response.incomplete", "response": {"status": "incomplete"}}])
assert chunks[0]["delta"]["stop_reason"] == "max_tokens"
def test_dict_function_call_output_maps_to_tool_use(self):
chunks = _process_all(
[
{
"type": "response.completed",
"response": {
"status": "completed",
"output": [{"type": "function_call", "name": "get_weather"}],
},
}
]
)
assert chunks[0]["delta"]["stop_reason"] == "tool_use"
def test_object_shaped_usage_still_extracted(self):
usage = SimpleNamespace(
input_tokens=3,
output_tokens=5,
cache_creation_input_tokens=0,
cache_read_input_tokens=0,
)
response = SimpleNamespace(status="completed", usage=usage, output=[])
chunks = _process_all([{"type": "response.completed", "response": response}])
assert chunks[0]["usage"] == {"input_tokens": 3, "output_tokens": 5}