mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(anthropic): read usage and status from dict-shaped Responses completed events
The response.completed/incomplete branch used getattr-only access for status, usage, token fields, and output, while the rest of the wrapper is dict-aware. Dict-shaped payloads therefore always produced message_delta.usage of 0/0 (disabling spend tracking and TPM enforcement), mapped response.incomplete to end_turn instead of max_tokens, and missed function_call outputs for the tool_use stop_reason. Also removes two dead *_tokens_details assignments that were immediately overwritten. Addresses the usage-extraction finding in #32086 Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
parent
cd6e8cdf23
commit
78111d8947
2 changed files with 78 additions and 13 deletions
|
|
@ -9,6 +9,12 @@ from litellm import verbose_logger
|
|||
from litellm._uuid import uuid
|
||||
|
||||
|
||||
def _get_field(obj: Any, key: str, default: Any = None) -> Any:
|
||||
if isinstance(obj, dict):
|
||||
return obj.get(key, default)
|
||||
return getattr(obj, key, default)
|
||||
|
||||
|
||||
class AnthropicResponsesStreamWrapper:
|
||||
"""
|
||||
Wraps a Responses API streaming iterator and re-emits events in Anthropic SSE format.
|
||||
|
|
@ -231,22 +237,19 @@ class AnthropicResponsesStreamWrapper:
|
|||
cache_read_tokens = 0
|
||||
|
||||
if response_obj is not None:
|
||||
status = getattr(response_obj, "status", None)
|
||||
status = _get_field(response_obj, "status")
|
||||
if status == "incomplete":
|
||||
stop_reason = "max_tokens"
|
||||
usage = getattr(response_obj, "usage", None)
|
||||
usage = _get_field(response_obj, "usage")
|
||||
if usage is not None:
|
||||
input_tokens = getattr(usage, "input_tokens", 0) or 0
|
||||
output_tokens = getattr(usage, "output_tokens", 0) or 0
|
||||
cache_creation_tokens = getattr(usage, "input_tokens_details", None) # type: ignore[assignment]
|
||||
cache_read_tokens = getattr(usage, "output_tokens_details", None) # type: ignore[assignment]
|
||||
# Prefer direct cache fields if present
|
||||
cache_creation_tokens = int(getattr(usage, "cache_creation_input_tokens", 0) or 0)
|
||||
cache_read_tokens = int(getattr(usage, "cache_read_input_tokens", 0) or 0)
|
||||
input_tokens = _get_field(usage, "input_tokens", 0) or 0
|
||||
output_tokens = _get_field(usage, "output_tokens", 0) or 0
|
||||
cache_creation_tokens = int(_get_field(usage, "cache_creation_input_tokens", 0) or 0)
|
||||
cache_read_tokens = int(_get_field(usage, "cache_read_input_tokens", 0) or 0)
|
||||
|
||||
# Check if tool_use was in the output to override stop_reason
|
||||
if response_obj is not None:
|
||||
output = getattr(response_obj, "output", []) or []
|
||||
output = _get_field(response_obj, "output", []) or []
|
||||
for out_item in output:
|
||||
out_type = getattr(out_item, "type", None) or (
|
||||
out_item.get("type") if isinstance(out_item, dict) else None
|
||||
|
|
|
|||
|
|
@ -5,10 +5,9 @@ Tests for AnthropicResponsesStreamWrapper
|
|||
|
||||
import os
|
||||
import sys
|
||||
from types import SimpleNamespace
|
||||
|
||||
sys.path.insert(
|
||||
0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../.."))
|
||||
)
|
||||
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../../..")))
|
||||
|
||||
from litellm.llms.anthropic.experimental_pass_through.responses_adapters.streaming_iterator import (
|
||||
AnthropicResponsesStreamWrapper,
|
||||
|
|
@ -77,3 +76,66 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded:
|
|||
("content_block_start", 0),
|
||||
("content_block_delta", 0),
|
||||
]
|
||||
|
||||
|
||||
class TestDictShapedCompletedEvents:
|
||||
"""Usage, status, and output must be read from dict-shaped
|
||||
`response.completed` payloads, not only attribute-shaped ones.
|
||||
|
||||
Previously this branch used getattr-only access, so dict-shaped events
|
||||
always produced usage 0/0 (disabling spend tracking and TPM enforcement,
|
||||
https://github.com/BerriAI/litellm/issues/32086) and mapped
|
||||
`response.incomplete` to end_turn instead of max_tokens.
|
||||
"""
|
||||
|
||||
def test_dict_usage_is_extracted(self):
|
||||
chunks = _process_all(
|
||||
[
|
||||
{
|
||||
"type": "response.completed",
|
||||
"response": {
|
||||
"status": "completed",
|
||||
"usage": {
|
||||
"input_tokens": 11,
|
||||
"output_tokens": 42,
|
||||
"cache_read_input_tokens": 7,
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
assert chunks[0]["type"] == "message_delta"
|
||||
assert chunks[0]["usage"] == {
|
||||
"input_tokens": 11,
|
||||
"output_tokens": 42,
|
||||
"cache_read_input_tokens": 7,
|
||||
}
|
||||
|
||||
def test_dict_incomplete_maps_to_max_tokens(self):
|
||||
chunks = _process_all([{"type": "response.incomplete", "response": {"status": "incomplete"}}])
|
||||
assert chunks[0]["delta"]["stop_reason"] == "max_tokens"
|
||||
|
||||
def test_dict_function_call_output_maps_to_tool_use(self):
|
||||
chunks = _process_all(
|
||||
[
|
||||
{
|
||||
"type": "response.completed",
|
||||
"response": {
|
||||
"status": "completed",
|
||||
"output": [{"type": "function_call", "name": "get_weather"}],
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
assert chunks[0]["delta"]["stop_reason"] == "tool_use"
|
||||
|
||||
def test_object_shaped_usage_still_extracted(self):
|
||||
usage = SimpleNamespace(
|
||||
input_tokens=3,
|
||||
output_tokens=5,
|
||||
cache_creation_input_tokens=0,
|
||||
cache_read_input_tokens=0,
|
||||
)
|
||||
response = SimpleNamespace(status="completed", usage=usage, output=[])
|
||||
chunks = _process_all([{"type": "response.completed", "response": response}])
|
||||
assert chunks[0]["usage"] == {"input_tokens": 3, "output_tokens": 5}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue