[Bug Fix] Fix 'dict' object has no attribute 'usage' crash dropping spend logs for streaming /v1/responses

On the streaming /v1/responses path, the ResponseCompletedEvent stored in the
success-logging callback has its .response field set to a plain dict at runtime
(the base response models allow extra fields). _get_assembled_streaming_response
assumed result.response was an object and accessed result.response.usage, which
raised 'dict' object has no attribute 'usage'. The exception propagated before
cost was computed, so streamed responses requests were silently not charged (no
LiteLLM_SpendLogs row), while still returning 200 to the client.

Read usage defensively (dict or object) and handle a dict-shaped Response API
usage, mirroring the existing usage-normalization helpers in this file. Add a
regression test.

Fixes #29913.

Signed-off-by: sherry255 <11342665+sherry255@users.noreply.github.com>
This commit is contained in:
sherry255 2026-06-09 01:24:34 +08:00
parent 35f6961526
commit 9d9bfcec4d
2 changed files with 66 additions and 11 deletions

View file

@ -3404,23 +3404,37 @@ class Logging(LiteLLMLoggingBaseClass):
(ResponseCompletedEvent, ResponseIncompleteEvent, ResponseFailedEvent),
):
## return unified Usage object
if isinstance(result.response.usage, ResponseAPIUsage):
# On the streaming /v1/responses path ``result.response`` may be a
# plain dict (the base response models allow extra fields), so read
# ``usage`` defensively instead of assuming an object attribute,
# which otherwise raises ``'dict' object has no attribute 'usage'``
# and drops the spend log for streamed responses (#29913).
response = result.response
usage = (
response.get("usage")
if isinstance(response, dict)
else getattr(response, "usage", None)
)
if isinstance(usage, ResponseAPIUsage) or (
isinstance(usage, dict)
and ResponseAPILoggingUtils._is_response_api_usage(usage)
):
transformed_usage = (
ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
result.response.usage
usage
)
)
# Set as dict instead of Usage object so model_dump() serializes it correctly
setattr(
result.response,
"usage",
(
transformed_usage.model_dump()
if hasattr(transformed_usage, "model_dump")
else dict(transformed_usage)
),
new_usage = (
transformed_usage.model_dump()
if hasattr(transformed_usage, "model_dump")
else dict(transformed_usage)
)
return result.response
if isinstance(response, dict):
response["usage"] = new_usage
else:
setattr(response, "usage", new_usage)
return response
else:
return None

View file

@ -36,6 +36,47 @@ def test_get_masked_api_base(logging_obj):
assert type(masked_api_base) == str
def test_get_assembled_streaming_response_dict_response_usage(logging_obj):
"""Regression test for #29913.
On the streaming ``/v1/responses`` path ``result.response`` is stored as a
plain dict at runtime, so ``_get_assembled_streaming_response`` must read
``usage`` defensively rather than assuming an object attribute. Previously
this raised ``'dict' object has no attribute 'usage'``, which crashed the
success-logging callback and dropped the spend log for streamed responses.
"""
import datetime
from litellm.types.llms.openai import (
ResponseCompletedEvent,
ResponsesAPIStreamEvents,
)
event = ResponseCompletedEvent.model_construct(
type=ResponsesAPIStreamEvents.RESPONSE_COMPLETED,
response={
"usage": {"input_tokens": 10, "output_tokens": 5, "total_tokens": 15}
},
)
now = datetime.datetime.now()
result = logging_obj._get_assembled_streaming_response(
result=event,
start_time=now,
end_time=now,
is_async=True,
streaming_chunks=[],
)
assert result is not None
# Response API usage ({input_tokens, output_tokens}) is normalized to the
# unified chat usage shape so cost can be computed downstream.
usage = result["usage"]
assert usage["prompt_tokens"] == 10
assert usage["completion_tokens"] == 5
assert usage["total_tokens"] == 15
def test_post_call_serializes_dict_with_datetime(logging_obj):
import datetime