mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
[Bug Fix] Fix 'dict' object has no attribute 'usage' crash dropping spend logs for streaming /v1/responses
On the streaming /v1/responses path, the ResponseCompletedEvent stored in the success-logging callback has its .response field set to a plain dict at runtime (the base response models allow extra fields). _get_assembled_streaming_response assumed result.response was an object and accessed result.response.usage, which raised 'dict' object has no attribute 'usage'. The exception propagated before cost was computed, so streamed responses requests were silently not charged (no LiteLLM_SpendLogs row), while still returning 200 to the client. Read usage defensively (dict or object) and handle a dict-shaped Response API usage, mirroring the existing usage-normalization helpers in this file. Add a regression test. Fixes #29913. Signed-off-by: sherry255 <11342665+sherry255@users.noreply.github.com>
This commit is contained in:
parent
35f6961526
commit
9d9bfcec4d
2 changed files with 66 additions and 11 deletions
|
|
@ -3404,23 +3404,37 @@ class Logging(LiteLLMLoggingBaseClass):
|
|||
(ResponseCompletedEvent, ResponseIncompleteEvent, ResponseFailedEvent),
|
||||
):
|
||||
## return unified Usage object
|
||||
if isinstance(result.response.usage, ResponseAPIUsage):
|
||||
# On the streaming /v1/responses path ``result.response`` may be a
|
||||
# plain dict (the base response models allow extra fields), so read
|
||||
# ``usage`` defensively instead of assuming an object attribute,
|
||||
# which otherwise raises ``'dict' object has no attribute 'usage'``
|
||||
# and drops the spend log for streamed responses (#29913).
|
||||
response = result.response
|
||||
usage = (
|
||||
response.get("usage")
|
||||
if isinstance(response, dict)
|
||||
else getattr(response, "usage", None)
|
||||
)
|
||||
if isinstance(usage, ResponseAPIUsage) or (
|
||||
isinstance(usage, dict)
|
||||
and ResponseAPILoggingUtils._is_response_api_usage(usage)
|
||||
):
|
||||
transformed_usage = (
|
||||
ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
|
||||
result.response.usage
|
||||
usage
|
||||
)
|
||||
)
|
||||
# Set as dict instead of Usage object so model_dump() serializes it correctly
|
||||
setattr(
|
||||
result.response,
|
||||
"usage",
|
||||
(
|
||||
transformed_usage.model_dump()
|
||||
if hasattr(transformed_usage, "model_dump")
|
||||
else dict(transformed_usage)
|
||||
),
|
||||
new_usage = (
|
||||
transformed_usage.model_dump()
|
||||
if hasattr(transformed_usage, "model_dump")
|
||||
else dict(transformed_usage)
|
||||
)
|
||||
return result.response
|
||||
if isinstance(response, dict):
|
||||
response["usage"] = new_usage
|
||||
else:
|
||||
setattr(response, "usage", new_usage)
|
||||
return response
|
||||
else:
|
||||
return None
|
||||
|
||||
|
|
|
|||
|
|
@ -36,6 +36,47 @@ def test_get_masked_api_base(logging_obj):
|
|||
assert type(masked_api_base) == str
|
||||
|
||||
|
||||
def test_get_assembled_streaming_response_dict_response_usage(logging_obj):
|
||||
"""Regression test for #29913.
|
||||
|
||||
On the streaming ``/v1/responses`` path ``result.response`` is stored as a
|
||||
plain dict at runtime, so ``_get_assembled_streaming_response`` must read
|
||||
``usage`` defensively rather than assuming an object attribute. Previously
|
||||
this raised ``'dict' object has no attribute 'usage'``, which crashed the
|
||||
success-logging callback and dropped the spend log for streamed responses.
|
||||
"""
|
||||
import datetime
|
||||
|
||||
from litellm.types.llms.openai import (
|
||||
ResponseCompletedEvent,
|
||||
ResponsesAPIStreamEvents,
|
||||
)
|
||||
|
||||
event = ResponseCompletedEvent.model_construct(
|
||||
type=ResponsesAPIStreamEvents.RESPONSE_COMPLETED,
|
||||
response={
|
||||
"usage": {"input_tokens": 10, "output_tokens": 5, "total_tokens": 15}
|
||||
},
|
||||
)
|
||||
|
||||
now = datetime.datetime.now()
|
||||
result = logging_obj._get_assembled_streaming_response(
|
||||
result=event,
|
||||
start_time=now,
|
||||
end_time=now,
|
||||
is_async=True,
|
||||
streaming_chunks=[],
|
||||
)
|
||||
|
||||
assert result is not None
|
||||
# Response API usage ({input_tokens, output_tokens}) is normalized to the
|
||||
# unified chat usage shape so cost can be computed downstream.
|
||||
usage = result["usage"]
|
||||
assert usage["prompt_tokens"] == 10
|
||||
assert usage["completion_tokens"] == 5
|
||||
assert usage["total_tokens"] == 15
|
||||
|
||||
|
||||
def test_post_call_serializes_dict_with_datetime(logging_obj):
|
||||
import datetime
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue