From 9d9bfcec4dccc4243beddb4e312e89402251aae6 Mon Sep 17 00:00:00 2001 From: sherry255 <11342665+sherry255@users.noreply.github.com> Date: Tue, 9 Jun 2026 01:24:34 +0800 Subject: [PATCH] [Bug Fix] Fix 'dict' object has no attribute 'usage' crash dropping spend logs for streaming /v1/responses On the streaming /v1/responses path, the ResponseCompletedEvent stored in the success-logging callback has its .response field set to a plain dict at runtime (the base response models allow extra fields). _get_assembled_streaming_response assumed result.response was an object and accessed result.response.usage, which raised 'dict' object has no attribute 'usage'. The exception propagated before cost was computed, so streamed responses requests were silently not charged (no LiteLLM_SpendLogs row), while still returning 200 to the client. Read usage defensively (dict or object) and handle a dict-shaped Response API usage, mirroring the existing usage-normalization helpers in this file. Add a regression test. Fixes #29913. Signed-off-by: sherry255 <11342665+sherry255@users.noreply.github.com> --- litellm/litellm_core_utils/litellm_logging.py | 36 +++++++++++----- .../test_litellm_logging.py | 41 +++++++++++++++++++ 2 files changed, 66 insertions(+), 11 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 2ab037afb0d..47b152d92c2 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -3404,23 +3404,37 @@ class Logging(LiteLLMLoggingBaseClass): (ResponseCompletedEvent, ResponseIncompleteEvent, ResponseFailedEvent), ): ## return unified Usage object - if isinstance(result.response.usage, ResponseAPIUsage): + # On the streaming /v1/responses path ``result.response`` may be a + # plain dict (the base response models allow extra fields), so read + # ``usage`` defensively instead of assuming an object attribute, + # which otherwise raises ``'dict' object has no attribute 'usage'`` + # and drops the spend log for streamed responses (#29913). + response = result.response + usage = ( + response.get("usage") + if isinstance(response, dict) + else getattr(response, "usage", None) + ) + if isinstance(usage, ResponseAPIUsage) or ( + isinstance(usage, dict) + and ResponseAPILoggingUtils._is_response_api_usage(usage) + ): transformed_usage = ( ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage( - result.response.usage + usage ) ) # Set as dict instead of Usage object so model_dump() serializes it correctly - setattr( - result.response, - "usage", - ( - transformed_usage.model_dump() - if hasattr(transformed_usage, "model_dump") - else dict(transformed_usage) - ), + new_usage = ( + transformed_usage.model_dump() + if hasattr(transformed_usage, "model_dump") + else dict(transformed_usage) ) - return result.response + if isinstance(response, dict): + response["usage"] = new_usage + else: + setattr(response, "usage", new_usage) + return response else: return None diff --git a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py index 07ab29c5231..43e5354d19a 100644 --- a/tests/test_litellm/litellm_core_utils/test_litellm_logging.py +++ b/tests/test_litellm/litellm_core_utils/test_litellm_logging.py @@ -36,6 +36,47 @@ def test_get_masked_api_base(logging_obj): assert type(masked_api_base) == str +def test_get_assembled_streaming_response_dict_response_usage(logging_obj): + """Regression test for #29913. + + On the streaming ``/v1/responses`` path ``result.response`` is stored as a + plain dict at runtime, so ``_get_assembled_streaming_response`` must read + ``usage`` defensively rather than assuming an object attribute. Previously + this raised ``'dict' object has no attribute 'usage'``, which crashed the + success-logging callback and dropped the spend log for streamed responses. + """ + import datetime + + from litellm.types.llms.openai import ( + ResponseCompletedEvent, + ResponsesAPIStreamEvents, + ) + + event = ResponseCompletedEvent.model_construct( + type=ResponsesAPIStreamEvents.RESPONSE_COMPLETED, + response={ + "usage": {"input_tokens": 10, "output_tokens": 5, "total_tokens": 15} + }, + ) + + now = datetime.datetime.now() + result = logging_obj._get_assembled_streaming_response( + result=event, + start_time=now, + end_time=now, + is_async=True, + streaming_chunks=[], + ) + + assert result is not None + # Response API usage ({input_tokens, output_tokens}) is normalized to the + # unified chat usage shape so cost can be computed downstream. + usage = result["usage"] + assert usage["prompt_tokens"] == 10 + assert usage["completion_tokens"] == 5 + assert usage["total_tokens"] == 15 + + def test_post_call_serializes_dict_with_datetime(logging_obj): import datetime