From 612d5a284d46b78db87ce71ef8254d015c5aa218 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Sat, 15 Mar 2025 09:55:33 -0700 Subject: [PATCH] refactor(litellm_logging.py): delegate returning a complete response to the streaming_handler Removes incorrect logic for calculating complete streaming response from litellm logging --- litellm/litellm_core_utils/litellm_logging.py | 12 ------------ litellm/litellm_core_utils/logging_utils.py | 2 +- .../litellm_core_utils/test_streaming_handler.py | 8 +++++--- 3 files changed, 6 insertions(+), 16 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index a369b7f3e36..6b7dc4ced47 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -2351,18 +2351,6 @@ class Logging(LiteLLMLoggingBaseClass): return result elif isinstance(result, ResponseCompletedEvent): return result.response - elif isinstance(result, ModelResponseStream): - complete_streaming_response: Optional[ - Union[ModelResponse, TextCompletionResponse] - ] = _assemble_complete_response_from_streaming_chunks( - result=result, - start_time=start_time, - end_time=end_time, - request_kwargs=self.model_call_details, - streaming_chunks=streaming_chunks, - is_async=is_async, - ) - return complete_streaming_response return None def _handle_anthropic_messages_response_logging(self, result: Any) -> ModelResponse: diff --git a/litellm/litellm_core_utils/logging_utils.py b/litellm/litellm_core_utils/logging_utils.py index 6782435af62..c2d959b3c08 100644 --- a/litellm/litellm_core_utils/logging_utils.py +++ b/litellm/litellm_core_utils/logging_utils.py @@ -77,7 +77,7 @@ def _assemble_complete_response_from_streaming_chunks( complete_streaming_response: Optional[ Union[ModelResponse, TextCompletionResponse] ] = None - if result.choices[0].finish_reason is not None: # if it's the last chunk + if getattr(result, "usage", None) is not None: # if it's the last chunk streaming_chunks.append(result) try: complete_streaming_response = litellm.stream_chunk_builder( diff --git a/tests/litellm/litellm_core_utils/test_streaming_handler.py b/tests/litellm/litellm_core_utils/test_streaming_handler.py index 76de59b57b8..19948b25dc7 100644 --- a/tests/litellm/litellm_core_utils/test_streaming_handler.py +++ b/tests/litellm/litellm_core_utils/test_streaming_handler.py @@ -1030,6 +1030,8 @@ def test_streaming_handler_with_usage(): ), ) - for chunk in response: - if hasattr(chunk, "usage"): - assert chunk.usage == final_usage_block + with patch("litellm.main.token_counter") as mock_token_counter: + for chunk in response: + if hasattr(chunk, "usage"): + assert chunk.usage == final_usage_block + assert mock_token_counter.assert_not_called()