refactor(litellm_logging.py): delegate returning a complete response to the streaming_handler

Removes incorrect logic for calculating complete streaming response from litellm logging
This commit is contained in:
Krrish Dholakia 2025-03-15 09:55:33 -07:00
parent cc82d42d25
commit 612d5a284d
3 changed files with 6 additions and 16 deletions

View file

@ -2351,18 +2351,6 @@ class Logging(LiteLLMLoggingBaseClass):
return result
elif isinstance(result, ResponseCompletedEvent):
return result.response
elif isinstance(result, ModelResponseStream):
complete_streaming_response: Optional[
Union[ModelResponse, TextCompletionResponse]
] = _assemble_complete_response_from_streaming_chunks(
result=result,
start_time=start_time,
end_time=end_time,
request_kwargs=self.model_call_details,
streaming_chunks=streaming_chunks,
is_async=is_async,
)
return complete_streaming_response
return None
def _handle_anthropic_messages_response_logging(self, result: Any) -> ModelResponse:

View file

@ -77,7 +77,7 @@ def _assemble_complete_response_from_streaming_chunks(
complete_streaming_response: Optional[
Union[ModelResponse, TextCompletionResponse]
] = None
if result.choices[0].finish_reason is not None: # if it's the last chunk
if getattr(result, "usage", None) is not None: # if it's the last chunk
streaming_chunks.append(result)
try:
complete_streaming_response = litellm.stream_chunk_builder(

View file

@ -1030,6 +1030,8 @@ def test_streaming_handler_with_usage():
),
)
for chunk in response:
if hasattr(chunk, "usage"):
assert chunk.usage == final_usage_block
with patch("litellm.main.token_counter") as mock_token_counter:
for chunk in response:
if hasattr(chunk, "usage"):
assert chunk.usage == final_usage_block
assert mock_token_counter.assert_not_called()