From e8a7116899b6045adf88ca29a50d81be8e4962e8 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 7 Mar 2026 16:18:51 -0800 Subject: [PATCH] fix(tests): fix repeating chunk and audio usage streaming tests (#23061) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Replace ModelResponse(stream=True) with ModelResponseStream in test_unit_test_custom_stream_wrapper_repeating_chunk — stream=True stores delta as a plain dict causing AttributeError in CustomStreamWrapper - Accept MidStreamFallbackError alongside InternalServerError in the repeating-chunk safety check assertion - Add @pytest.mark.flaky(retries=3) to the live OpenAI audio output usage test --- .../test_stream_chunk_builder.py | 1 + tests/local_testing/test_streaming.py | 30 ++++++++----------- 2 files changed, 14 insertions(+), 17 deletions(-) diff --git a/tests/local_testing/test_stream_chunk_builder.py b/tests/local_testing/test_stream_chunk_builder.py index ddb1546097c..e5d909812c1 100644 --- a/tests/local_testing/test_stream_chunk_builder.py +++ b/tests/local_testing/test_stream_chunk_builder.py @@ -636,6 +636,7 @@ def test_stream_chunk_builder_openai_prompt_caching(): assert response_usage_value == v +@pytest.mark.flaky(retries=3, delay=2) def test_stream_chunk_builder_openai_audio_output_usage(): from pydantic import BaseModel from openai import OpenAI diff --git a/tests/local_testing/test_streaming.py b/tests/local_testing/test_streaming.py index bbeaacccb00..ef2f89cdaf5 100644 --- a/tests/local_testing/test_streaming.py +++ b/tests/local_testing/test_streaming.py @@ -3075,22 +3075,18 @@ def test_unit_test_custom_stream_wrapper_repeating_chunk( """ litellm.set_verbose = False chunks = [ - litellm.ModelResponse( - **{ - "id": "chatcmpl-123", - "object": "chat.completion.chunk", - "created": 1694268190, - "model": "gpt-3.5-turbo-0125", - "system_fingerprint": "fp_44709d6fcb", - "choices": [ - { - "index": 0, - "delta": {"content": chunk_value}, - "finish_reason": "stop", - } - ], - }, - stream=True, + litellm.ModelResponseStream( + id="chatcmpl-123", + created=1694268190, + model="gpt-3.5-turbo-0125", + system_fingerprint="fp_44709d6fcb", + choices=[ + { + "index": 0, + "delta": {"content": chunk_value}, + "finish_reason": "stop", + } + ], ) ] * loop_amount completion_stream = ModelResponseListIterator(model_responses=chunks) @@ -3113,7 +3109,7 @@ def test_unit_test_custom_stream_wrapper_repeating_chunk( print(f"expected_chunk_fail: {expected_chunk_fail}") if (loop_amount > litellm.REPEATED_STREAMING_CHUNK_LIMIT) and expected_chunk_fail: - with pytest.raises(litellm.InternalServerError): + with pytest.raises((litellm.InternalServerError, litellm.exceptions.MidStreamFallbackError)): for chunk in response: continue else: