mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-25 01:02:15 +00:00
fix(streaming): handle empty choices array in streaming chunks
Some providers send streaming chunks with empty choices arrays (choices: []) as initialization or ping chunks. This causes IndexError crashes in two locations: 1. streaming_handler.py - raise_on_model_repetition() accesses choices[0] without checking if choices is non-empty. Fix: early return with counter reset when either of the last two chunks has empty choices. 2. streaming_chunk_builder_utils.py - build_base_response() assumes the first chunk with a 'choices' key has at least one element. Fix: require len(choices) > 0, use null-safe .get() access with fallback to role='assistant'. Added tests covering empty, mixed, and normal cases.
This commit is contained in:
parent
27de6547dc
commit
0793a3280e
3 changed files with 198 additions and 2 deletions
|
|
@ -119,8 +119,19 @@ class ChunkProcessor:
|
|||
model = ChunkProcessor._get_model_from_chunks(chunks, first_chunk_model)
|
||||
system_fingerprint = chunk.get("system_fingerprint", None)
|
||||
|
||||
first_chunk_with_choices = next((c for c in chunks if c.get("choices")), chunk)
|
||||
role = first_chunk_with_choices["choices"][0]["delta"]["role"]
|
||||
first_chunk_with_choices = next(
|
||||
(c for c in chunks if c.get("choices")),
|
||||
chunk,
|
||||
)
|
||||
if (
|
||||
first_chunk_with_choices.get("choices")
|
||||
and len(first_chunk_with_choices["choices"]) > 0
|
||||
):
|
||||
role = (first_chunk_with_choices["choices"][0].get("delta") or {}).get(
|
||||
"role", "assistant"
|
||||
)
|
||||
else:
|
||||
role = "assistant"
|
||||
finish_reason = "stop"
|
||||
for chunk in chunks:
|
||||
if "choices" in chunk and len(chunk["choices"]) > 0:
|
||||
|
|
|
|||
|
|
@ -269,6 +269,10 @@ class CustomStreamWrapper:
|
|||
if len(self.chunks) < 2:
|
||||
return
|
||||
|
||||
if not self.chunks[-1].choices or not self.chunks[-2].choices:
|
||||
self._repeated_messages_count = 1
|
||||
return
|
||||
|
||||
last_content = self.chunks[-1].choices[0].delta.content
|
||||
|
||||
if (
|
||||
|
|
|
|||
|
|
@ -0,0 +1,181 @@
|
|||
"""
|
||||
Test that streaming handles empty choices arrays gracefully.
|
||||
|
||||
Some providers send streaming chunks with empty choices arrays (choices: [])
|
||||
as initialization or ping chunks. LiteLLM should not crash when encountering
|
||||
these chunks.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../../.."))
|
||||
|
||||
from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper
|
||||
from litellm.litellm_core_utils.streaming_chunk_builder_utils import ChunkProcessor
|
||||
from litellm.types.utils import (
|
||||
Delta,
|
||||
ModelResponseStream,
|
||||
StreamingChoices,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def stream_wrapper() -> CustomStreamWrapper:
|
||||
return CustomStreamWrapper(
|
||||
completion_stream=None,
|
||||
model="test-model",
|
||||
logging_obj=MagicMock(),
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
|
||||
class TestRaiseOnModelRepetitionEmptyChoices:
|
||||
"""Test raise_on_model_repetition handles chunks with empty choices."""
|
||||
|
||||
def test_empty_choices_on_last_chunk(self, stream_wrapper):
|
||||
"""Should not crash when the last chunk has empty choices."""
|
||||
chunk_with_content = ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
index=0,
|
||||
delta=Delta(content="hello", role="assistant"),
|
||||
finish_reason=None,
|
||||
)
|
||||
]
|
||||
)
|
||||
chunk_empty_choices = ModelResponseStream(choices=[])
|
||||
|
||||
stream_wrapper.chunks = [chunk_with_content, chunk_empty_choices]
|
||||
stream_wrapper.raise_on_model_repetition()
|
||||
assert stream_wrapper._repeated_messages_count == 1
|
||||
|
||||
def test_empty_choices_on_second_to_last_chunk(self, stream_wrapper):
|
||||
"""Should not crash when the second-to-last chunk has empty choices."""
|
||||
chunk_empty_choices = ModelResponseStream(choices=[])
|
||||
chunk_with_content = ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
index=0,
|
||||
delta=Delta(content="hello", role="assistant"),
|
||||
finish_reason=None,
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
stream_wrapper.chunks = [chunk_empty_choices, chunk_with_content]
|
||||
stream_wrapper.raise_on_model_repetition()
|
||||
assert stream_wrapper._repeated_messages_count == 1
|
||||
|
||||
def test_both_chunks_empty_choices(self, stream_wrapper):
|
||||
"""Should not crash when both last chunks have empty choices."""
|
||||
chunk1 = ModelResponseStream(choices=[])
|
||||
chunk2 = ModelResponseStream(choices=[])
|
||||
|
||||
stream_wrapper.chunks = [chunk1, chunk2]
|
||||
stream_wrapper.raise_on_model_repetition()
|
||||
assert stream_wrapper._repeated_messages_count == 1
|
||||
|
||||
def test_normal_chunks_still_detect_repetition(self, stream_wrapper):
|
||||
"""Repetition detection still works for normal chunks with content."""
|
||||
chunk1 = ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
index=0,
|
||||
delta=Delta(content="repeated content here!", role="assistant"),
|
||||
finish_reason=None,
|
||||
)
|
||||
]
|
||||
)
|
||||
chunk2 = ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
index=0,
|
||||
delta=Delta(content="repeated content here!", role="assistant"),
|
||||
finish_reason=None,
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
stream_wrapper.chunks = [chunk1, chunk2]
|
||||
stream_wrapper.raise_on_model_repetition()
|
||||
assert stream_wrapper._repeated_messages_count == 2
|
||||
|
||||
def test_counter_resets_after_empty_chunk(self, stream_wrapper):
|
||||
"""Counter should reset to 1 after encountering empty choices."""
|
||||
# Simulate accumulated count
|
||||
stream_wrapper._repeated_messages_count = 5
|
||||
|
||||
chunk_with_content = ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
index=0,
|
||||
delta=Delta(content="hello", role="assistant"),
|
||||
finish_reason=None,
|
||||
)
|
||||
]
|
||||
)
|
||||
chunk_empty = ModelResponseStream(choices=[])
|
||||
|
||||
stream_wrapper.chunks = [chunk_with_content, chunk_empty]
|
||||
stream_wrapper.raise_on_model_repetition()
|
||||
assert stream_wrapper._repeated_messages_count == 1
|
||||
|
||||
|
||||
class TestBuildBaseResponseEmptyChoices:
|
||||
"""Test build_base_response handles chunks with empty choices."""
|
||||
|
||||
def test_all_chunks_have_empty_choices(self):
|
||||
"""Should fallback to role='assistant' when no chunk has valid choices."""
|
||||
chunks = [
|
||||
{
|
||||
"id": "1",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1234567890,
|
||||
"model": "test-model",
|
||||
"choices": [],
|
||||
},
|
||||
{
|
||||
"id": "2",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1234567890,
|
||||
"model": "test-model",
|
||||
"choices": [],
|
||||
},
|
||||
]
|
||||
|
||||
processor = ChunkProcessor(chunks=chunks, messages=None)
|
||||
response = processor.build_base_response(chunks)
|
||||
assert response["choices"][0]["message"]["role"] == "assistant"
|
||||
|
||||
def test_mixed_empty_and_valid_choices(self):
|
||||
"""Should find the first chunk with non-empty choices."""
|
||||
chunks = [
|
||||
{
|
||||
"id": "1",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1234567890,
|
||||
"model": "test-model",
|
||||
"choices": [],
|
||||
},
|
||||
{
|
||||
"id": "2",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 1234567890,
|
||||
"model": "test-model",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"role": "assistant", "content": "hi"},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
},
|
||||
]
|
||||
|
||||
processor = ChunkProcessor(chunks=chunks, messages=None)
|
||||
response = processor.build_base_response(chunks)
|
||||
assert response["choices"][0]["message"]["role"] == "assistant"
|
||||
Loading…
Add table
Reference in a new issue