This commit is contained in:
devin-ai-integration[bot] 2026-08-27 18:41:50 -05:00 • committed by GitHub
commit 98b749df0f
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 158 additions and 3 deletions

View file

@ -561,7 +561,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
will_merge_into_held = (
self.holding_stop_reason_chunk is not None and getattr(chunk, "usage", None) is not None
)
is_final_chunk = chunk.choices[0].finish_reason is not None
is_final_chunk = bool(chunk.choices) and chunk.choices[0].finish_reason is not None
processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic(
response=chunk,
current_content_block_index=self.current_content_block_index,
@ -795,7 +795,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
will_merge_into_held = (
self.holding_stop_reason_chunk is not None and getattr(chunk, "usage", None) is not None
)
is_final_chunk = chunk.choices[0].finish_reason is not None
is_final_chunk = bool(chunk.choices) and chunk.choices[0].finish_reason is not None
processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic(
response=chunk,
current_content_block_index=self.current_content_block_index,
@ -1029,6 +1029,8 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
@staticmethod
def _is_blank_delta(chunk: "ModelResponseStream") -> bool:
if not chunk.choices:
return True
choice: Final = chunk.choices[0]
if choice.finish_reason is not None:
return False
@ -1057,6 +1059,8 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
# Example logic - customize based on your needs:
# If chunk indicates a tool call
if not chunk.choices:
return False
if chunk.choices[0].finish_reason is not None:
return False

View file

@ -1570,7 +1570,7 @@ class LiteLLMAnthropicMessagesAdapter:
applied_edits: list[AppliedEdit] | None = None,
) -> ContentBlockDelta | MessageBlockDelta:
## base case - final chunk w/ finish reason
if response.choices[0].finish_reason is not None:
if response.choices and response.choices[0].finish_reason is not None:
delta: Final = MessageDelta(
stop_reason=self._translate_openai_finish_reason_to_anthropic(response.choices[0].finish_reason),
)

View file

@ -160,6 +160,8 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
return fn_name, None
def _is_reasoning_end(self, chunk):
if not chunk.choices:
return False
delta: Final = chunk.choices[0].delta
# if this indicates reasoning content, don't consider reasoning ended
@ -818,6 +820,8 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
# Change: Never return a value, just enqueue output item events
if self.sent_output_item_added_event:
return
if not chunk.choices:
return
delta: Final = chunk.choices[0].delta
self._sequence_number += 1
@ -1145,6 +1149,8 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
It's unclear how users expect litellm to translate multiple-choices-per-chunk to the responses API output.
"""
if not choices:
return ""
choice: Final = choices[0]
chat_completion_delta: Final[ChatCompletionDelta] = choice.delta
return chat_completion_delta.content or ""

View file

@ -143,6 +143,60 @@ def test_delayed_usage_chunk_preserves_cache_tokens():
assert message_delta["usage"]["cache_creation_input_tokens"] == 20
def test_trailing_empty_choices_usage_chunk_emits_message_delta_usage():
"""Regression for LIT-4767.
The trailing usage-only chunk an OpenAI-compatible provider sends when
``include_usage`` is set has ``choices: []``. The adapter used to index
``choices[0]`` unguarded (``is_final_chunk`` / ``_should_start_new_content_block``)
and crash with IndexError. It must instead merge the usage into the held
stop-reason chunk so ``message_delta`` still reports it.
"""
chunks = [
ModelResponseStream(
choices=[StreamingChoices(index=0, delta=Delta(content="Two."), finish_reason=None)],
),
ModelResponseStream(
choices=[StreamingChoices(index=0, delta=Delta(), finish_reason="stop")],
),
ModelResponseStream(
choices=[],
usage=Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15),
),
]
wrapper = AnthropicStreamWrapper(completion_stream=iter(chunks), model="gpt-4o")
events = list(wrapper)
message_delta = next(event for event in events if event.get("type") == "message_delta")
assert message_delta["usage"]["input_tokens"] == 10
assert message_delta["usage"]["output_tokens"] == 5
def test_leading_empty_choices_chunk_does_not_crash_stream():
"""Azure emits a leading ``prompt_filter_results`` chunk with ``choices: []``
before any content. It must be tolerated and the following content emitted."""
chunks = [
ModelResponseStream(choices=[]),
ModelResponseStream(
choices=[StreamingChoices(index=0, delta=Delta(content="Hi"), finish_reason=None)],
),
ModelResponseStream(
choices=[StreamingChoices(index=0, delta=Delta(), finish_reason="stop")],
usage=Usage(prompt_tokens=3, completion_tokens=1, total_tokens=4),
),
]
async def _aiter() -> "AsyncIterator[ModelResponseStream]":
for chunk in chunks:
yield chunk
wrapper = AnthropicStreamWrapper(completion_stream=_aiter(), model="gpt-4o")
sse = _collect_async(wrapper)
assert "Hi" in sse
assert "message_stop" in sse
def test_splitter_passes_through_non_combined_chunks():
"""A chunk with content but no finish_reason is not split."""
chunk = ModelResponseStream(

View file

@ -0,0 +1,91 @@
"""
Regression tests for LIT-4767.
When an upstream OpenAI-compatible provider emits a chunk with ``choices: []``
(the trailing usage-only chunk every provider sends when ``include_usage`` is
set, or Azure's leading ``prompt_filter_results`` chunk), the Responses bridge
iterator used to index ``choices[0]`` unguarded and die with
``IndexError: list index out of range``, killing the whole stream.
The empty-choices chunk must be tolerated without crashing, and the usage it
carries must still reach ``response.completed``.
"""
from unittest.mock import AsyncMock
from litellm.responses.litellm_completion_transformation.streaming_iterator import (
LiteLLMCompletionStreamingIterator,
)
from litellm.types.llms.openai import ResponsesAPIStreamEvents
from litellm.types.utils import (
Delta,
ModelResponseStream,
StreamingChoices,
Usage,
)
def _iterator() -> LiteLLMCompletionStreamingIterator:
return LiteLLMCompletionStreamingIterator(
model="gpt-4o",
litellm_custom_stream_wrapper=AsyncMock(),
request_input="hi",
responses_api_request={},
custom_llm_provider="openai",
)
def _empty_choices_usage_chunk() -> ModelResponseStream:
chunk = ModelResponseStream(id="chunk-usage", model="gpt-4o", choices=[])
chunk.usage = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15)
return chunk
def test_ensure_output_item_for_empty_choices_chunk_does_not_crash():
"""First chunk with no choices must not raise (traceback frame in the ticket)."""
iterator = _iterator()
# Would raise IndexError before the fix.
assert iterator._ensure_output_item_for_chunk(_empty_choices_usage_chunk()) is None
assert iterator.sent_output_item_added_event is False
def test_transform_empty_choices_chunk_returns_no_delta():
"""The mid/trailing usage chunk flows through transform without crashing."""
iterator = _iterator()
# Would raise IndexError in _get_delta_string_from_streaming_choices before the fix.
assert iterator._transform_chat_completion_chunk_to_response_api_chunk(_empty_choices_usage_chunk()) is None
def test_is_reasoning_end_false_for_empty_choices_chunk():
iterator = _iterator()
assert iterator._is_reasoning_end(_empty_choices_usage_chunk()) is False
def test_empty_choices_usage_chunk_still_reaches_response_completed():
"""End-to-end: a text chunk followed by a choices=[] usage chunk must emit
response.completed carrying the usage rather than dying mid-stream."""
class _SyncWrapper:
def __init__(self, chunks):
self._it = iter(chunks)
self.logging_obj = None
self.stream_options = {"include_usage": True}
def __next__(self):
return next(self._it)
text_chunk = ModelResponseStream(
id="chunk-1",
model="gpt-4o",
choices=[StreamingChoices(index=0, delta=Delta(role="assistant", content="Hi"), finish_reason=None)],
)
iterator = _iterator()
iterator.litellm_logging_obj = None
iterator.litellm_custom_stream_wrapper = _SyncWrapper([text_chunk, _empty_choices_usage_chunk()])
events = list(iterator)
completed = [e for e in events if getattr(e, "type", None) == ResponsesAPIStreamEvents.RESPONSE_COMPLETED]
assert len(completed) == 1
assert completed[0].response.usage is not None
assert completed[0].response.usage.total_tokens == 15