mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Merge 73488599f0 into 49affa7c01
This commit is contained in:
commit
98b749df0f
5 changed files with 158 additions and 3 deletions
|
|
@ -561,7 +561,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
will_merge_into_held = (
|
||||
self.holding_stop_reason_chunk is not None and getattr(chunk, "usage", None) is not None
|
||||
)
|
||||
is_final_chunk = chunk.choices[0].finish_reason is not None
|
||||
is_final_chunk = bool(chunk.choices) and chunk.choices[0].finish_reason is not None
|
||||
processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic(
|
||||
response=chunk,
|
||||
current_content_block_index=self.current_content_block_index,
|
||||
|
|
@ -795,7 +795,7 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
will_merge_into_held = (
|
||||
self.holding_stop_reason_chunk is not None and getattr(chunk, "usage", None) is not None
|
||||
)
|
||||
is_final_chunk = chunk.choices[0].finish_reason is not None
|
||||
is_final_chunk = bool(chunk.choices) and chunk.choices[0].finish_reason is not None
|
||||
processed_chunk = LiteLLMAnthropicMessagesAdapter().translate_streaming_openai_response_to_anthropic(
|
||||
response=chunk,
|
||||
current_content_block_index=self.current_content_block_index,
|
||||
|
|
@ -1029,6 +1029,8 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
|
||||
@staticmethod
|
||||
def _is_blank_delta(chunk: "ModelResponseStream") -> bool:
|
||||
if not chunk.choices:
|
||||
return True
|
||||
choice: Final = chunk.choices[0]
|
||||
if choice.finish_reason is not None:
|
||||
return False
|
||||
|
|
@ -1057,6 +1059,8 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
|
||||
# Example logic - customize based on your needs:
|
||||
# If chunk indicates a tool call
|
||||
if not chunk.choices:
|
||||
return False
|
||||
if chunk.choices[0].finish_reason is not None:
|
||||
return False
|
||||
|
||||
|
|
|
|||
|
|
@ -1570,7 +1570,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
applied_edits: list[AppliedEdit] | None = None,
|
||||
) -> ContentBlockDelta | MessageBlockDelta:
|
||||
## base case - final chunk w/ finish reason
|
||||
if response.choices[0].finish_reason is not None:
|
||||
if response.choices and response.choices[0].finish_reason is not None:
|
||||
delta: Final = MessageDelta(
|
||||
stop_reason=self._translate_openai_finish_reason_to_anthropic(response.choices[0].finish_reason),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -160,6 +160,8 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
|
|||
return fn_name, None
|
||||
|
||||
def _is_reasoning_end(self, chunk):
|
||||
if not chunk.choices:
|
||||
return False
|
||||
delta: Final = chunk.choices[0].delta
|
||||
|
||||
# if this indicates reasoning content, don't consider reasoning ended
|
||||
|
|
@ -818,6 +820,8 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
|
|||
# Change: Never return a value, just enqueue output item events
|
||||
if self.sent_output_item_added_event:
|
||||
return
|
||||
if not chunk.choices:
|
||||
return
|
||||
delta: Final = chunk.choices[0].delta
|
||||
|
||||
self._sequence_number += 1
|
||||
|
|
@ -1145,6 +1149,8 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
|
|||
|
||||
It's unclear how users expect litellm to translate multiple-choices-per-chunk to the responses API output.
|
||||
"""
|
||||
if not choices:
|
||||
return ""
|
||||
choice: Final = choices[0]
|
||||
chat_completion_delta: Final[ChatCompletionDelta] = choice.delta
|
||||
return chat_completion_delta.content or ""
|
||||
|
|
|
|||
|
|
@ -143,6 +143,60 @@ def test_delayed_usage_chunk_preserves_cache_tokens():
|
|||
assert message_delta["usage"]["cache_creation_input_tokens"] == 20
|
||||
|
||||
|
||||
def test_trailing_empty_choices_usage_chunk_emits_message_delta_usage():
|
||||
"""Regression for LIT-4767.
|
||||
|
||||
The trailing usage-only chunk an OpenAI-compatible provider sends when
|
||||
``include_usage`` is set has ``choices: []``. The adapter used to index
|
||||
``choices[0]`` unguarded (``is_final_chunk`` / ``_should_start_new_content_block``)
|
||||
and crash with IndexError. It must instead merge the usage into the held
|
||||
stop-reason chunk so ``message_delta`` still reports it.
|
||||
"""
|
||||
chunks = [
|
||||
ModelResponseStream(
|
||||
choices=[StreamingChoices(index=0, delta=Delta(content="Two."), finish_reason=None)],
|
||||
),
|
||||
ModelResponseStream(
|
||||
choices=[StreamingChoices(index=0, delta=Delta(), finish_reason="stop")],
|
||||
),
|
||||
ModelResponseStream(
|
||||
choices=[],
|
||||
usage=Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15),
|
||||
),
|
||||
]
|
||||
wrapper = AnthropicStreamWrapper(completion_stream=iter(chunks), model="gpt-4o")
|
||||
events = list(wrapper)
|
||||
|
||||
message_delta = next(event for event in events if event.get("type") == "message_delta")
|
||||
assert message_delta["usage"]["input_tokens"] == 10
|
||||
assert message_delta["usage"]["output_tokens"] == 5
|
||||
|
||||
|
||||
def test_leading_empty_choices_chunk_does_not_crash_stream():
|
||||
"""Azure emits a leading ``prompt_filter_results`` chunk with ``choices: []``
|
||||
before any content. It must be tolerated and the following content emitted."""
|
||||
chunks = [
|
||||
ModelResponseStream(choices=[]),
|
||||
ModelResponseStream(
|
||||
choices=[StreamingChoices(index=0, delta=Delta(content="Hi"), finish_reason=None)],
|
||||
),
|
||||
ModelResponseStream(
|
||||
choices=[StreamingChoices(index=0, delta=Delta(), finish_reason="stop")],
|
||||
usage=Usage(prompt_tokens=3, completion_tokens=1, total_tokens=4),
|
||||
),
|
||||
]
|
||||
|
||||
async def _aiter() -> "AsyncIterator[ModelResponseStream]":
|
||||
for chunk in chunks:
|
||||
yield chunk
|
||||
|
||||
wrapper = AnthropicStreamWrapper(completion_stream=_aiter(), model="gpt-4o")
|
||||
sse = _collect_async(wrapper)
|
||||
|
||||
assert "Hi" in sse
|
||||
assert "message_stop" in sse
|
||||
|
||||
|
||||
def test_splitter_passes_through_non_combined_chunks():
|
||||
"""A chunk with content but no finish_reason is not split."""
|
||||
chunk = ModelResponseStream(
|
||||
|
|
|
|||
|
|
@ -0,0 +1,91 @@
|
|||
"""
|
||||
Regression tests for LIT-4767.
|
||||
|
||||
When an upstream OpenAI-compatible provider emits a chunk with ``choices: []``
|
||||
(the trailing usage-only chunk every provider sends when ``include_usage`` is
|
||||
set, or Azure's leading ``prompt_filter_results`` chunk), the Responses bridge
|
||||
iterator used to index ``choices[0]`` unguarded and die with
|
||||
``IndexError: list index out of range``, killing the whole stream.
|
||||
|
||||
The empty-choices chunk must be tolerated without crashing, and the usage it
|
||||
carries must still reach ``response.completed``.
|
||||
"""
|
||||
|
||||
from unittest.mock import AsyncMock
|
||||
|
||||
from litellm.responses.litellm_completion_transformation.streaming_iterator import (
|
||||
LiteLLMCompletionStreamingIterator,
|
||||
)
|
||||
from litellm.types.llms.openai import ResponsesAPIStreamEvents
|
||||
from litellm.types.utils import (
|
||||
Delta,
|
||||
ModelResponseStream,
|
||||
StreamingChoices,
|
||||
Usage,
|
||||
)
|
||||
|
||||
|
||||
def _iterator() -> LiteLLMCompletionStreamingIterator:
|
||||
return LiteLLMCompletionStreamingIterator(
|
||||
model="gpt-4o",
|
||||
litellm_custom_stream_wrapper=AsyncMock(),
|
||||
request_input="hi",
|
||||
responses_api_request={},
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
|
||||
def _empty_choices_usage_chunk() -> ModelResponseStream:
|
||||
chunk = ModelResponseStream(id="chunk-usage", model="gpt-4o", choices=[])
|
||||
chunk.usage = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15)
|
||||
return chunk
|
||||
|
||||
|
||||
def test_ensure_output_item_for_empty_choices_chunk_does_not_crash():
|
||||
"""First chunk with no choices must not raise (traceback frame in the ticket)."""
|
||||
iterator = _iterator()
|
||||
# Would raise IndexError before the fix.
|
||||
assert iterator._ensure_output_item_for_chunk(_empty_choices_usage_chunk()) is None
|
||||
assert iterator.sent_output_item_added_event is False
|
||||
|
||||
|
||||
def test_transform_empty_choices_chunk_returns_no_delta():
|
||||
"""The mid/trailing usage chunk flows through transform without crashing."""
|
||||
iterator = _iterator()
|
||||
# Would raise IndexError in _get_delta_string_from_streaming_choices before the fix.
|
||||
assert iterator._transform_chat_completion_chunk_to_response_api_chunk(_empty_choices_usage_chunk()) is None
|
||||
|
||||
|
||||
def test_is_reasoning_end_false_for_empty_choices_chunk():
|
||||
iterator = _iterator()
|
||||
assert iterator._is_reasoning_end(_empty_choices_usage_chunk()) is False
|
||||
|
||||
|
||||
def test_empty_choices_usage_chunk_still_reaches_response_completed():
|
||||
"""End-to-end: a text chunk followed by a choices=[] usage chunk must emit
|
||||
response.completed carrying the usage rather than dying mid-stream."""
|
||||
|
||||
class _SyncWrapper:
|
||||
def __init__(self, chunks):
|
||||
self._it = iter(chunks)
|
||||
self.logging_obj = None
|
||||
self.stream_options = {"include_usage": True}
|
||||
|
||||
def __next__(self):
|
||||
return next(self._it)
|
||||
|
||||
text_chunk = ModelResponseStream(
|
||||
id="chunk-1",
|
||||
model="gpt-4o",
|
||||
choices=[StreamingChoices(index=0, delta=Delta(role="assistant", content="Hi"), finish_reason=None)],
|
||||
)
|
||||
iterator = _iterator()
|
||||
iterator.litellm_logging_obj = None
|
||||
iterator.litellm_custom_stream_wrapper = _SyncWrapper([text_chunk, _empty_choices_usage_chunk()])
|
||||
|
||||
events = list(iterator)
|
||||
|
||||
completed = [e for e in events if getattr(e, "type", None) == ResponsesAPIStreamEvents.RESPONSE_COMPLETED]
|
||||
assert len(completed) == 1
|
||||
assert completed[0].response.usage is not None
|
||||
assert completed[0].response.usage.total_tokens == 15
|
||||
Loading…
Add table
Reference in a new issue