mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-17 23:51:30 +00:00
test(responses): fold the empty-choices regression tests into the mapped streaming iterator test file
Some checks failed
ai-gateway image / ai-gateway release image (push) Has been cancelled
LiteLLM Rust / rust-lint (push) Has been cancelled
LiteLLM Rust / rust-test (push) Has been cancelled
Terraform Modules / fmt, validate, test (aws) (push) Has been cancelled
Terraform Modules / fmt, validate, test (gcp) (push) Has been cancelled
Terraform Provider / gofmt, vet, build, test (push) Has been cancelled
Terraform Provider / Provider endpoints vs proxy OpenAPI schema (push) Has been cancelled
Some checks failed
ai-gateway image / ai-gateway release image (push) Has been cancelled
LiteLLM Rust / rust-lint (push) Has been cancelled
LiteLLM Rust / rust-test (push) Has been cancelled
Terraform Modules / fmt, validate, test (aws) (push) Has been cancelled
Terraform Modules / fmt, validate, test (gcp) (push) Has been cancelled
Terraform Provider / gofmt, vet, build, test (push) Has been cancelled
Terraform Provider / Provider endpoints vs proxy OpenAPI schema (push) Has been cancelled
This commit is contained in:
parent
a6b0d4266f
commit
69e9edb91a
2 changed files with 43 additions and 91 deletions
|
|
@ -1,91 +0,0 @@
|
|||
"""
|
||||
Regression tests for LIT-4767.
|
||||
|
||||
When an upstream OpenAI-compatible provider emits a chunk with ``choices: []``
|
||||
(the trailing usage-only chunk every provider sends when ``include_usage`` is
|
||||
set, or Azure's leading ``prompt_filter_results`` chunk), the Responses bridge
|
||||
iterator used to index ``choices[0]`` unguarded and die with
|
||||
``IndexError: list index out of range``, killing the whole stream.
|
||||
|
||||
The empty-choices chunk must be tolerated without crashing, and the usage it
|
||||
carries must still reach ``response.completed``.
|
||||
"""
|
||||
|
||||
from unittest.mock import AsyncMock
|
||||
|
||||
from litellm.responses.litellm_completion_transformation.streaming_iterator import (
|
||||
LiteLLMCompletionStreamingIterator,
|
||||
)
|
||||
from litellm.types.llms.openai import ResponsesAPIStreamEvents
|
||||
from litellm.types.utils import (
|
||||
Delta,
|
||||
ModelResponseStream,
|
||||
StreamingChoices,
|
||||
Usage,
|
||||
)
|
||||
|
||||
|
||||
def _iterator() -> LiteLLMCompletionStreamingIterator:
|
||||
return LiteLLMCompletionStreamingIterator(
|
||||
model="gpt-4o",
|
||||
litellm_custom_stream_wrapper=AsyncMock(),
|
||||
request_input="hi",
|
||||
responses_api_request={},
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
|
||||
def _empty_choices_usage_chunk() -> ModelResponseStream:
|
||||
chunk = ModelResponseStream(id="chunk-usage", model="gpt-4o", choices=[])
|
||||
chunk.usage = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15)
|
||||
return chunk
|
||||
|
||||
|
||||
def test_ensure_output_item_for_empty_choices_chunk_does_not_crash():
|
||||
"""First chunk with no choices must not raise (traceback frame in the ticket)."""
|
||||
iterator = _iterator()
|
||||
# Would raise IndexError before the fix.
|
||||
assert iterator._ensure_output_item_for_chunk(_empty_choices_usage_chunk()) is None
|
||||
assert iterator.sent_output_item_added_event is False
|
||||
|
||||
|
||||
def test_transform_empty_choices_chunk_returns_no_delta():
|
||||
"""The mid/trailing usage chunk flows through transform without crashing."""
|
||||
iterator = _iterator()
|
||||
# Would raise IndexError in _get_delta_string_from_streaming_choices before the fix.
|
||||
assert iterator._transform_chat_completion_chunk_to_response_api_chunk(_empty_choices_usage_chunk()) is None
|
||||
|
||||
|
||||
def test_is_reasoning_end_false_for_empty_choices_chunk():
|
||||
iterator = _iterator()
|
||||
assert iterator._is_reasoning_end(_empty_choices_usage_chunk()) is False
|
||||
|
||||
|
||||
def test_empty_choices_usage_chunk_still_reaches_response_completed():
|
||||
"""End-to-end: a text chunk followed by a choices=[] usage chunk must emit
|
||||
response.completed carrying the usage rather than dying mid-stream."""
|
||||
|
||||
class _SyncWrapper:
|
||||
def __init__(self, chunks):
|
||||
self._it = iter(chunks)
|
||||
self.logging_obj = None
|
||||
self.stream_options = {"include_usage": True}
|
||||
|
||||
def __next__(self):
|
||||
return next(self._it)
|
||||
|
||||
text_chunk = ModelResponseStream(
|
||||
id="chunk-1",
|
||||
model="gpt-4o",
|
||||
choices=[StreamingChoices(index=0, delta=Delta(role="assistant", content="Hi"), finish_reason=None)],
|
||||
)
|
||||
iterator = _iterator()
|
||||
iterator.litellm_logging_obj = None
|
||||
iterator.litellm_custom_stream_wrapper = _SyncWrapper([text_chunk, _empty_choices_usage_chunk()])
|
||||
|
||||
events = list(iterator)
|
||||
|
||||
completed = [e for e in events if getattr(e, "type", None) == ResponsesAPIStreamEvents.RESPONSE_COMPLETED]
|
||||
assert len(completed) == 1
|
||||
assert completed[0].response.usage is not None
|
||||
assert completed[0].response.usage.total_tokens == 15
|
||||
|
|
@ -752,6 +752,49 @@ def test_completed_event_restores_usage_hidden_by_stream_options_none():
|
|||
assert completed.response.usage.output_tokens == 5
|
||||
|
||||
|
||||
def _empty_choices_chunk(usage: Usage | None = None) -> ModelResponseStream:
|
||||
return ModelResponseStream(id=CHAT_COMPLETION_ID, model="claude-haiku-4-5", choices=[], usage=usage)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_leading_empty_choices_chunk_does_not_kill_the_stream():
|
||||
"""
|
||||
Azure leads some streams with a `prompt_filter_results` chunk whose `choices` is empty.
|
||||
The bridge used to index `choices[0]` on it and die before the first token.
|
||||
"""
|
||||
iterator = _build_iterator([_empty_choices_chunk(), _chunk("Hello"), _chunk("!", finish_reason="stop")])
|
||||
|
||||
events = [event async for event in iterator]
|
||||
|
||||
event_types = [getattr(event, "type", None) for event in events]
|
||||
assert event_types.count(ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED) == 1
|
||||
assert "".join(event.delta for event in events if event.type == ResponsesAPIStreamEvents.OUTPUT_TEXT_DELTA) == "Hello!"
|
||||
assert event_types[-1] == ResponsesAPIStreamEvents.RESPONSE_COMPLETED
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_trailing_empty_choices_usage_chunk_reaches_response_completed():
|
||||
"""
|
||||
With `stream_options.include_usage` (which the bridge always sets) the last upstream chunk
|
||||
carries only usage and an empty `choices`. It must not crash the stream, and its usage must
|
||||
still land on `response.completed`.
|
||||
"""
|
||||
usage: Final = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15)
|
||||
iterator = _build_iterator([_chunk("Hello"), _chunk("", finish_reason="stop"), _empty_choices_chunk(usage)])
|
||||
|
||||
events = [event async for event in iterator]
|
||||
|
||||
completed = next(
|
||||
event for event in events if getattr(event, "type", None) == ResponsesAPIStreamEvents.RESPONSE_COMPLETED
|
||||
)
|
||||
assert completed.response.usage.input_tokens == 10
|
||||
assert completed.response.usage.output_tokens == 5
|
||||
|
||||
|
||||
def test_is_reasoning_end_ignores_empty_choices_chunk():
|
||||
assert _build_iterator([])._is_reasoning_end(_empty_choices_chunk()) is False
|
||||
|
||||
|
||||
def test_object_tool_call_arguments_stream_as_valid_json():
|
||||
"""A provider that sends decoded object arguments must still stream valid JSON.
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue