mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge pull request #35314 from BerriAI/devin_ai_fix_lit5034_empty_choices
fix(anthropic adapter): stop indexing choices[0] on choiceless streaming chunks
This commit is contained in:
commit
281e52ac49
2 changed files with 117 additions and 0 deletions
|
|
@ -348,6 +348,26 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
merged_chunk["context_management"] = ContextManagementResponse(applied_edits=list(self.applied_edits))
|
||||
return self._augment_message_delta_usage(merged_chunk)
|
||||
|
||||
def _handle_choiceless_chunk(self, chunk: "ModelResponseStream") -> bool:
|
||||
"""Consume an OpenAI-compatible chunk that carries no ``choices``.
|
||||
|
||||
``choices`` is legitimately empty on metadata-only chunks; the final
|
||||
usage chunk emitted when ``stream_options.include_usage`` is set is the
|
||||
common case (vLLM and other OpenAI-compatible servers do this). Such a
|
||||
chunk carries no content-block information, so the caller must not run
|
||||
the content-block state machine over it.
|
||||
|
||||
Returns True when a merged ``message_delta`` was queued (usage folded
|
||||
into the held stop-reason chunk); False when the chunk should be
|
||||
skipped entirely.
|
||||
"""
|
||||
if self.holding_stop_reason_chunk is not None and _optional_attr(chunk, "usage") is not None:
|
||||
self.chunk_queue.append(self._merge_usage_into_held_stop_reason_chunk(chunk))
|
||||
self.queued_usage_chunk = True
|
||||
self.holding_stop_reason_chunk = None
|
||||
return True
|
||||
return False
|
||||
|
||||
def _ensure_context_management_attached(self, message_delta_chunk: MessageBlockDelta) -> MessageBlockDelta:
|
||||
"""Attach ``context_management`` to a ``message_delta`` chunk if
|
||||
``self.applied_edits`` is non-empty and the chunk does not already
|
||||
|
|
@ -509,6 +529,11 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
if chunk == "None" or chunk is None:
|
||||
raise Exception
|
||||
|
||||
if not getattr(chunk, "choices", None):
|
||||
if self._handle_choiceless_chunk(chunk):
|
||||
return self.chunk_queue.popleft()
|
||||
continue
|
||||
|
||||
should_start_new_block = self._should_start_new_content_block(chunk)
|
||||
is_opening_first_block = self.sent_content_block_start is False
|
||||
if is_opening_first_block and self._is_blank_delta(chunk):
|
||||
|
|
@ -732,6 +757,11 @@ class AnthropicStreamWrapper(AdapterCompletionStreamWrapper):
|
|||
if chunk == "None" or chunk is None:
|
||||
raise Exception
|
||||
|
||||
if not getattr(chunk, "choices", None):
|
||||
if self._handle_choiceless_chunk(chunk):
|
||||
return self.chunk_queue.popleft()
|
||||
continue
|
||||
|
||||
should_start_new_block = self._should_start_new_content_block(chunk)
|
||||
is_opening_first_block = self.sent_content_block_start is False
|
||||
if is_opening_first_block and self._is_blank_delta(chunk):
|
||||
|
|
|
|||
|
|
@ -0,0 +1,87 @@
|
|||
"""
|
||||
Regression tests for OpenAI-compatible chunks with an empty ``choices`` list.
|
||||
|
||||
``choices: []`` is valid OpenAI-compatible streaming: vLLM (and OpenAI itself,
|
||||
when ``stream_options.include_usage`` is set) emits a final usage chunk with no
|
||||
choices, and some gateways emit metadata-only chunks mid-stream. The adapter
|
||||
used to index ``chunk.choices[0]`` unconditionally, so such a chunk raised
|
||||
``IndexError: list index out of range`` and killed the ``/v1/messages`` stream.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
from typing import Any, AsyncIterator, Dict, List, Optional
|
||||
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import (
|
||||
AnthropicStreamWrapper,
|
||||
)
|
||||
from litellm.types.utils import Delta, ModelResponseStream, StreamingChoices, Usage
|
||||
|
||||
|
||||
def _text_chunk(text: str) -> ModelResponseStream:
|
||||
return ModelResponseStream(
|
||||
choices=[StreamingChoices(index=0, delta=Delta(content=text), finish_reason=None)]
|
||||
)
|
||||
|
||||
|
||||
def _finish_chunk() -> ModelResponseStream:
|
||||
return ModelResponseStream(choices=[StreamingChoices(index=0, delta=Delta(), finish_reason="stop")])
|
||||
|
||||
|
||||
def _empty_choices_chunk(usage: Optional[Usage] = None) -> ModelResponseStream:
|
||||
return ModelResponseStream(choices=[], usage=usage)
|
||||
|
||||
|
||||
def _collect_async(wrapper: AnthropicStreamWrapper) -> str:
|
||||
async def _run() -> str:
|
||||
return "".join(
|
||||
[raw.decode() if isinstance(raw, bytes) else raw async for raw in wrapper.async_anthropic_sse_wrapper()]
|
||||
)
|
||||
|
||||
return asyncio.run(_run())
|
||||
|
||||
|
||||
def _message_delta(sse: str) -> Dict[str, Any]:
|
||||
return next(
|
||||
json.loads(line[len("data: ") :])
|
||||
for block in sse.split("\n\n")
|
||||
for line in block.splitlines()
|
||||
if line.startswith("data: ") and '"message_delta"' in line
|
||||
)
|
||||
|
||||
|
||||
def test_leading_metadata_chunk_without_choices_does_not_kill_stream():
|
||||
"""A metadata-only chunk before any content must be skipped, not indexed."""
|
||||
chunks: List[ModelResponseStream] = [
|
||||
_empty_choices_chunk(),
|
||||
_text_chunk("Hello"),
|
||||
_text_chunk(" there"),
|
||||
_finish_chunk(),
|
||||
]
|
||||
wrapper = AnthropicStreamWrapper(completion_stream=iter(chunks), model="mock-model")
|
||||
events = list(wrapper)
|
||||
|
||||
text = "".join(
|
||||
event["delta"]["text"] for event in events if event.get("type") == "content_block_delta"
|
||||
)
|
||||
assert text == "Hello there"
|
||||
assert events[-1]["type"] == "message_stop"
|
||||
|
||||
|
||||
def test_final_usage_chunk_without_choices_is_merged_into_message_delta():
|
||||
"""The vLLM/OpenAI final usage chunk carries no choices; its usage must
|
||||
still land on the Anthropic ``message_delta``."""
|
||||
usage = Usage(prompt_tokens=10, completion_tokens=3, total_tokens=13)
|
||||
|
||||
async def _aiter() -> "AsyncIterator[ModelResponseStream]":
|
||||
for chunk in [_text_chunk("Hi"), _finish_chunk(), _empty_choices_chunk(usage)]:
|
||||
yield chunk
|
||||
|
||||
sse = _collect_async(AnthropicStreamWrapper(completion_stream=_aiter(), model="mock-model"))
|
||||
|
||||
message_delta = _message_delta(sse)
|
||||
assert message_delta["delta"]["stop_reason"] == "end_turn"
|
||||
assert message_delta["usage"]["input_tokens"] == 10
|
||||
assert message_delta["usage"]["output_tokens"] == 3
|
||||
assert "Hi" in sse
|
||||
assert "message_stop" in sse
|
||||
Loading…
Add table
Reference in a new issue