From 65918b0495bda147c97b64ee72e3b16da2113dee Mon Sep 17 00:00:00 2001 From: Sriniketh24 Date: Sun, 24 May 2026 08:03:20 +0530 Subject: [PATCH] Fix streaming usage chunk to have empty choices per OpenAI spec When `stream_options: {"include_usage": true}` is set, the final streaming chunk carrying token usage must have `choices: []` per the OpenAI API specification. Previously, `model_response_creator()` filled in a default `StreamingChoices(finish_reason=None)` placeholder, causing downstream clients (LangChain.js, Dify, etc.) to double-count tokens or misinterpret the usage-only event as a content chunk. The fix sets `response.choices = []` on the synthetic usage chunk in both the sync (`__next__`) and async (`__anext__`) paths, immediately after `model_response_creator()` builds the response. Fixes #28735 --- .../litellm_core_utils/streaming_handler.py | 8 ++ .../test_streaming_handler.py | 88 +++++++++++++++++-- 2 files changed, 87 insertions(+), 9 deletions(-) diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index fa7faf3035d..1a8774e2d04 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -1928,6 +1928,10 @@ class CustomStreamWrapper: ) response = self.model_response_creator() + # Per the OpenAI streaming spec, the final usage chunk + # must have choices: [] (not a default placeholder choice). + # https://platform.openai.com/docs/api-reference/chat/streaming + response.choices = [] if complete_streaming_response is not None: setattr( response, @@ -2157,6 +2161,10 @@ class CustomStreamWrapper: ) response = self.model_response_creator() + # Per the OpenAI streaming spec, the final usage chunk + # must have choices: [] (not a default placeholder choice). + # https://platform.openai.com/docs/api-reference/chat/streaming + response.choices = [] if complete_streaming_response is not None: setattr( response, diff --git a/tests/test_litellm/litellm_core_utils/test_streaming_handler.py b/tests/test_litellm/litellm_core_utils/test_streaming_handler.py index 49d3c51e340..1b28c666913 100644 --- a/tests/test_litellm/litellm_core_utils/test_streaming_handler.py +++ b/tests/test_litellm/litellm_core_utils/test_streaming_handler.py @@ -2036,23 +2036,19 @@ async def test_azure_streaming_role_preserved_with_include_usage(sync_mode: bool chunks.append(chunk) # The prompt_filter chunk should be forwarded with choices=[] - assert len(chunks[0].choices) == 0, ( - f"Expected prompt_filter chunk with choices=[], got {len(chunks[0].choices)} choices" - ) + assert ( + len(chunks[0].choices) == 0 + ), f"Expected prompt_filter chunk with choices=[], got {len(chunks[0].choices)} choices" # At least one chunk must have role='assistant' in its delta has_role = any( - len(c.choices) > 0 - and getattr(c.choices[0].delta, "role", None) == "assistant" + len(c.choices) > 0 and getattr(c.choices[0].delta, "role", None) == "assistant" for c in chunks ) assert has_role, ( "No chunk contained role='assistant' in delta (issue #24221). " "Chunk deltas: " - + str([ - c.choices[0].delta if c.choices else "no choices" - for c in chunks - ]) + + str([c.choices[0].delta if c.choices else "no choices" for c in chunks]) ) @@ -2124,3 +2120,77 @@ def test_gemini_legacy_vertex_tool_calls_finish_reason_with_stop_enum(): f"Expected 'tool_calls' but got {final.choices[0].finish_reason!r}. " "STOP enum was not normalised through map_finish_reason()." ) + + +class TestUsageChunkEmptyChoices: + """The OpenAI streaming spec requires the final usage chunk to have choices: []. + + See: https://platform.openai.com/docs/api-reference/chat/streaming + Regression test for https://github.com/BerriAI/litellm/issues/28735 + """ + + def _build_wrapper(self, logging_obj): + """Build a CustomStreamWrapper that has already sent the last content chunk.""" + chunks = [ + ModelResponseStream( + id="chatcmpl-test", + choices=[ + StreamingChoices( + finish_reason=None, + index=0, + delta=Delta(content="hello", role="assistant"), + ) + ], + ), + ModelResponseStream( + id="chatcmpl-test", + choices=[ + StreamingChoices( + finish_reason="stop", + index=0, + delta=Delta(content=None), + ) + ], + ), + ] + completion_stream = ModelResponseListIterator(model_responses=chunks) + wrapper = CustomStreamWrapper( + completion_stream=completion_stream, + model="gpt-4", + logging_obj=logging_obj, + custom_llm_provider="openai", + stream_options={"include_usage": True}, + ) + return wrapper + + def test_sync_usage_chunk_has_empty_choices(self, logging_obj): + """Sync __next__: the final usage-only chunk must have choices=[].""" + wrapper = self._build_wrapper(logging_obj) + + collected = [] + for chunk in wrapper: + collected.append(chunk) + + # The last chunk should be the usage chunk + usage_chunk = collected[-1] + assert hasattr(usage_chunk, "usage"), "Final chunk should carry usage" + assert usage_chunk.choices == [], ( + f"Usage chunk must have choices=[] per OpenAI spec, " + f"got {usage_chunk.choices!r}" + ) + + @pytest.mark.asyncio + async def test_async_usage_chunk_has_empty_choices(self, logging_obj): + """Async __anext__: the final usage-only chunk must have choices=[].""" + wrapper = self._build_wrapper(logging_obj) + + collected = [] + async for chunk in wrapper: + collected.append(chunk) + + usage_chunk = collected[-1] + assert hasattr(usage_chunk, "usage"), "Final chunk should carry usage" + assert usage_chunk.choices == [], ( + f"Usage chunk must have choices=[] per OpenAI spec, " + f"got {usage_chunk.choices!r}" + )