From 4f7f949e1197ffcc6006b8fac176bffeb588d775 Mon Sep 17 00:00:00 2001 From: Lai Wei <8022184+roywei@users.noreply.github.com> Date: Sat, 3 Oct 2026 00:05:29 -0700 Subject: [PATCH 1/2] fix(streaming): keep finish_reason and provider usage when rebuilding streamed text completions stream_chunk_builder_text_completion read finish_reason from chunks[-1] and recounted usage from messages. With include_usage the last chunk is a usage-only trailer, so finish_reason came back None (or raised IndexError when the trailer had no choices), and prompt_tokens was 0 because a text-completion prompt is not in messages. Scan for the last finish_reason and prefer the provider-reported usage, as the chat path already does. Co-Authored-By: Claude Opus 5.5 --- litellm/main.py | 38 +++++++++++++++++++++++++++----------- tests/unit/test_main.py | 27 +++++++++++++++++++++++++++ 2 files changed, 54 insertions(+), 11 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index 6f72b6ff1ab..373b7f7df59 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -8843,8 +8843,14 @@ def stream_chunk_builder_text_completion(chunks: list, messages: Sequence | None created: Final = chunks[0]["created"] model: Final = chunks[0]["model"] system_fingerprint: Final = chunks[0].get("system_fingerprint", None) - finish_reason: Final = chunks[-1]["choices"][0]["finish_reason"] - logprobs: Final = chunks[-1]["choices"][0]["logprobs"] + # With stream_options.include_usage the last chunk is a usage-only trailer, and some providers + # (e.g. vLLM) send finish_reason on a text-less chunk before it, so scan rather than read chunks[-1]. + chunks_with_choices: Final = [chunk for chunk in chunks if chunk["choices"]] + finish_reason: Final = next( + (c["choices"][0]["finish_reason"] for c in reversed(chunks_with_choices) if c["choices"][0]["finish_reason"]), + None, + ) + logprobs: Final = chunks_with_choices[-1]["choices"][0]["logprobs"] if chunks_with_choices else None content_list: Final = [] for chunk in chunks: @@ -8857,16 +8863,26 @@ def stream_chunk_builder_text_completion(chunks: list, messages: Sequence | None # Combine the "content" strings into a single string || combine the 'function' strings into a single string combined_content: Final = "".join(content_list) - try: - prompt_tokens = token_counter(model=model, messages=messages) - except Exception: # don't allow this failing to block a complete streaming response from being returned - print_verbose("token_counter failed, assuming prompt tokens is 0") - prompt_tokens = 0 - completion_tokens: Final = token_counter( - model=model, - text=combined_content, - count_response_tokens=True, # count_response_tokens is a Flag to tell token counter this is a response, No need to add extra tokens we do for input messages + # Prefer the usage the provider reported (the include_usage trailer) over a local recount, which + # cannot see a text-completion prompt (it is not in `messages`) and so reports 0 prompt tokens. + provider_usage: Final = next( + (c.get("usage") for c in reversed(chunks) if c.get("usage") and c["usage"].get("total_tokens")), + None, ) + if provider_usage is not None: + prompt_tokens = provider_usage.get("prompt_tokens") or 0 + completion_tokens = provider_usage.get("completion_tokens") or 0 + else: + try: + prompt_tokens = token_counter(model=model, messages=messages) + except Exception: # don't allow this failing to block a complete streaming response from being returned + print_verbose("token_counter failed, assuming prompt tokens is 0") + prompt_tokens = 0 + completion_tokens = token_counter( + model=model, + text=combined_content, + count_response_tokens=True, # count_response_tokens is a Flag to tell token counter this is a response, No need to add extra tokens we do for input messages + ) response: Final = { "id": id, diff --git a/tests/unit/test_main.py b/tests/unit/test_main.py index e159e564a71..4d0fea1f3c1 100644 --- a/tests/unit/test_main.py +++ b/tests/unit/test_main.py @@ -3263,6 +3263,33 @@ def test_stream_chunk_builder_text_completion_combines_text_and_usage(): assert response.usage.total_tokens == response.usage.prompt_tokens + response.usage.completion_tokens +@pytest.mark.parametrize("trailer_choices", [[], [{"text": None, "index": 0, "logprobs": None, "finish_reason": None}]]) +def test_stream_chunk_builder_text_completion_keeps_finish_reason_and_provider_usage(trailer_choices): + """vLLM-style include_usage stream: finish_reason arrives on a text-less chunk, then a + usage-only trailer. The rebuilt response must keep both instead of reading chunks[-1] + and recounting the prompt from `messages` (which a text completion doesn't have).""" + from litellm.main import stream_chunk_builder_text_completion + from litellm.types.utils import TextCompletionResponse + + def chunk(choices, **extra): + return TextCompletionResponse( + id="cmpl-1", object="text_completion", created=1, model="my-model", choices=choices, **extra + ) + + chunks = [ + chunk([{"text": "Hello", "index": 0, "logprobs": None, "finish_reason": None}]), + chunk([{"text": " world", "index": 0, "logprobs": None, "finish_reason": None}]), + chunk([{"text": "", "index": 0, "logprobs": None, "finish_reason": "length"}]), + chunk(trailer_choices, usage={"prompt_tokens": 7, "completion_tokens": 2, "total_tokens": 9}), + ] + + response = stream_chunk_builder_text_completion(chunks=chunks, messages=None) + + assert response.choices[0].text == "Hello world" + assert response.choices[0].finish_reason == "length" + assert (response.usage.prompt_tokens, response.usage.completion_tokens, response.usage.total_tokens) == (7, 2, 9) + + def test_completion_forwards_store_and_prompt_cache_key_to_openai(): """ Regression test for https://github.com/BerriAI/litellm/issues/33184 From 9eb748effdc3eb42ad17010bcab8c47fa59f818c Mon Sep 17 00:00:00 2001 From: Lai Wei Date: Sat, 3 Oct 2026 00:34:12 -0700 Subject: [PATCH 2/2] fix(streaming): type the text completion chunk builder so its scans add no basedpyright debt Read chunk fields as typed attributes instead of untyped subscripts, which clears the reportUnknownArgumentType breach the lint gate flagged --- litellm/main.py | 41 ++++++++++++++++++----------------------- 1 file changed, 18 insertions(+), 23 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index 373b7f7df59..99d560ac70e 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -13,6 +13,7 @@ import asyncio import contextvars import datetime import inspect +import itertools import json import os import random @@ -8837,49 +8838,43 @@ def config_completion(**kwargs): ) -def stream_chunk_builder_text_completion(chunks: list, messages: Sequence | None = None) -> TextCompletionResponse: - id: Final = chunks[0]["id"] - object: Final = chunks[0]["object"] - created: Final = chunks[0]["created"] - model: Final = chunks[0]["model"] +def stream_chunk_builder_text_completion( + chunks: Sequence[TextCompletionResponse], messages: Sequence | None = None +) -> TextCompletionResponse: + id: Final = chunks[0].id + object: Final = chunks[0].object + created: Final = chunks[0].created + model: Final = chunks[0].model system_fingerprint: Final = chunks[0].get("system_fingerprint", None) # With stream_options.include_usage the last chunk is a usage-only trailer, and some providers # (e.g. vLLM) send finish_reason on a text-less chunk before it, so scan rather than read chunks[-1]. - chunks_with_choices: Final = [chunk for chunk in chunks if chunk["choices"]] + chunks_with_choices: Final = [chunk for chunk in chunks if chunk.choices] finish_reason: Final = next( - (c["choices"][0]["finish_reason"] for c in reversed(chunks_with_choices) if c["choices"][0]["finish_reason"]), + (c.choices[0].finish_reason for c in reversed(chunks_with_choices) if c.choices[0].finish_reason), None, ) - logprobs: Final = chunks_with_choices[-1]["choices"][0]["logprobs"] if chunks_with_choices else None + logprobs: Final = chunks_with_choices[-1].choices[0].logprobs if chunks_with_choices else None - content_list: Final = [] - for chunk in chunks: - choices = chunk["choices"] - for choice in choices: - if choice is not None and hasattr(choice, "text") and choice.get("text") is not None: - _choice = choice.get("text") - content_list.append(_choice) - - # Combine the "content" strings into a single string || combine the 'function' strings into a single string - combined_content: Final = "".join(content_list) + all_choices: Final = itertools.chain.from_iterable(chunk.choices for chunk in chunks) + combined_content: Final = "".join(choice.text for choice in all_choices if choice.text) # Prefer the usage the provider reported (the include_usage trailer) over a local recount, which # cannot see a text-completion prompt (it is not in `messages`) and so reports 0 prompt tokens. provider_usage: Final = next( - (c.get("usage") for c in reversed(chunks) if c.get("usage") and c["usage"].get("total_tokens")), + (c.usage for c in reversed(chunks) if c.usage and c.usage.total_tokens), None, ) if provider_usage is not None: - prompt_tokens = provider_usage.get("prompt_tokens") or 0 - completion_tokens = provider_usage.get("completion_tokens") or 0 + prompt_tokens = provider_usage.prompt_tokens + completion_tokens = provider_usage.completion_tokens else: try: - prompt_tokens = token_counter(model=model, messages=messages) + prompt_tokens = token_counter(model=model or "", messages=messages) except Exception: # don't allow this failing to block a complete streaming response from being returned print_verbose("token_counter failed, assuming prompt tokens is 0") prompt_tokens = 0 completion_tokens = token_counter( - model=model, + model=model or "", text=combined_content, count_response_tokens=True, # count_response_tokens is a Flag to tell token counter this is a response, No need to add extra tokens we do for input messages )