fix(streaming): keep finish_reason and provider usage when rebuilding streamed text completions

stream_chunk_builder_text_completion read finish_reason from chunks[-1] and recounted usage from messages. With include_usage the last chunk is a usage-only trailer, so finish_reason came back None (or raised IndexError when the trailer had no choices), and prompt_tokens was 0 because a text-completion prompt is not in messages. Scan for the last finish_reason and prefer the provider-reported usage, as the chat path already does.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Lai Wei 2026-10-03 00:05:29 -07:00
parent 8b11b682f4
commit 4f7f949e11
2 changed files with 54 additions and 11 deletions

View file

@ -8843,8 +8843,14 @@ def stream_chunk_builder_text_completion(chunks: list, messages: Sequence | None
created: Final = chunks[0]["created"]
model: Final = chunks[0]["model"]
system_fingerprint: Final = chunks[0].get("system_fingerprint", None)
finish_reason: Final = chunks[-1]["choices"][0]["finish_reason"]
logprobs: Final = chunks[-1]["choices"][0]["logprobs"]
# With stream_options.include_usage the last chunk is a usage-only trailer, and some providers
# (e.g. vLLM) send finish_reason on a text-less chunk before it, so scan rather than read chunks[-1].
chunks_with_choices: Final = [chunk for chunk in chunks if chunk["choices"]]
finish_reason: Final = next(
(c["choices"][0]["finish_reason"] for c in reversed(chunks_with_choices) if c["choices"][0]["finish_reason"]),
None,
)
logprobs: Final = chunks_with_choices[-1]["choices"][0]["logprobs"] if chunks_with_choices else None
content_list: Final = []
for chunk in chunks:
@ -8857,16 +8863,26 @@ def stream_chunk_builder_text_completion(chunks: list, messages: Sequence | None
# Combine the "content" strings into a single string || combine the 'function' strings into a single string
combined_content: Final = "".join(content_list)
try:
prompt_tokens = token_counter(model=model, messages=messages)
except Exception: # don't allow this failing to block a complete streaming response from being returned
print_verbose("token_counter failed, assuming prompt tokens is 0")
prompt_tokens = 0
completion_tokens: Final = token_counter(
model=model,
text=combined_content,
count_response_tokens=True, # count_response_tokens is a Flag to tell token counter this is a response, No need to add extra tokens we do for input messages
# Prefer the usage the provider reported (the include_usage trailer) over a local recount, which
# cannot see a text-completion prompt (it is not in `messages`) and so reports 0 prompt tokens.
provider_usage: Final = next(
(c.get("usage") for c in reversed(chunks) if c.get("usage") and c["usage"].get("total_tokens")),
None,
)
if provider_usage is not None:
prompt_tokens = provider_usage.get("prompt_tokens") or 0
completion_tokens = provider_usage.get("completion_tokens") or 0
else:
try:
prompt_tokens = token_counter(model=model, messages=messages)
except Exception: # don't allow this failing to block a complete streaming response from being returned
print_verbose("token_counter failed, assuming prompt tokens is 0")
prompt_tokens = 0
completion_tokens = token_counter(
model=model,
text=combined_content,
count_response_tokens=True, # count_response_tokens is a Flag to tell token counter this is a response, No need to add extra tokens we do for input messages
)
response: Final = {
"id": id,

View file

@ -3263,6 +3263,33 @@ def test_stream_chunk_builder_text_completion_combines_text_and_usage():
assert response.usage.total_tokens == response.usage.prompt_tokens + response.usage.completion_tokens
@pytest.mark.parametrize("trailer_choices", [[], [{"text": None, "index": 0, "logprobs": None, "finish_reason": None}]])
def test_stream_chunk_builder_text_completion_keeps_finish_reason_and_provider_usage(trailer_choices):
"""vLLM-style include_usage stream: finish_reason arrives on a text-less chunk, then a
usage-only trailer. The rebuilt response must keep both instead of reading chunks[-1]
and recounting the prompt from `messages` (which a text completion doesn't have)."""
from litellm.main import stream_chunk_builder_text_completion
from litellm.types.utils import TextCompletionResponse
def chunk(choices, **extra):
return TextCompletionResponse(
id="cmpl-1", object="text_completion", created=1, model="my-model", choices=choices, **extra
)
chunks = [
chunk([{"text": "Hello", "index": 0, "logprobs": None, "finish_reason": None}]),
chunk([{"text": " world", "index": 0, "logprobs": None, "finish_reason": None}]),
chunk([{"text": "", "index": 0, "logprobs": None, "finish_reason": "length"}]),
chunk(trailer_choices, usage={"prompt_tokens": 7, "completion_tokens": 2, "total_tokens": 9}),
]
response = stream_chunk_builder_text_completion(chunks=chunks, messages=None)
assert response.choices[0].text == "Hello world"
assert response.choices[0].finish_reason == "length"
assert (response.usage.prompt_tokens, response.usage.completion_tokens, response.usage.total_tokens) == (7, 2, 9)
def test_completion_forwards_store_and_prompt_cache_key_to_openai():
"""
Regression test for https://github.com/BerriAI/litellm/issues/33184