fix(streaming): keep litellm Usage on text-completion usage chunks (#43047)

* fix(streaming): keep litellm Usage on text-completion usage chunks

* fix(streaming): convert provider usage to litellm Usage instead of dropping it
This commit is contained in:
yuneng-jiang 2026-09-26 18:53:12 -07:00 • committed by GitHub
parent 013d5fa015
commit be35b22dfc
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 46 additions and 2 deletions

View file

@ -1504,8 +1504,11 @@ class CustomStreamWrapper:
self.tool_call = True
if hasattr(chunk, "usage") and chunk.usage is not None:
model_response.usage = chunk.usage
chunk_usage: Final = getattr(chunk, "usage", None)
if isinstance(chunk_usage, Usage):
model_response.usage = chunk_usage
elif isinstance(chunk_usage, BaseModel):
model_response.usage = Usage(**chunk_usage.model_dump())
## RETURN ARG
result: Final = self.return_processed_chunk_logic(

View file

@ -2859,6 +2859,47 @@ def test_dispatch_text_completion_openai_with_usage(
assert model_response.usage.total_tokens == 8
@pytest.mark.parametrize("custom_llm_provider", ["text-completion-openai", "azure_text"])
def test_text_completion_usage_chunk_keeps_provider_usage_as_litellm_usage(
initialized_custom_stream_wrapper: CustomStreamWrapper,
custom_llm_provider: str,
):
from openai.types.completion import Completion
from openai.types.completion_usage import CompletionUsage
initialized_custom_stream_wrapper.custom_llm_provider = custom_llm_provider
initialized_custom_stream_wrapper.model = "gpt-3.5-turbo-instruct"
initialized_custom_stream_wrapper.send_stream_usage = True
initialized_custom_stream_wrapper.received_finish_reason = "length"
provider_usage: Final = CompletionUsage.model_validate(
{
"prompt_tokens": 7,
"completion_tokens": 4,
"total_tokens": 11,
"completion_tokens_details": {"reasoning_tokens": 3},
"prompt_tokens_details": {"cached_tokens": 2},
"cost": 0.0123,
}
)
chunk: Final = Completion.model_construct(
id="cmpl-usage",
choices=[],
created=1,
model="gpt-3.5-turbo-instruct",
object="text_completion",
usage=provider_usage,
)
returned: Final = initialized_custom_stream_wrapper.chunk_creator(chunk=chunk)
assert isinstance(returned.usage, Usage)
dumped: Final = returned.model_dump()["usage"]
assert (dumped["prompt_tokens"], dumped["completion_tokens"], dumped["total_tokens"]) == (7, 4, 11)
assert dumped["cost"] == provider_usage.model_dump()["cost"]
assert dumped["completion_tokens_details"]["reasoning_tokens"] == 3
assert dumped["prompt_tokens_details"]["cached_tokens"] == 2
@pytest.mark.asyncio
async def test_custom_stream_wrapper_anext_does_not_block_event_loop_for_sync_iterators(
logging_obj: Logging,