From 517a929ccdfdc0fb1ba5433e2a3c0a7c37653ec5 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 7 Mar 2026 17:03:53 -0800 Subject: [PATCH] fix(streaming): return None from count_reasoning_tokens when no reasoning content found (#23076) Prevents spurious reasoning_tokens=0 from being injected into completion_tokens_details for non-reasoning models (e.g. gpt-4o-audio-preview). Previously count_reasoning_tokens always returned int (defaulting to 0), so the if reasoning_tokens is not None guard in calculate_usage always fired and wrote reasoning_tokens=0 into completion_tokens_details even when the model never emitted any reasoning content. This caused stream_chunk_builder to produce a usage dict with an extra reasoning_tokens=0 field that was not in the original streaming chunk, breaking the equality assertion in test_stream_chunk_builder_openai_audio_output_usage. Fix: return Optional[int] -- None means no reasoning content seen, 0 means reasoning_content was present but counted zero tokens. --- litellm/litellm_core_utils/streaming_chunk_builder_utils.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py index 143d87ebf34..ba35a2c7cad 100644 --- a/litellm/litellm_core_utils/streaming_chunk_builder_utils.py +++ b/litellm/litellm_core_utils/streaming_chunk_builder_utils.py @@ -476,13 +476,15 @@ class ChunkProcessor: "prompt_tokens_details": prompt_tokens_details, } - def count_reasoning_tokens(self, response: ModelResponse) -> int: - reasoning_tokens = 0 + def count_reasoning_tokens(self, response: ModelResponse) -> Optional[int]: + reasoning_tokens: Optional[int] = None for choice in response.choices: if ( hasattr(cast(Choices, choice).message, "reasoning_content") and cast(Choices, choice).message.reasoning_content is not None ): + if reasoning_tokens is None: + reasoning_tokens = 0 reasoning_tokens += token_counter( text=cast(Choices, choice).message.reasoning_content, count_response_tokens=True,