fix(streaming): return None from count_reasoning_tokens when no reasoning content found (#23076)

Prevents spurious reasoning_tokens=0 from being injected into
completion_tokens_details for non-reasoning models (e.g. gpt-4o-audio-preview).

Previously count_reasoning_tokens always returned int (defaulting to 0),
so the if reasoning_tokens is not None guard in calculate_usage always
fired and wrote reasoning_tokens=0 into completion_tokens_details even
when the model never emitted any reasoning content. This caused
stream_chunk_builder to produce a usage dict with an extra
reasoning_tokens=0 field that was not in the original streaming chunk,
breaking the equality assertion in test_stream_chunk_builder_openai_audio_output_usage.

Fix: return Optional[int] -- None means no reasoning content seen,
0 means reasoning_content was present but counted zero tokens.
This commit is contained in:
Ishaan Jaff 2026-03-07 17:03:53 -08:00 • committed by GitHub
parent ada8877aeb
commit 517a929ccd
No known key found for this signature in database
GPG key ID: B5690EEEBB952194

View file

@ -476,13 +476,15 @@ class ChunkProcessor:
"prompt_tokens_details": prompt_tokens_details,
}
def count_reasoning_tokens(self, response: ModelResponse) -> int:
reasoning_tokens = 0
def count_reasoning_tokens(self, response: ModelResponse) -> Optional[int]:
reasoning_tokens: Optional[int] = None
for choice in response.choices:
if (
hasattr(cast(Choices, choice).message, "reasoning_content")
and cast(Choices, choice).message.reasoning_content is not None
):
if reasoning_tokens is None:
reasoning_tokens = 0
reasoning_tokens += token_counter(
text=cast(Choices, choice).message.reasoning_content,
count_response_tokens=True,