mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(responses): preserve reasoning_tokens through chat->responses usage translation (#32837)
* fix(responses): preserve reasoning_tokens through chat->responses usage translation Remove the unconditional else-branch that wrote reasoning_tokens=0 whenever completion_tokens_details.reasoning_tokens was None or absent. Also change OutputTokensDetails.reasoning_tokens from int=0 to Optional[int]=None so that re-instantiation without explicit reasoning_tokens no longer silently zeroes out the field, and remove the same hardcoded zero from the mock_responses_api_response initializer. * test(responses): update assertions to match Optional[int] reasoning_tokens default * fix(responses): preserve explicit reasoning_tokens=0 in usage translation Align the reasoning_tokens guard with the is-not-None guards used for text_tokens and image_tokens: a provider-reported zero passes through while an absent value stays omitted. --------- Co-authored-by: Deepanshu <deepanshu.lulla@alpha-sense.com>
This commit is contained in:
parent
4737e75c86
commit
ee6e8077ae
5 changed files with 124 additions and 23 deletions
|
|
@ -2020,8 +2020,6 @@ class LiteLLMCompletionResponsesConfig:
|
|||
output_details_dict: dict[str, int] = {}
|
||||
if hasattr(completion_details, "reasoning_tokens") and completion_details.reasoning_tokens is not None:
|
||||
output_details_dict["reasoning_tokens"] = completion_details.reasoning_tokens
|
||||
else:
|
||||
output_details_dict["reasoning_tokens"] = 0
|
||||
|
||||
if hasattr(completion_details, "text_tokens") and completion_details.text_tokens is not None:
|
||||
output_details_dict["text_tokens"] = completion_details.text_tokens
|
||||
|
|
|
|||
|
|
@ -127,7 +127,7 @@ def mock_responses_api_response(
|
|||
"input_tokens": 36,
|
||||
"input_tokens_details": {"cached_tokens": 0},
|
||||
"output_tokens": 87,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"output_tokens_details": {},
|
||||
"total_tokens": 123,
|
||||
},
|
||||
"user": None,
|
||||
|
|
|
|||
|
|
@ -1185,7 +1185,7 @@ class ResponsesAPIRequestParams(ResponsesAPIOptionalRequestParams, total=False):
|
|||
|
||||
|
||||
class OutputTokensDetails(BaseLiteLLMOpenAIResponseObject):
|
||||
reasoning_tokens: int = 0
|
||||
reasoning_tokens: Optional[int] = None
|
||||
|
||||
text_tokens: Optional[int] = None
|
||||
|
||||
|
|
|
|||
|
|
@ -1959,6 +1959,110 @@ class TestUsageTransformation:
|
|||
assert response_usage.output_tokens_details.text_tokens == 50
|
||||
assert response_usage.output_tokens_details.image_tokens == 100
|
||||
|
||||
def test_reasoning_tokens_not_forced_to_zero_when_absent(self):
|
||||
# Regression: previously the else branch wrote reasoning_tokens=0 even when
|
||||
# completion_tokens_details had no reasoning (reasoning_tokens=None). That caused
|
||||
# the proxy to always report reasoning_tokens=0 for non-thinking responses.
|
||||
usage = Usage(
|
||||
prompt_tokens=10,
|
||||
completion_tokens=50,
|
||||
total_tokens=60,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
text_tokens=50,
|
||||
# reasoning_tokens intentionally absent -> None
|
||||
),
|
||||
)
|
||||
|
||||
chat_completion_response = ModelResponse(
|
||||
id="test-response-id",
|
||||
created=1234567890,
|
||||
model="claude-haiku-4-5",
|
||||
object="chat.completion",
|
||||
usage=usage,
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
message=Message(content="Hello!", role="assistant"),
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
response_usage = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
|
||||
chat_completion_response=chat_completion_response
|
||||
)
|
||||
|
||||
assert response_usage.output_tokens_details is not None
|
||||
assert response_usage.output_tokens_details.reasoning_tokens is None
|
||||
|
||||
def test_reasoning_tokens_preserved_when_thinking_occurred(self):
|
||||
# Regression: reasoning_tokens must survive the chat->responses translation
|
||||
# when the provider actually did thinking.
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=612,
|
||||
total_tokens=712,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
reasoning_tokens=512,
|
||||
text_tokens=100,
|
||||
),
|
||||
)
|
||||
|
||||
chat_completion_response = ModelResponse(
|
||||
id="test-response-id",
|
||||
created=1234567890,
|
||||
model="claude-haiku-4-5",
|
||||
object="chat.completion",
|
||||
usage=usage,
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
message=Message(content="Hello!", role="assistant"),
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
response_usage = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
|
||||
chat_completion_response=chat_completion_response
|
||||
)
|
||||
|
||||
assert response_usage.output_tokens_details is not None
|
||||
assert response_usage.output_tokens_details.reasoning_tokens == 512
|
||||
|
||||
def test_reasoning_tokens_explicit_zero_preserved(self):
|
||||
usage = Usage(
|
||||
prompt_tokens=10,
|
||||
completion_tokens=50,
|
||||
total_tokens=60,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
reasoning_tokens=0,
|
||||
text_tokens=50,
|
||||
),
|
||||
)
|
||||
|
||||
chat_completion_response = ModelResponse(
|
||||
id="test-response-id",
|
||||
created=1234567890,
|
||||
model="gpt-5.6",
|
||||
object="chat.completion",
|
||||
usage=usage,
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
message=Message(content="Hello!", role="assistant"),
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
response_usage = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
|
||||
chat_completion_response=chat_completion_response
|
||||
)
|
||||
|
||||
assert response_usage.output_tokens_details is not None
|
||||
assert response_usage.output_tokens_details.reasoning_tokens == 0
|
||||
|
||||
|
||||
class TestStreamingIDConsistency:
|
||||
"""Test cases for consistent IDs across streaming events (issue #14962)"""
|
||||
|
|
|
|||
|
|
@ -269,29 +269,30 @@ def test_transform_usage_with_zero_values():
|
|||
"""
|
||||
Test transformation when token details are explicitly set to 0.
|
||||
|
||||
This ensures 0 values are preserved and not treated as None.
|
||||
cached_tokens=0 is preserved (cache was available; nothing was cached).
|
||||
reasoning_tokens=0 is preserved the same way: an explicit provider-reported
|
||||
zero passes through, while an absent value (None) is omitted.
|
||||
"""
|
||||
completion_response = create_mock_completion_response(
|
||||
model="gpt-4",
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
cached_tokens=0, # Explicitly 0
|
||||
reasoning_tokens=0, # Explicitly 0
|
||||
cached_tokens=0, # Explicitly 0 — preserved
|
||||
reasoning_tokens=0, # Explicitly 0 — preserved
|
||||
)
|
||||
|
||||
responses_usage = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
|
||||
completion_response
|
||||
)
|
||||
|
||||
# Should preserve 0 values
|
||||
assert responses_usage.input_tokens_details is not None
|
||||
assert responses_usage.input_tokens_details.cached_tokens == 0
|
||||
|
||||
assert responses_usage.output_tokens_details is not None
|
||||
assert responses_usage.output_tokens_details.reasoning_tokens == 0
|
||||
|
||||
print("✓ Transformation preserves explicit 0 values")
|
||||
print("✓ Transformation preserves explicit reasoning_tokens=0 and omits absent values")
|
||||
|
||||
|
||||
def test_input_tokens_details_requires_cached_tokens():
|
||||
|
|
@ -315,25 +316,23 @@ def test_input_tokens_details_requires_cached_tokens():
|
|||
print("✓ InputTokensDetails correctly defaults cached_tokens to 0")
|
||||
|
||||
|
||||
def test_output_tokens_details_requires_reasoning_tokens():
|
||||
def test_output_tokens_details_reasoning_tokens():
|
||||
"""
|
||||
Test that OutputTokensDetails has reasoning_tokens as an int with default value 0.
|
||||
Test OutputTokensDetails.reasoning_tokens field semantics.
|
||||
|
||||
This ensures backward compatibility while making the field non-optional.
|
||||
reasoning_tokens is Optional[int] = None: present only when reasoning actually occurred.
|
||||
"""
|
||||
# Should work with reasoning_tokens=0
|
||||
details1 = OutputTokensDetails(reasoning_tokens=0)
|
||||
assert details1.reasoning_tokens == 0
|
||||
details_explicit_zero = OutputTokensDetails(reasoning_tokens=0)
|
||||
assert details_explicit_zero.reasoning_tokens == 0
|
||||
|
||||
# Should work with reasoning_tokens=100
|
||||
details2 = OutputTokensDetails(reasoning_tokens=100)
|
||||
assert details2.reasoning_tokens == 100
|
||||
details_positive = OutputTokensDetails(reasoning_tokens=100)
|
||||
assert details_positive.reasoning_tokens == 100
|
||||
|
||||
# Should work without reasoning_tokens (defaults to 0)
|
||||
details3 = OutputTokensDetails()
|
||||
assert details3.reasoning_tokens == 0
|
||||
# Default is None — absence means reasoning did not occur (or was not tracked)
|
||||
details_default = OutputTokensDetails()
|
||||
assert details_default.reasoning_tokens is None
|
||||
|
||||
print("✓ OutputTokensDetails correctly defaults reasoning_tokens to 0")
|
||||
print("✓ OutputTokensDetails.reasoning_tokens defaults to None")
|
||||
|
||||
|
||||
def test_all_providers_transformation_scenarios():
|
||||
|
|
@ -419,7 +418,7 @@ if __name__ == "__main__":
|
|||
test_transform_usage_with_both_token_details()
|
||||
test_transform_usage_with_zero_values()
|
||||
test_input_tokens_details_requires_cached_tokens()
|
||||
test_output_tokens_details_requires_reasoning_tokens()
|
||||
test_output_tokens_details_reasoning_tokens()
|
||||
test_all_providers_transformation_scenarios()
|
||||
|
||||
print("\n" + "=" * 60)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue