fix(cost): bill realtime reasoning tokens nested in text_tokens once

OpenAI and Azure realtime usage reports output_tokens == text_tokens + audio_tokens
with reasoning_tokens counted inside text_tokens, so generic_cost_per_token billed
the reasoning share twice. When the output token details sum past completion_tokens,
the nested reasoning overlap is now subtracted from text_tokens before pricing;
shapes where text_tokens already excludes reasoning are unchanged.
This commit is contained in:
mateo-berri 2026-09-04 19:11:27 -07:00
parent 639b3f4f62
commit 001b531ae9
3 changed files with 111 additions and 1 deletions

View file

@ -852,6 +852,17 @@ class CompletionTokensDetailsResult(TypedDict):
video_tokens: int
def _text_tokens_without_nested_reasoning(
completion_tokens: int,
text_tokens: int,
reasoning_tokens: int,
other_modality_tokens: int,
) -> int:
reported_total: Final = text_tokens + reasoning_tokens + other_modality_tokens
nested_reasoning_tokens: Final = min(reasoning_tokens, text_tokens, max(reported_total - completion_tokens, 0))
return text_tokens - nested_reasoning_tokens
def parse_completion_tokens_details(usage: Usage) -> CompletionTokensDetailsResult:
audio_tokens: Final = (
cast(
@ -860,7 +871,7 @@ def parse_completion_tokens_details(usage: Usage) -> CompletionTokensDetailsResu
)
or 0
)
text_tokens: Final = (
reported_text_tokens: Final = (
cast(
int | None,
getattr(usage.completion_tokens_details, "text_tokens", None),
@ -882,6 +893,12 @@ def parse_completion_tokens_details(usage: Usage) -> CompletionTokensDetailsResu
or 0
)
video_tokens: Final = _coerce_token_count(getattr(usage.completion_tokens_details, "video_tokens", 0))
text_tokens: Final = _text_tokens_without_nested_reasoning(
completion_tokens=usage.completion_tokens,
text_tokens=reported_text_tokens,
reasoning_tokens=reasoning_tokens,
other_modality_tokens=audio_tokens + image_tokens + video_tokens,
)
return CompletionTokensDetailsResult(
audio_tokens=audio_tokens,

View file

@ -4723,3 +4723,54 @@ def test_route_image_generation_cost_falls_back_to_requested_size(monkeypatch, r
)
assert cost == expected_cost
def test_generic_cost_per_token_bills_reasoning_nested_in_text_tokens_once(_local_model_cost_map):
"""
Realtime usage (OpenAI and Azure) reports output_tokens == text_tokens + audio_tokens with
reasoning_tokens already counted inside text_tokens, so reasoning must not be billed on top.
"""
model = "gpt-realtime-2.1-mini"
usage = Usage(
prompt_tokens=346,
completion_tokens=29,
total_tokens=375,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=152, image_tokens=194, audio_tokens=0, cached_tokens=128
),
completion_tokens_details=CompletionTokensDetailsWrapper(
text_tokens=29, audio_tokens=0, reasoning_tokens=19
),
)
prompt_cost, completion_cost = generic_cost_per_token(model=model, usage=usage, custom_llm_provider="openai")
breakdown = get_token_type_cost_breakdown(model=model, custom_llm_provider="openai", usage=usage)
info = litellm.get_model_info(model=model, custom_llm_provider="openai")
assert completion_cost == pytest.approx(29 * info["output_cost_per_token"])
assert completion_cost - breakdown.reasoning_cost == pytest.approx(10 * info["output_cost_per_token"])
assert prompt_cost == pytest.approx(
24 * info["input_cost_per_token"]
+ 128 * info["cache_read_input_token_cost"]
+ 194 * info["input_cost_per_image_token"]
)
def test_generic_cost_per_token_keeps_billing_reasoning_reported_beside_text_tokens(_local_model_cost_map):
"""Providers whose text_tokens exclude reasoning (text + reasoning == completion) stay billed in full."""
model = "gpt-realtime-2.1-mini"
usage = Usage(
prompt_tokens=100,
completion_tokens=44,
total_tokens=144,
completion_tokens_details=CompletionTokensDetailsWrapper(
text_tokens=25, audio_tokens=0, reasoning_tokens=19
),
)
_, completion_cost = generic_cost_per_token(model=model, usage=usage, custom_llm_provider="openai")
info = litellm.get_model_info(model=model, custom_llm_provider="openai")
assert completion_cost == pytest.approx(44 * info["output_cost_per_token"])

View file

@ -4492,3 +4492,45 @@ def test_batch_cost_calculator_gpt_6_astra_bills_half_the_standard_rate(_local_m
assert prompt_cost == pytest.approx(1000 * 5e-6)
assert completion_cost == pytest.approx(500 * 2.5e-5)
def test_handle_realtime_stream_cost_calculation_bills_nested_reasoning_tokens_once(_local_model_cost_map):
"""Realtime response.done nests reasoning_tokens inside text_tokens, so they are billed once."""
results: OpenAIRealtimeStreamList = [
{"type": "session.created", "session": {"model": "gpt-realtime-2.1-mini"}},
{
"type": "response.done",
"response": {
"usage": {
"total_tokens": 260,
"input_tokens": 237,
"output_tokens": 23,
"input_token_details": {
"text_tokens": 43,
"audio_tokens": 0,
"image_tokens": 194,
"cached_tokens": 0,
"cached_tokens_details": {"text_tokens": 0, "audio_tokens": 0, "image_tokens": 0},
},
"output_token_details": {"text_tokens": 23, "audio_tokens": 0, "reasoning_tokens": 18},
}
},
},
]
combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(
results=results,
)
total_cost = handle_realtime_stream_cost_calculation(
results=results,
combined_usage_object=combined_usage_object,
custom_llm_provider="azure",
litellm_model_name="azure/gpt-realtime-2.1-mini",
)
info = litellm.get_model_info(model="azure/gpt-realtime-2.1-mini", custom_llm_provider="azure")
expected = (
43 * info["input_cost_per_token"] + 194 * info["input_cost_per_image_token"] + 23 * info["output_cost_per_token"]
)
assert total_cost == pytest.approx(expected)
assert total_cost == pytest.approx(0.0002362)