fix(cost): store cost breakdown for /v1/realtime sessions

Realtime cost calculation computed totals but never populated logging_obj.cost_breakdown, so spend logs and the UI Metrics/Cost Breakdown showed no input/output cost details.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
shivam 2026-06-09 17:10:15 -07:00
parent 50522157dc
commit 5e2df556d8
No known key found for this signature in database
2 changed files with 69 additions and 1 deletions

View file

@ -1567,6 +1567,7 @@ def completion_cost( # noqa: PLR0915
custom_llm_provider=custom_llm_provider,
litellm_model_name=model,
data_residency=data_residency,
litellm_logging_obj=litellm_logging_obj,
)
elif call_type == _MCP_CALL_TYPE:
from litellm.proxy._experimental.mcp_server.cost_calculator import (
@ -2494,6 +2495,7 @@ def handle_realtime_stream_cost_calculation(
custom_llm_provider: str,
litellm_model_name: str,
data_residency: Optional[str] = None,
litellm_logging_obj: Optional[LitellmLoggingObject] = None,
) -> float:
"""
Handles the cost calculation for realtime stream responses.
@ -2533,4 +2535,12 @@ def handle_realtime_stream_cost_calculation(
break # exit if we find a valid model
total_cost = input_cost_per_token + output_cost_per_token
_store_cost_breakdown_in_logging_obj(
litellm_logging_obj=litellm_logging_obj,
prompt_tokens_cost_usd_dollar=input_cost_per_token,
completion_tokens_cost_usd_dollar=output_cost_per_token,
cost_for_built_in_tools_cost_usd_dollar=0.0,
total_cost_usd_dollar=total_cost,
)
return total_cost

View file

@ -385,7 +385,65 @@ def test_handle_realtime_stream_cost_calculation():
)
assert cost == 0.0 # No usage, no cost
def test_handle_realtime_stream_cost_calculation_stores_cost_breakdown():
"""Regression: realtime cost must populate logging_obj.cost_breakdown so the
spend logs / UI show input vs output cost (issue: cost_breakdown was None for
/v1/realtime even though a total spend was computed)."""
from datetime import datetime
from litellm.litellm_core_utils.litellm_logging import Logging
results: OpenAIRealtimeStreamList = [
{"type": "session.created", "session": {"model": "gpt-4o-realtime-preview"}},
{
"type": "response.done",
"response": {
"usage": {
"input_tokens": 100,
"output_tokens": 50,
"total_tokens": 150,
}
},
},
]
combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(
results=results,
)
logging_obj = Logging(
model="gpt-4o-realtime-preview",
messages=[],
stream=False,
call_type="_arealtime",
start_time=datetime.now(),
litellm_call_id="realtime-cost-breakdown-test",
function_id="realtime-cost-breakdown-test",
)
total_cost = handle_realtime_stream_cost_calculation(
results=results,
combined_usage_object=combined_usage_object,
custom_llm_provider="openai",
litellm_model_name="gpt-4o-realtime-preview",
litellm_logging_obj=logging_obj,
)
assert total_cost > 0
assert logging_obj.cost_breakdown is not None
assert logging_obj.cost_breakdown["input_cost"] > 0
assert logging_obj.cost_breakdown["output_cost"] > 0
assert (
abs(
logging_obj.cost_breakdown["input_cost"]
+ logging_obj.cost_breakdown["output_cost"]
- total_cost
)
< 1e-9
)
assert abs(logging_obj.cost_breakdown["total_cost"] - total_cost) < 1e-9
def test_realtime_stream_combines_text_and_audio_token_details():
"""Realtime response.done usage with input_token_details / output_token_details."""
from litellm.cost_calculator import RealtimeAPITokenUsageProcessor