diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 88029615ba8..cf8fa602be5 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -1567,6 +1567,7 @@ def completion_cost( # noqa: PLR0915 custom_llm_provider=custom_llm_provider, litellm_model_name=model, data_residency=data_residency, + litellm_logging_obj=litellm_logging_obj, ) elif call_type == _MCP_CALL_TYPE: from litellm.proxy._experimental.mcp_server.cost_calculator import ( @@ -2494,6 +2495,7 @@ def handle_realtime_stream_cost_calculation( custom_llm_provider: str, litellm_model_name: str, data_residency: Optional[str] = None, + litellm_logging_obj: Optional[LitellmLoggingObject] = None, ) -> float: """ Handles the cost calculation for realtime stream responses. @@ -2533,4 +2535,12 @@ def handle_realtime_stream_cost_calculation( break # exit if we find a valid model total_cost = input_cost_per_token + output_cost_per_token + _store_cost_breakdown_in_logging_obj( + litellm_logging_obj=litellm_logging_obj, + prompt_tokens_cost_usd_dollar=input_cost_per_token, + completion_tokens_cost_usd_dollar=output_cost_per_token, + cost_for_built_in_tools_cost_usd_dollar=0.0, + total_cost_usd_dollar=total_cost, + ) + return total_cost diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 82a4a60bf82..2b60cfc9ccd 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -385,7 +385,65 @@ def test_handle_realtime_stream_cost_calculation(): ) assert cost == 0.0 # No usage, no cost - + +def test_handle_realtime_stream_cost_calculation_stores_cost_breakdown(): + """Regression: realtime cost must populate logging_obj.cost_breakdown so the + spend logs / UI show input vs output cost (issue: cost_breakdown was None for + /v1/realtime even though a total spend was computed).""" + from datetime import datetime + + from litellm.litellm_core_utils.litellm_logging import Logging + + results: OpenAIRealtimeStreamList = [ + {"type": "session.created", "session": {"model": "gpt-4o-realtime-preview"}}, + { + "type": "response.done", + "response": { + "usage": { + "input_tokens": 100, + "output_tokens": 50, + "total_tokens": 150, + } + }, + }, + ] + combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results( + results=results, + ) + + logging_obj = Logging( + model="gpt-4o-realtime-preview", + messages=[], + stream=False, + call_type="_arealtime", + start_time=datetime.now(), + litellm_call_id="realtime-cost-breakdown-test", + function_id="realtime-cost-breakdown-test", + ) + + total_cost = handle_realtime_stream_cost_calculation( + results=results, + combined_usage_object=combined_usage_object, + custom_llm_provider="openai", + litellm_model_name="gpt-4o-realtime-preview", + litellm_logging_obj=logging_obj, + ) + + assert total_cost > 0 + assert logging_obj.cost_breakdown is not None + assert logging_obj.cost_breakdown["input_cost"] > 0 + assert logging_obj.cost_breakdown["output_cost"] > 0 + assert ( + abs( + logging_obj.cost_breakdown["input_cost"] + + logging_obj.cost_breakdown["output_cost"] + - total_cost + ) + < 1e-9 + ) + assert abs(logging_obj.cost_breakdown["total_cost"] - total_cost) < 1e-9 + + def test_realtime_stream_combines_text_and_audio_token_details(): """Realtime response.done usage with input_token_details / output_token_details.""" from litellm.cost_calculator import RealtimeAPITokenUsageProcessor