From 5e2df556d8065de5d8baf66e391d089570d88b05 Mon Sep 17 00:00:00 2001 From: shivam Date: Tue, 9 Jun 2026 17:10:15 -0700 Subject: [PATCH 1/3] fix(cost): store cost breakdown for /v1/realtime sessions Realtime cost calculation computed totals but never populated logging_obj.cost_breakdown, so spend logs and the UI Metrics/Cost Breakdown showed no input/output cost details. Co-authored-by: Cursor --- litellm/cost_calculator.py | 10 ++++ tests/test_litellm/test_cost_calculator.py | 60 +++++++++++++++++++++- 2 files changed, 69 insertions(+), 1 deletion(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 88029615ba8..cf8fa602be5 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -1567,6 +1567,7 @@ def completion_cost( # noqa: PLR0915 custom_llm_provider=custom_llm_provider, litellm_model_name=model, data_residency=data_residency, + litellm_logging_obj=litellm_logging_obj, ) elif call_type == _MCP_CALL_TYPE: from litellm.proxy._experimental.mcp_server.cost_calculator import ( @@ -2494,6 +2495,7 @@ def handle_realtime_stream_cost_calculation( custom_llm_provider: str, litellm_model_name: str, data_residency: Optional[str] = None, + litellm_logging_obj: Optional[LitellmLoggingObject] = None, ) -> float: """ Handles the cost calculation for realtime stream responses. @@ -2533,4 +2535,12 @@ def handle_realtime_stream_cost_calculation( break # exit if we find a valid model total_cost = input_cost_per_token + output_cost_per_token + _store_cost_breakdown_in_logging_obj( + litellm_logging_obj=litellm_logging_obj, + prompt_tokens_cost_usd_dollar=input_cost_per_token, + completion_tokens_cost_usd_dollar=output_cost_per_token, + cost_for_built_in_tools_cost_usd_dollar=0.0, + total_cost_usd_dollar=total_cost, + ) + return total_cost diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 82a4a60bf82..2b60cfc9ccd 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -385,7 +385,65 @@ def test_handle_realtime_stream_cost_calculation(): ) assert cost == 0.0 # No usage, no cost - + +def test_handle_realtime_stream_cost_calculation_stores_cost_breakdown(): + """Regression: realtime cost must populate logging_obj.cost_breakdown so the + spend logs / UI show input vs output cost (issue: cost_breakdown was None for + /v1/realtime even though a total spend was computed).""" + from datetime import datetime + + from litellm.litellm_core_utils.litellm_logging import Logging + + results: OpenAIRealtimeStreamList = [ + {"type": "session.created", "session": {"model": "gpt-4o-realtime-preview"}}, + { + "type": "response.done", + "response": { + "usage": { + "input_tokens": 100, + "output_tokens": 50, + "total_tokens": 150, + } + }, + }, + ] + combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results( + results=results, + ) + + logging_obj = Logging( + model="gpt-4o-realtime-preview", + messages=[], + stream=False, + call_type="_arealtime", + start_time=datetime.now(), + litellm_call_id="realtime-cost-breakdown-test", + function_id="realtime-cost-breakdown-test", + ) + + total_cost = handle_realtime_stream_cost_calculation( + results=results, + combined_usage_object=combined_usage_object, + custom_llm_provider="openai", + litellm_model_name="gpt-4o-realtime-preview", + litellm_logging_obj=logging_obj, + ) + + assert total_cost > 0 + assert logging_obj.cost_breakdown is not None + assert logging_obj.cost_breakdown["input_cost"] > 0 + assert logging_obj.cost_breakdown["output_cost"] > 0 + assert ( + abs( + logging_obj.cost_breakdown["input_cost"] + + logging_obj.cost_breakdown["output_cost"] + - total_cost + ) + < 1e-9 + ) + assert abs(logging_obj.cost_breakdown["total_cost"] - total_cost) < 1e-9 + + def test_realtime_stream_combines_text_and_audio_token_details(): """Realtime response.done usage with input_token_details / output_token_details.""" from litellm.cost_calculator import RealtimeAPITokenUsageProcessor From 09f611e7b50d93ea224ab51053619ed67def4f7b Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Thu, 2 Jul 2026 17:03:05 +0000 Subject: [PATCH 2/3] fix: include realtime transcription cost in breakdown --- litellm/cost_calculator.py | 18 ++++++++++-------- tests/test_litellm/test_cost_calculator.py | 16 ++++++++++++++++ 2 files changed, 26 insertions(+), 8 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index bfdce95d89a..9d9564a7110 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -2564,7 +2564,16 @@ def handle_realtime_stream_cost_calculation( input_cost_per_token += _input_cost_per_token output_cost_per_token += _output_cost_per_token break # exit if we find a valid model - total_cost = input_cost_per_token + output_cost_per_token + transcription_cost = ( + handle_realtime_transcription_cost_calculation( + results=results, + custom_llm_provider=custom_llm_provider, + litellm_model_name=litellm_model_name, + ) + if any(r.get("type") == _TRANSCRIPTION_COMPLETED_EVENT_TYPE for r in results) + else 0.0 + ) + total_cost = input_cost_per_token + output_cost_per_token + transcription_cost _store_cost_breakdown_in_logging_obj( litellm_logging_obj=litellm_logging_obj, @@ -2574,13 +2583,6 @@ def handle_realtime_stream_cost_calculation( total_cost_usd_dollar=total_cost, ) - if any(r.get("type") == _TRANSCRIPTION_COMPLETED_EVENT_TYPE for r in results): - total_cost += handle_realtime_transcription_cost_calculation( - results=results, - custom_llm_provider=custom_llm_provider, - litellm_model_name=litellm_model_name, - ) - return total_cost diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 4ba64f13b69..3a5d43b0300 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -637,6 +637,10 @@ def test_realtime_transcription_duration_cost(monkeypatch): ($0.017/min). The .completed events carry usage {type: duration, seconds: N}; cost must equal total_seconds * input_cost_per_second. """ + from datetime import datetime + + from litellm.litellm_core_utils.litellm_logging import Logging + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) @@ -667,17 +671,29 @@ def test_realtime_transcription_duration_cost(monkeypatch): combined = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results( results=results ) + logging_obj = Logging( + model="gpt-realtime-whisper", + messages=[], + stream=False, + call_type="_arealtime", + start_time=datetime.now(), + litellm_call_id="realtime-transcription-cost-breakdown-test", + function_id="realtime-transcription-cost-breakdown-test", + ) cost = handle_realtime_stream_cost_calculation( results=results, combined_usage_object=combined, custom_llm_provider="openai", litellm_model_name="gpt-realtime-whisper", + litellm_logging_obj=logging_obj, ) # 90 seconds at $0.017/minute. expected = 90.0 * (0.017 / 60) assert abs(cost - expected) < 1e-9 assert cost > 0 # guards against the duration branch being dropped + assert logging_obj.cost_breakdown is not None + assert abs(logging_obj.cost_breakdown["total_cost"] - cost) < 1e-9 def test_realtime_transcription_duration_cost_resolves_model_from_litellm_name( From f0d41e4d1696951dfb18e2fa48e7d67be00faf76 Mon Sep 17 00:00:00 2001 From: Shivam Rawat Date: Sat, 4 Jul 2026 12:22:53 -0700 Subject: [PATCH 3/3] fix: attribute realtime transcription cost in cost breakdown Pass transcription_cost through additional_costs so cost_breakdown's input_cost + output_cost + additional_costs sums to total_cost instead of silently folding it into total_cost only. Co-authored-by: Cursor --- litellm/cost_calculator.py | 1 + tests/test_litellm/test_cost_calculator.py | 12 ++++++++++++ 2 files changed, 13 insertions(+) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index 243060ee46e..2cb94e02d9c 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -2356,6 +2356,7 @@ def handle_realtime_stream_cost_calculation( completion_tokens_cost_usd_dollar=output_cost_per_token, cost_for_built_in_tools_cost_usd_dollar=0.0, total_cost_usd_dollar=total_cost, + additional_costs={"transcription_cost": transcription_cost} if transcription_cost > 0 else None, ) return total_cost diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 41feec83a1c..0198a396b9b 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -695,6 +695,18 @@ def test_realtime_transcription_duration_cost(monkeypatch): assert logging_obj.cost_breakdown is not None assert abs(logging_obj.cost_breakdown["total_cost"] - cost) < 1e-9 + # The transcription cost must be attributed in the breakdown, not just folded + # into total_cost, or input_cost + output_cost + additional_costs won't sum to total_cost. + additional_costs = logging_obj.cost_breakdown.get("additional_costs") + assert additional_costs is not None + assert abs(additional_costs["transcription_cost"] - expected) < 1e-9 + attributed_total = ( + logging_obj.cost_breakdown["input_cost"] + + logging_obj.cost_breakdown["output_cost"] + + additional_costs["transcription_cost"] + ) + assert abs(attributed_total - logging_obj.cost_breakdown["total_cost"]) < 1e-9 + def test_realtime_transcription_duration_cost_resolves_model_from_litellm_name( monkeypatch,