From 02aa7b2803681181c05687acb2bd0642f30ffb95 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 23 May 2026 03:47:21 +0000 Subject: [PATCH] fix(gemini realtime): propagate usageMetadata on tool-call response.done Gemini Live emits usageMetadata as a sibling top-level key alongside the toolCall frame; the tool-call branch was unconditionally building response.done from get_empty_usage(), so tokens consumed by tool-call turns were recorded as zero spend and bypassed LiteLLM budget accounting. Mirror the non-tool-call RESPONSE_DONE path: when the same frame carries usageMetadata, run VertexGeminiConfig._calculate_usage and forward the real token counts. --- .../llms/gemini/realtime/transformation.py | 23 ++++- .../test_gemini_realtime_transformation.py | 94 +++++++++++++++++++ 2 files changed, 112 insertions(+), 5 deletions(-) diff --git a/litellm/llms/gemini/realtime/transformation.py b/litellm/llms/gemini/realtime/transformation.py index c5a02df499f..2346d3b7a73 100644 --- a/litellm/llms/gemini/realtime/transformation.py +++ b/litellm/llms/gemini/realtime/transformation.py @@ -1450,12 +1450,25 @@ class GeminiRealtimeConfig(BaseRealtimeConfig): ) ) - # response.done - close the response so clients can submit tool results. - # Include an empty usage block for parity with the non-tool-call - # response.done path; OpenAI-compatible clients expect `usage` - # to always be present on response.done events. + # response.done - close the response so clients can submit tool + # results. Mirror the non-tool-call RESPONSE_DONE path: if Gemini + # delivered ``usageMetadata`` alongside this ``toolCall`` frame, + # propagate the real token counts so spend/budget accounting + # records the tokens consumed by the tool-call turn. Otherwise + # fall back to an empty usage block (OpenAI-compatible clients + # expect ``usage`` to always be present on response.done). + if "usageMetadata" in json_message: + _tool_call_chat_completion_usage = ( + VertexGeminiConfig._calculate_usage( + completion_response=cast( + BidiGenerateContentServerMessage, json_message + ), + ) + ) + else: + _tool_call_chat_completion_usage = get_empty_usage() tool_call_responses_api_usage = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage( - get_empty_usage(), + _tool_call_chat_completion_usage, ) tool_call_done_event = OpenAIRealtimeDoneEvent( type="response.done", diff --git a/tests/test_litellm/llms/gemini/realtime/test_gemini_realtime_transformation.py b/tests/test_litellm/llms/gemini/realtime/test_gemini_realtime_transformation.py index 122eb3fe69a..d25a20c9f25 100644 --- a/tests/test_litellm/llms/gemini/realtime/test_gemini_realtime_transformation.py +++ b/tests/test_litellm/llms/gemini/realtime/test_gemini_realtime_transformation.py @@ -905,6 +905,100 @@ def test_gemini_empty_tool_call_with_sibling_usage_metadata_does_not_crash(): assert result["current_output_item_id"] == "item_existing" +def test_gemini_tool_call_response_done_includes_usage_from_sibling_metadata(): + """A ``toolCall`` frame with a sibling ``usageMetadata`` must propagate the + real token counts onto the emitted ``response.done`` so spend/budget + accounting records tokens consumed by the tool-call turn — otherwise an + authenticated client can repeatedly drive tool calls with zero spend.""" + config = GeminiRealtimeConfig() + logging_obj = MagicMock() + logging_obj.litellm_trace_id = "trace_tool_call_usage" + + result = config.transform_realtime_response( + json.dumps( + { + "toolCall": { + "functionCalls": [ + { + "id": "call_usage", + "name": "get_weather", + "args": {"location": "NYC"}, + } + ] + }, + "usageMetadata": { + "promptTokenCount": 17, + "responseTokenCount": 4, + "totalTokenCount": 21, + }, + } + ), + "gemini-2.5-flash", + logging_obj, + realtime_response_transform_input={ + "session_configuration_request": None, + "current_output_item_id": None, + "current_response_id": None, + "current_conversation_id": None, + "current_delta_chunks": [], + "current_item_chunks": [], + "current_delta_type": None, + }, + ) + + response_done = next( + ev for ev in result["response"] if ev.get("type") == "response.done" + ) + usage = response_done["response"]["usage"] + assert usage["input_tokens"] == 17 + assert usage["output_tokens"] == 4 + assert usage["total_tokens"] == 21 + + +def test_gemini_tool_call_response_done_falls_back_to_empty_usage(): + """Without sibling ``usageMetadata`` the tool-call ``response.done`` still + carries a valid empty usage block so OpenAI-compatible clients (which + expect ``usage`` on every ``response.done``) don't break.""" + config = GeminiRealtimeConfig() + logging_obj = MagicMock() + logging_obj.litellm_trace_id = "trace_tool_call_no_usage" + + result = config.transform_realtime_response( + json.dumps( + { + "toolCall": { + "functionCalls": [ + { + "id": "call_no_usage", + "name": "get_weather", + "args": {"location": "NYC"}, + } + ] + } + } + ), + "gemini-2.5-flash", + logging_obj, + realtime_response_transform_input={ + "session_configuration_request": None, + "current_output_item_id": None, + "current_response_id": None, + "current_conversation_id": None, + "current_delta_chunks": [], + "current_item_chunks": [], + "current_delta_type": None, + }, + ) + + response_done = next( + ev for ev in result["response"] if ev.get("type") == "response.done" + ) + usage = response_done["response"]["usage"] + assert usage["input_tokens"] == 0 + assert usage["output_tokens"] == 0 + assert usage["total_tokens"] == 0 + + def test_gemini_function_call_output_includes_name(): """Verify function_call_output includes name field from stored mapping.""" config = GeminiRealtimeConfig()