mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-09 22:31:41 +00:00
fix(gemini realtime): propagate usageMetadata on tool-call response.done
Gemini Live emits usageMetadata as a sibling top-level key alongside the toolCall frame; the tool-call branch was unconditionally building response.done from get_empty_usage(), so tokens consumed by tool-call turns were recorded as zero spend and bypassed LiteLLM budget accounting. Mirror the non-tool-call RESPONSE_DONE path: when the same frame carries usageMetadata, run VertexGeminiConfig._calculate_usage and forward the real token counts.
This commit is contained in:
parent
f54874f707
commit
02aa7b2803
2 changed files with 112 additions and 5 deletions
|
|
@ -1450,12 +1450,25 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
|
|||
)
|
||||
)
|
||||
|
||||
# response.done - close the response so clients can submit tool results.
|
||||
# Include an empty usage block for parity with the non-tool-call
|
||||
# response.done path; OpenAI-compatible clients expect `usage`
|
||||
# to always be present on response.done events.
|
||||
# response.done - close the response so clients can submit tool
|
||||
# results. Mirror the non-tool-call RESPONSE_DONE path: if Gemini
|
||||
# delivered ``usageMetadata`` alongside this ``toolCall`` frame,
|
||||
# propagate the real token counts so spend/budget accounting
|
||||
# records the tokens consumed by the tool-call turn. Otherwise
|
||||
# fall back to an empty usage block (OpenAI-compatible clients
|
||||
# expect ``usage`` to always be present on response.done).
|
||||
if "usageMetadata" in json_message:
|
||||
_tool_call_chat_completion_usage = (
|
||||
VertexGeminiConfig._calculate_usage(
|
||||
completion_response=cast(
|
||||
BidiGenerateContentServerMessage, json_message
|
||||
),
|
||||
)
|
||||
)
|
||||
else:
|
||||
_tool_call_chat_completion_usage = get_empty_usage()
|
||||
tool_call_responses_api_usage = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
|
||||
get_empty_usage(),
|
||||
_tool_call_chat_completion_usage,
|
||||
)
|
||||
tool_call_done_event = OpenAIRealtimeDoneEvent(
|
||||
type="response.done",
|
||||
|
|
|
|||
|
|
@ -905,6 +905,100 @@ def test_gemini_empty_tool_call_with_sibling_usage_metadata_does_not_crash():
|
|||
assert result["current_output_item_id"] == "item_existing"
|
||||
|
||||
|
||||
def test_gemini_tool_call_response_done_includes_usage_from_sibling_metadata():
|
||||
"""A ``toolCall`` frame with a sibling ``usageMetadata`` must propagate the
|
||||
real token counts onto the emitted ``response.done`` so spend/budget
|
||||
accounting records tokens consumed by the tool-call turn — otherwise an
|
||||
authenticated client can repeatedly drive tool calls with zero spend."""
|
||||
config = GeminiRealtimeConfig()
|
||||
logging_obj = MagicMock()
|
||||
logging_obj.litellm_trace_id = "trace_tool_call_usage"
|
||||
|
||||
result = config.transform_realtime_response(
|
||||
json.dumps(
|
||||
{
|
||||
"toolCall": {
|
||||
"functionCalls": [
|
||||
{
|
||||
"id": "call_usage",
|
||||
"name": "get_weather",
|
||||
"args": {"location": "NYC"},
|
||||
}
|
||||
]
|
||||
},
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 17,
|
||||
"responseTokenCount": 4,
|
||||
"totalTokenCount": 21,
|
||||
},
|
||||
}
|
||||
),
|
||||
"gemini-2.5-flash",
|
||||
logging_obj,
|
||||
realtime_response_transform_input={
|
||||
"session_configuration_request": None,
|
||||
"current_output_item_id": None,
|
||||
"current_response_id": None,
|
||||
"current_conversation_id": None,
|
||||
"current_delta_chunks": [],
|
||||
"current_item_chunks": [],
|
||||
"current_delta_type": None,
|
||||
},
|
||||
)
|
||||
|
||||
response_done = next(
|
||||
ev for ev in result["response"] if ev.get("type") == "response.done"
|
||||
)
|
||||
usage = response_done["response"]["usage"]
|
||||
assert usage["input_tokens"] == 17
|
||||
assert usage["output_tokens"] == 4
|
||||
assert usage["total_tokens"] == 21
|
||||
|
||||
|
||||
def test_gemini_tool_call_response_done_falls_back_to_empty_usage():
|
||||
"""Without sibling ``usageMetadata`` the tool-call ``response.done`` still
|
||||
carries a valid empty usage block so OpenAI-compatible clients (which
|
||||
expect ``usage`` on every ``response.done``) don't break."""
|
||||
config = GeminiRealtimeConfig()
|
||||
logging_obj = MagicMock()
|
||||
logging_obj.litellm_trace_id = "trace_tool_call_no_usage"
|
||||
|
||||
result = config.transform_realtime_response(
|
||||
json.dumps(
|
||||
{
|
||||
"toolCall": {
|
||||
"functionCalls": [
|
||||
{
|
||||
"id": "call_no_usage",
|
||||
"name": "get_weather",
|
||||
"args": {"location": "NYC"},
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
),
|
||||
"gemini-2.5-flash",
|
||||
logging_obj,
|
||||
realtime_response_transform_input={
|
||||
"session_configuration_request": None,
|
||||
"current_output_item_id": None,
|
||||
"current_response_id": None,
|
||||
"current_conversation_id": None,
|
||||
"current_delta_chunks": [],
|
||||
"current_item_chunks": [],
|
||||
"current_delta_type": None,
|
||||
},
|
||||
)
|
||||
|
||||
response_done = next(
|
||||
ev for ev in result["response"] if ev.get("type") == "response.done"
|
||||
)
|
||||
usage = response_done["response"]["usage"]
|
||||
assert usage["input_tokens"] == 0
|
||||
assert usage["output_tokens"] == 0
|
||||
assert usage["total_tokens"] == 0
|
||||
|
||||
|
||||
def test_gemini_function_call_output_includes_name():
|
||||
"""Verify function_call_output includes name field from stored mapping."""
|
||||
config = GeminiRealtimeConfig()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue