fix(gemini realtime): buffer standalone usageMetadata for next response.done
Some checks failed
Unit Tests: Proxy DB Operations / assert-shard-coverage (push) Has been cancelled
Unit Tests: Security / security (push) Has been cancelled
Unit Tests: Proxy DB Operations / auth-checks (push) Has been cancelled
Unit Tests: Proxy DB Operations / budgets (push) Has been cancelled
Unit Tests: Proxy DB Operations / custom-logging (push) Has been cancelled
Unit Tests: Proxy DB Operations / db-and-spend (push) Has been cancelled
Unit Tests: Proxy DB Operations / endpoints-and-responses (push) Has been cancelled
Unit Tests: Proxy DB Operations / guardrails-hooks (push) Has been cancelled
Unit Tests: Proxy DB Operations / jwt-and-keys (push) Has been cancelled
Unit Tests: Proxy DB Operations / key-generation (push) Has been cancelled
Unit Tests: Proxy DB Operations / logging-misc (push) Has been cancelled
Unit Tests: Proxy DB Operations / proxy-runtime (push) Has been cancelled
Unit Tests: Proxy DB Operations / proxy-server-core (push) Has been cancelled
Unit Tests: Proxy DB Operations / schema-migration (push) Has been cancelled
Unit Tests: Proxy DB Operations / proxy-utils (push) Has been cancelled

Gemini Live can emit usageMetadata as a standalone WebSocket frame between
turns. The previous transformer treated those frames as no-ops, so token
counts arriving outside the closing turnComplete/toolCall frame were
dropped from spend and budget accounting. An authenticated client could
drive turns whose usage was recorded as zero, bypassing budgets.

Buffer any standalone usageMetadata on the config instance and attribute
the deferred counts to the next emitted response.done (tool-call or
normal). In-frame usageMetadata remains authoritative and clears the
buffer.
This commit is contained in:
mateo-berri 2026-05-23 20:57:14 +00:00
parent a8b592b634
commit b102bc3512
No known key found for this signature in database
2 changed files with 252 additions and 7 deletions

View file

@ -91,6 +91,12 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
super().__init__()
# Store call_id → function_name mapping for tool call round-trip
self._tool_call_id_to_name: "OrderedDict[str, str]" = OrderedDict()
# Buffer ``usageMetadata`` that Gemini Live emits as a standalone
# frame (between turns) so the next ``response.done`` attributes the
# tokens consumed. Without this an authenticated client can drive
# tool-call or normal turns whose token usage is recorded as zero,
# bypassing spend and budget accounting.
self._pending_usage_metadata: Optional[dict] = None
def validate_environment(
self, headers: dict, model: str, api_key: Optional[str] = None
@ -855,6 +861,32 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
returned_items.append(response_output_item_done)
return returned_items
def _consume_usage_metadata_for_response_done(self, frame: dict) -> Optional[dict]:
"""Return the ``usageMetadata`` to attribute to a ``response.done``.
Gemini Live emits ``usageMetadata`` either alongside the closing
frame (``serverContent.turnComplete`` / ``toolCall``) or as a
standalone frame between turns. The standalone form would otherwise
be discarded by the no-op branch in ``transform_realtime_response``
and the consumed tokens silently dropped from spend/budget
accounting. ``_pending_usage_metadata`` buffers any such standalone
frames so the next emitted ``response.done`` carries the deferred
token counts.
Returns the in-frame ``usageMetadata`` if present (and clears the
buffer since the in-frame counts are the authoritative attribution
for this turn), otherwise returns the buffered counts. ``None`` is
returned when neither is available so the caller can fall back to
``get_empty_usage()``.
"""
in_frame = frame.get("usageMetadata") if isinstance(frame, dict) else None
if isinstance(in_frame, dict):
self._pending_usage_metadata = None
return in_frame
buffered = self._pending_usage_metadata
self._pending_usage_metadata = None
return buffered
def transform_tool_call_events(
self,
tool_call_message: dict,
@ -1014,9 +1046,15 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
_modalities = [
modality.lower() for modality in cast(List[str], gemini_modalities)
]
if "usageMetadata" in message:
resolved_usage_metadata = self._consume_usage_metadata_for_response_done(
cast(dict, message)
)
if resolved_usage_metadata is not None:
_chat_completion_usage = VertexGeminiConfig._calculate_usage(
completion_response=message,
completion_response=cast(
BidiGenerateContentServerMessage,
{**cast(dict, message), "usageMetadata": resolved_usage_metadata},
),
)
else:
_chat_completion_usage = get_empty_usage()
@ -1462,14 +1500,26 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
# results. Mirror the non-tool-call RESPONSE_DONE path: if Gemini
# delivered ``usageMetadata`` alongside this ``toolCall`` frame,
# propagate the real token counts so spend/budget accounting
# records the tokens consumed by the tool-call turn. Otherwise
# fall back to an empty usage block (OpenAI-compatible clients
# expect ``usage`` to always be present on response.done).
if "usageMetadata" in json_message:
# records the tokens consumed by the tool-call turn. Standalone
# ``usageMetadata`` frames emitted in a separate WebSocket frame
# are buffered on the instance so the next ``response.done``
# picks them up (otherwise an authenticated client could drive
# tool-call turns whose token usage is recorded as zero,
# bypassing budgets). Falls back to an empty usage block when
# neither is available (OpenAI-compatible clients expect
# ``usage`` to always be present on response.done).
resolved_tool_call_usage_metadata = (
self._consume_usage_metadata_for_response_done(json_message)
)
if resolved_tool_call_usage_metadata is not None:
_tool_call_chat_completion_usage = (
VertexGeminiConfig._calculate_usage(
completion_response=cast(
BidiGenerateContentServerMessage, json_message
BidiGenerateContentServerMessage,
{
**json_message,
"usageMetadata": resolved_tool_call_usage_metadata,
},
),
)
)
@ -1586,6 +1636,14 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
and not (key == "serverContent" and server_content_handled)
and not (key == "toolCall" and tool_call_handled)
]
# Buffer standalone usage metadata so the next response.done can
# attribute the token counts. Without this, an authenticated
# client driving turns whose usageMetadata is emitted in a
# separate frame would have those tokens recorded as zero spend,
# bypassing budget enforcement.
standalone_usage_metadata = json_message.get("usageMetadata")
if isinstance(standalone_usage_metadata, dict):
self._pending_usage_metadata = standalone_usage_metadata
if not unhandled_known_keys:
return {
"response": returned_message,

View file

@ -1317,3 +1317,190 @@ def test_gemini_standalone_usage_metadata_does_not_crash_websocket():
assert result["current_output_item_id"] == "item_existing"
assert result["current_response_id"] == "resp_existing"
assert result["current_conversation_id"] == "conv_existing"
def test_gemini_standalone_usage_metadata_is_attributed_to_next_tool_call_response_done():
"""A standalone ``usageMetadata`` frame emitted between turns must not
silently drop the consumed tokens. The next tool-call ``response.done``
must carry those token counts so an authenticated client cannot drive
tool-call turns whose token usage is recorded as zero, bypassing
spend/budget accounting."""
config = GeminiRealtimeConfig()
logging_obj = MagicMock()
logging_obj.litellm_trace_id = "trace_standalone_usage_then_tool_call"
standalone_result = config.transform_realtime_response(
json.dumps(
{
"usageMetadata": {
"promptTokenCount": 31,
"responseTokenCount": 9,
"totalTokenCount": 40,
}
}
),
"gemini-2.5-flash",
logging_obj,
realtime_response_transform_input={
"session_configuration_request": None,
"current_output_item_id": None,
"current_response_id": None,
"current_conversation_id": None,
"current_delta_chunks": [],
"current_item_chunks": [],
"current_delta_type": None,
},
)
assert standalone_result["response"] == []
tool_call_result = config.transform_realtime_response(
json.dumps(
{
"toolCall": {
"functionCalls": [
{
"id": "call_buffered",
"name": "get_weather",
"args": {"location": "NYC"},
}
]
}
}
),
"gemini-2.5-flash",
logging_obj,
realtime_response_transform_input={
"session_configuration_request": None,
"current_output_item_id": None,
"current_response_id": None,
"current_conversation_id": None,
"current_delta_chunks": [],
"current_item_chunks": [],
"current_delta_type": None,
},
)
response_done = next(
ev for ev in tool_call_result["response"] if ev.get("type") == "response.done"
)
usage = response_done["response"]["usage"]
assert usage["input_tokens"] == 31
assert usage["output_tokens"] == 9
assert usage["total_tokens"] == 40
# Buffer must be cleared after attribution so a subsequent tool-call
# turn without its own usage does not double-count the previous frame.
assert config._pending_usage_metadata is None
def test_gemini_standalone_usage_metadata_is_attributed_to_next_response_done():
"""A standalone ``usageMetadata`` frame must also flow into the normal
(non-tool-call) ``response.done`` path so audio/text turns whose usage
arrives in a separate frame are still billed correctly."""
config = GeminiRealtimeConfig()
logging_obj = MagicMock()
logging_obj.litellm_trace_id = "trace_standalone_usage_then_turn_complete"
config.transform_realtime_response(
json.dumps(
{
"usageMetadata": {
"promptTokenCount": 5,
"responseTokenCount": 11,
"totalTokenCount": 16,
}
}
),
"gemini-2.5-flash",
logging_obj,
realtime_response_transform_input={
"session_configuration_request": None,
"current_output_item_id": None,
"current_response_id": None,
"current_conversation_id": None,
"current_delta_chunks": [],
"current_item_chunks": [],
"current_delta_type": None,
},
)
turn_complete_result = config.transform_realtime_response(
json.dumps({"serverContent": {"turnComplete": True}}),
"gemini-2.5-flash",
logging_obj,
realtime_response_transform_input={
"session_configuration_request": None,
"current_output_item_id": None,
"current_response_id": None,
"current_conversation_id": None,
"current_delta_chunks": [],
"current_item_chunks": [],
"current_delta_type": None,
},
)
response_done = next(
ev
for ev in turn_complete_result["response"]
if ev.get("type") == "response.done"
)
usage = response_done["response"]["usage"]
assert usage["input_tokens"] == 5
assert usage["output_tokens"] == 11
assert usage["total_tokens"] == 16
assert config._pending_usage_metadata is None
def test_gemini_in_frame_usage_metadata_clears_pending_buffer():
"""When ``usageMetadata`` arrives in the same frame as the closing
``toolCall`` / ``turnComplete``, the in-frame counts are authoritative
and any buffered standalone metadata must be discarded so a later
turn's ``response.done`` does not double-count tokens."""
config = GeminiRealtimeConfig()
config._pending_usage_metadata = {
"promptTokenCount": 99,
"responseTokenCount": 99,
"totalTokenCount": 198,
}
logging_obj = MagicMock()
logging_obj.litellm_trace_id = "trace_in_frame_clears_buffer"
result = config.transform_realtime_response(
json.dumps(
{
"toolCall": {
"functionCalls": [
{
"id": "call_in_frame",
"name": "get_weather",
"args": {"location": "NYC"},
}
]
},
"usageMetadata": {
"promptTokenCount": 3,
"responseTokenCount": 2,
"totalTokenCount": 5,
},
}
),
"gemini-2.5-flash",
logging_obj,
realtime_response_transform_input={
"session_configuration_request": None,
"current_output_item_id": None,
"current_response_id": None,
"current_conversation_id": None,
"current_delta_chunks": [],
"current_item_chunks": [],
"current_delta_type": None,
},
)
response_done = next(
ev for ev in result["response"] if ev.get("type") == "response.done"
)
usage = response_done["response"]["usage"]
assert usage["input_tokens"] == 3
assert usage["output_tokens"] == 2
assert usage["total_tokens"] == 5
assert config._pending_usage_metadata is None