From 9efcc02776860b381211d07d619a5cc8134ed91f Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Fri, 22 May 2026 21:06:06 +0000 Subject: [PATCH] fix(realtime): correct conversation_id, VAD disable, modality state, empty toolCall - Gemini tool-call response.done now includes conversation_id so clients can match it against the preceding response.created. - Vertex AI setup no longer overrides an explicit guardrail-injected create_response: False back to disabled: False; the guardrail's intent to disable VAD auto-response is now respected. - Modality handler is now passed the locally-updated response/item IDs rather than the original input snapshot, preventing stale IDs after a prior tool-call/response.done in the same JSON message resets them. - Skip emitting orphaned response.created/response.done events when Gemini sends an empty functionCalls array. Co-authored-by: Yassin Kortam --- .../llms/gemini/realtime/transformation.py | 26 +++++++++++++++++-- .../llms/vertex_ai/realtime/transformation.py | 20 +++++++++----- 2 files changed, 37 insertions(+), 9 deletions(-) diff --git a/litellm/llms/gemini/realtime/transformation.py b/litellm/llms/gemini/realtime/transformation.py index a1b0a965db9..af3cf5579a7 100644 --- a/litellm/llms/gemini/realtime/transformation.py +++ b/litellm/llms/gemini/realtime/transformation.py @@ -1227,7 +1227,13 @@ class GeminiRealtimeConfig(BaseRealtimeConfig): ) returned_message.append(transformed_message) elif openai_event == ResponsesAPIStreamEvents.FUNCTION_CALL_ARGUMENTS_DONE: - # Handle toolCall from Gemini + # Handle toolCall from Gemini. Skip entirely if there are no + # function calls in the payload — emitting an orphaned + # response.created/response.done pair with no output items + # would confuse OpenAI-compatible clients. + if not value.get("functionCalls"): + continue + # Emit response.created preamble if this is the first event in the response if current_response_id is None: current_response_id = f"resp_{uuid.uuid4()}" @@ -1323,6 +1329,7 @@ class GeminiRealtimeConfig(BaseRealtimeConfig): } for te in tool_call_events ], + conversation_id=current_conversation_id, ), ) ) @@ -1351,10 +1358,25 @@ class GeminiRealtimeConfig(BaseRealtimeConfig): or openai_event == OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DELTA or openai_event == OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DONE ): + # Pass the locally-updated state (rather than the original + # input snapshot) so that prior iterations of this loop — + # e.g. a tool-call or response.done that just reset + # current_response_id/current_output_item_id to None — are + # honoured by the modality handler. + _modality_input: RealtimeResponseTransformInput = { + **realtime_response_transform_input, + "current_output_item_id": current_output_item_id, + "current_response_id": current_response_id, + "current_conversation_id": current_conversation_id, + "current_delta_chunks": current_delta_chunks, + "current_item_chunks": current_item_chunks, + "current_delta_type": current_delta_type, + "session_configuration_request": session_configuration_request, + } _returned_message = self.handle_openai_modality_event( openai_event, json_message, - realtime_response_transform_input, + _modality_input, delta_type="text" if "text" in openai_event.value else "audio", ) returned_message.extend(_returned_message["returned_message"]) diff --git a/litellm/llms/vertex_ai/realtime/transformation.py b/litellm/llms/vertex_ai/realtime/transformation.py index a739ef8678d..d2d89a52c8d 100644 --- a/litellm/llms/vertex_ai/realtime/transformation.py +++ b/litellm/llms/vertex_ai/realtime/transformation.py @@ -172,17 +172,23 @@ class VertexAIRealtimeConfig(GeminiRealtimeConfig): # the client provided a partial ``turn_detection`` (e.g. only # ``silence_duration_ms``). ``map_automatic_turn_detection`` sets # ``disabled=True`` whenever ``create_response`` is absent or - # ``False`` — the latter being how transcription guardrails - # suppress automatic responses — which would otherwise silently - # disable VAD here and break speech detection / transcription - # events. Vertex Live has no "VAD on, no auto-response" mode, so - # always keep VAD active; ``create_response: True`` already maps - # to ``disabled=False`` and is therefore unaffected. + # ``False``. Force ``disabled=False`` only when the client did + # not explicitly request ``create_response: False`` — that path + # is how transcription guardrails suppress automatic responses, + # and overriding it here would silently bypass the guardrail. + # Vertex Live has no "VAD on, no auto-response" mode, so callers + # that need that behaviour must accept that VAD is off. + client_turn_detection = session_params.get("turn_detection") + client_disabled_auto_response = ( + isinstance(client_turn_detection, dict) + and client_turn_detection.get("create_response") is False + ) realtime_input_config = setup_config.setdefault("realtimeInputConfig", {}) automatic_detection = realtime_input_config.setdefault( "automaticActivityDetection", {} ) - automatic_detection["disabled"] = False + if not client_disabled_auto_response: + automatic_detection["disabled"] = False automatic_detection.setdefault("silenceDurationMs", 800) setup_config.setdefault("inputAudioTranscription", {})