fix(realtime): correct conversation_id, VAD disable, modality state, empty toolCall

- Gemini tool-call response.done now includes conversation_id so clients
  can match it against the preceding response.created.
- Vertex AI setup no longer overrides an explicit guardrail-injected
  create_response: False back to disabled: False; the guardrail's intent
  to disable VAD auto-response is now respected.
- Modality handler is now passed the locally-updated response/item IDs
  rather than the original input snapshot, preventing stale IDs after a
  prior tool-call/response.done in the same JSON message resets them.
- Skip emitting orphaned response.created/response.done events when
  Gemini sends an empty functionCalls array.

Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
Cursor Agent 2026-05-22 21:06:06 +00:00
parent 24b8e17a4b
commit 9efcc02776
No known key found for this signature in database
2 changed files with 37 additions and 9 deletions

View file

@ -1227,7 +1227,13 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
)
returned_message.append(transformed_message)
elif openai_event == ResponsesAPIStreamEvents.FUNCTION_CALL_ARGUMENTS_DONE:
# Handle toolCall from Gemini
# Handle toolCall from Gemini. Skip entirely if there are no
# function calls in the payload — emitting an orphaned
# response.created/response.done pair with no output items
# would confuse OpenAI-compatible clients.
if not value.get("functionCalls"):
continue
# Emit response.created preamble if this is the first event in the response
if current_response_id is None:
current_response_id = f"resp_{uuid.uuid4()}"
@ -1323,6 +1329,7 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
}
for te in tool_call_events
],
conversation_id=current_conversation_id,
),
)
)
@ -1351,10 +1358,25 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
or openai_event == OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DELTA
or openai_event == OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DONE
):
# Pass the locally-updated state (rather than the original
# input snapshot) so that prior iterations of this loop —
# e.g. a tool-call or response.done that just reset
# current_response_id/current_output_item_id to None — are
# honoured by the modality handler.
_modality_input: RealtimeResponseTransformInput = {
**realtime_response_transform_input,
"current_output_item_id": current_output_item_id,
"current_response_id": current_response_id,
"current_conversation_id": current_conversation_id,
"current_delta_chunks": current_delta_chunks,
"current_item_chunks": current_item_chunks,
"current_delta_type": current_delta_type,
"session_configuration_request": session_configuration_request,
}
_returned_message = self.handle_openai_modality_event(
openai_event,
json_message,
realtime_response_transform_input,
_modality_input,
delta_type="text" if "text" in openai_event.value else "audio",
)
returned_message.extend(_returned_message["returned_message"])

View file

@ -172,17 +172,23 @@ class VertexAIRealtimeConfig(GeminiRealtimeConfig):
# the client provided a partial ``turn_detection`` (e.g. only
# ``silence_duration_ms``). ``map_automatic_turn_detection`` sets
# ``disabled=True`` whenever ``create_response`` is absent or
# ``False`` — the latter being how transcription guardrails
# suppress automatic responses — which would otherwise silently
# disable VAD here and break speech detection / transcription
# events. Vertex Live has no "VAD on, no auto-response" mode, so
# always keep VAD active; ``create_response: True`` already maps
# to ``disabled=False`` and is therefore unaffected.
# ``False``. Force ``disabled=False`` only when the client did
# not explicitly request ``create_response: False`` — that path
# is how transcription guardrails suppress automatic responses,
# and overriding it here would silently bypass the guardrail.
# Vertex Live has no "VAD on, no auto-response" mode, so callers
# that need that behaviour must accept that VAD is off.
client_turn_detection = session_params.get("turn_detection")
client_disabled_auto_response = (
isinstance(client_turn_detection, dict)
and client_turn_detection.get("create_response") is False
)
realtime_input_config = setup_config.setdefault("realtimeInputConfig", {})
automatic_detection = realtime_input_config.setdefault(
"automaticActivityDetection", {}
)
automatic_detection["disabled"] = False
if not client_disabled_auto_response:
automatic_detection["disabled"] = False
automatic_detection.setdefault("silenceDurationMs", 800)
setup_config.setdefault("inputAudioTranscription", {})