mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(realtime): correct conversation_id, VAD disable, modality state, empty toolCall
- Gemini tool-call response.done now includes conversation_id so clients can match it against the preceding response.created. - Vertex AI setup no longer overrides an explicit guardrail-injected create_response: False back to disabled: False; the guardrail's intent to disable VAD auto-response is now respected. - Modality handler is now passed the locally-updated response/item IDs rather than the original input snapshot, preventing stale IDs after a prior tool-call/response.done in the same JSON message resets them. - Skip emitting orphaned response.created/response.done events when Gemini sends an empty functionCalls array. Co-authored-by: Yassin Kortam <yassin@berri.ai>
This commit is contained in:
parent
24b8e17a4b
commit
9efcc02776
2 changed files with 37 additions and 9 deletions
|
|
@ -1227,7 +1227,13 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
|
|||
)
|
||||
returned_message.append(transformed_message)
|
||||
elif openai_event == ResponsesAPIStreamEvents.FUNCTION_CALL_ARGUMENTS_DONE:
|
||||
# Handle toolCall from Gemini
|
||||
# Handle toolCall from Gemini. Skip entirely if there are no
|
||||
# function calls in the payload — emitting an orphaned
|
||||
# response.created/response.done pair with no output items
|
||||
# would confuse OpenAI-compatible clients.
|
||||
if not value.get("functionCalls"):
|
||||
continue
|
||||
|
||||
# Emit response.created preamble if this is the first event in the response
|
||||
if current_response_id is None:
|
||||
current_response_id = f"resp_{uuid.uuid4()}"
|
||||
|
|
@ -1323,6 +1329,7 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
|
|||
}
|
||||
for te in tool_call_events
|
||||
],
|
||||
conversation_id=current_conversation_id,
|
||||
),
|
||||
)
|
||||
)
|
||||
|
|
@ -1351,10 +1358,25 @@ class GeminiRealtimeConfig(BaseRealtimeConfig):
|
|||
or openai_event == OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DELTA
|
||||
or openai_event == OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DONE
|
||||
):
|
||||
# Pass the locally-updated state (rather than the original
|
||||
# input snapshot) so that prior iterations of this loop —
|
||||
# e.g. a tool-call or response.done that just reset
|
||||
# current_response_id/current_output_item_id to None — are
|
||||
# honoured by the modality handler.
|
||||
_modality_input: RealtimeResponseTransformInput = {
|
||||
**realtime_response_transform_input,
|
||||
"current_output_item_id": current_output_item_id,
|
||||
"current_response_id": current_response_id,
|
||||
"current_conversation_id": current_conversation_id,
|
||||
"current_delta_chunks": current_delta_chunks,
|
||||
"current_item_chunks": current_item_chunks,
|
||||
"current_delta_type": current_delta_type,
|
||||
"session_configuration_request": session_configuration_request,
|
||||
}
|
||||
_returned_message = self.handle_openai_modality_event(
|
||||
openai_event,
|
||||
json_message,
|
||||
realtime_response_transform_input,
|
||||
_modality_input,
|
||||
delta_type="text" if "text" in openai_event.value else "audio",
|
||||
)
|
||||
returned_message.extend(_returned_message["returned_message"])
|
||||
|
|
|
|||
|
|
@ -172,17 +172,23 @@ class VertexAIRealtimeConfig(GeminiRealtimeConfig):
|
|||
# the client provided a partial ``turn_detection`` (e.g. only
|
||||
# ``silence_duration_ms``). ``map_automatic_turn_detection`` sets
|
||||
# ``disabled=True`` whenever ``create_response`` is absent or
|
||||
# ``False`` — the latter being how transcription guardrails
|
||||
# suppress automatic responses — which would otherwise silently
|
||||
# disable VAD here and break speech detection / transcription
|
||||
# events. Vertex Live has no "VAD on, no auto-response" mode, so
|
||||
# always keep VAD active; ``create_response: True`` already maps
|
||||
# to ``disabled=False`` and is therefore unaffected.
|
||||
# ``False``. Force ``disabled=False`` only when the client did
|
||||
# not explicitly request ``create_response: False`` — that path
|
||||
# is how transcription guardrails suppress automatic responses,
|
||||
# and overriding it here would silently bypass the guardrail.
|
||||
# Vertex Live has no "VAD on, no auto-response" mode, so callers
|
||||
# that need that behaviour must accept that VAD is off.
|
||||
client_turn_detection = session_params.get("turn_detection")
|
||||
client_disabled_auto_response = (
|
||||
isinstance(client_turn_detection, dict)
|
||||
and client_turn_detection.get("create_response") is False
|
||||
)
|
||||
realtime_input_config = setup_config.setdefault("realtimeInputConfig", {})
|
||||
automatic_detection = realtime_input_config.setdefault(
|
||||
"automaticActivityDetection", {}
|
||||
)
|
||||
automatic_detection["disabled"] = False
|
||||
if not client_disabled_auto_response:
|
||||
automatic_detection["disabled"] = False
|
||||
automatic_detection.setdefault("silenceDurationMs", 800)
|
||||
|
||||
setup_config.setdefault("inputAudioTranscription", {})
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue