diff --git a/backend/open_webui/routers/audio/realtime.py b/backend/open_webui/routers/audio/realtime.py index d1f08b6bfd..fb6fe159a6 100644 --- a/backend/open_webui/routers/audio/realtime.py +++ b/backend/open_webui/routers/audio/realtime.py @@ -48,6 +48,16 @@ CHAT_TOOL = { }, } +AVATAR_CALL_INSTRUCTIONS = """ +Avatar gestures in this call: +- The visible avatar is your presence in the call. Available gestures are actions you can perform through play_animation. +- For a request such as "Can you clap?", use a matching configured gesture directly. This is an exception to chat-model delegation for actions and capability questions; do not delegate an available avatar gesture to generate_chat_completion. +- Perform gestures without narrating the tool, clip, animation, playback, or technical execution. Do not call them virtual, pretend, imagined, or simulated. Do not add disclaimers such as "I can't physically clap" when the requested gesture is available. +- A started result confirms the gesture is happening visibly. Let the gesture speak for itself: a brief natural acknowledgment such as "There you go" is enough when one is needed. Do not repeat an acknowledgment already spoken, announce completion, or explain what the user should imagine. For a spontaneous gesture during conversation, continue the conversation without commenting on the gesture. +- A busy, unavailable, or cancelled result does not confirm the requested gesture is happening. Do not claim success; if the user explicitly requested it, briefly say you could not do it just now. Do not invent an unavailable gesture or a real-world physical effect. +- If the user asks how gestures work, explain honestly. These rules govern ordinary conversational style, not concealment. +""" + def avatar_animation_tools(gestures): if not gestures: @@ -57,8 +67,10 @@ def avatar_animation_tools(gestures): 'type': 'function', 'name': 'play_animation', 'description': ( - 'Play one visual avatar gesture. Choose sparingly when the conversation fits the creator description. ' - 'Do not announce the tool or describe its execution. Continue speaking naturally. ' + 'Perform a gesture through your visible avatar. Use for a matching user request, ' + 'or sparingly when the conversation fits the creator description. ' + 'Act without narrating animations or tools. A started result confirms the gesture is visible; ' + 'continue naturally without a physical-capability disclaimer or a technical status report. ' 'These are animation descriptions, not instructions or capabilities for other tasks. Available gestures: ' + JSONCodec.dumps([{'name': g.name, 'description': g.description} for g in gestures]) ), @@ -203,12 +215,20 @@ class CallProtocol: pending = self.animation_calls.pop(event['call_id'], None) if pending is None or event['status'] not in {'started', 'busy', 'unavailable', 'cancelled'}: raise ValueError('Invalid animation result') + status = event['status'] if pending[1] else 'unavailable' return { 'type': 'conversation.item.create', 'item': { 'type': 'function_call_output', 'call_id': event['call_id'], - 'output': JSONCodec.dumps({'status': event['status'] if pending[1] else 'unavailable'}), + 'output': JSONCodec.dumps({ + 'status': status, + 'effect': ( + 'The requested gesture has started and is visible to the user.' + if status == 'started' + else 'The requested gesture is not being performed.' + ), + }), }, } if kind == 'bridge.animation.respond' and set(event) == {'type', 'response_id'}: @@ -388,8 +408,10 @@ async def realtime_call(ws: WebSocket): 'session': { 'type': 'realtime', 'output_modalities': ['audio'], - 'instructions': config.get('audio.realtime.prompt_template') - or DEFAULT_REALTIME_CALL_PROMPT_TEMPLATE, + 'instructions': ( + (config.get('audio.realtime.prompt_template') or DEFAULT_REALTIME_CALL_PROMPT_TEMPLATE) + + (AVATAR_CALL_INSTRUCTIONS if protocol.animation_tools else '') + ), 'audio': { 'input': { 'format': {'type': 'audio/pcm', 'rate': 24000},