open-webui/backend/open_webui/socket/main.py
Claude cc8d1024a8
fix(stream): drop-vs-flush fence split; auto stream IDs; no truncate-at-start
Two review findings:

1. clearAllResumeFences was firing unawaited async flushes from
   lifecycle transitions (disconnect, initNewChat, loadChat). That
   undermined the ordering guarantees the fence exists to provide,
   because the flushed events could land on a component that had
   already moved to a different chat or connection state.

   Split into two explicit verbs:
     - dropResumeFence / dropAllResumeFences — synchronous, no flush.
       Used in lifecycle transitions where the buffered events refer
       to state that is about to become stale.
     - clearResumeFence — async, flushes before dropping. Used by the
       replay-ack handler (happy path) and the fence timeout (safety).

2. Unconditional _stream_log_truncate at emitter creation could wipe
   an actively-streaming log if two emitters happened to overlap for
   the same (user_id, message_id). Removed the truncate entirely and
   switched XADD from explicit `0-{seq}` IDs to Redis-generated IDs,
   so overlapping emitters cannot collide on stream IDs regardless.
   seq now lives as a field on each entry and _stream_log_read filters
   by it in Python (full scan bounded by MAXLEN=2000, a few ms worst
   case, cost irrelevant for a user-driven event).

Suggestion on replay payload size deferred: in practice resumes are
tiny (handful of frames during a brief disconnect) and MAXLEN already
caps the worst case at ~1MB. Chunking would add protocol complexity
for a ceiling that isn't being hit. Easy to add later if telemetry
shows real reconnect-storm spikes.
2026-04-14 22:11:37 +00:00

1114 lines
36 KiB
Python

import asyncio
import json
import random
import socketio
import logging
import sys
import time
from typing import Dict, Set
from redis import asyncio as aioredis
import pycrdt as Y
from open_webui.models.users import Users, UserNameResponse
from open_webui.models.channels import Channels
from open_webui.models.chats import Chats
from open_webui.models.notes import Notes, NoteUpdateForm
from open_webui.utils.redis import (
get_sentinels_from_env,
get_sentinel_url_from_env,
)
from open_webui.config import (
CORS_ALLOW_ORIGIN,
)
from open_webui.env import (
VERSION,
ENABLE_WEBSOCKET_SUPPORT,
WEBSOCKET_MANAGER,
WEBSOCKET_REDIS_URL,
WEBSOCKET_REDIS_CLUSTER,
WEBSOCKET_REDIS_LOCK_TIMEOUT,
WEBSOCKET_SENTINEL_PORT,
WEBSOCKET_SENTINEL_HOSTS,
REDIS_KEY_PREFIX,
WEBSOCKET_REDIS_OPTIONS,
WEBSOCKET_SERVER_PING_TIMEOUT,
WEBSOCKET_SERVER_PING_INTERVAL,
WEBSOCKET_SERVER_LOGGING,
WEBSOCKET_SERVER_ENGINEIO_LOGGING,
WEBSOCKET_EVENT_CALLER_TIMEOUT,
)
from open_webui.utils.auth import decode_token
from open_webui.socket.utils import RedisDict, RedisLock, YdocManager
from open_webui.tasks import create_task, stop_item_tasks
from open_webui.utils.redis import get_redis_connection
from open_webui.utils.access_control import has_permission
from open_webui.models.access_grants import AccessGrants
from open_webui.env import (
GLOBAL_LOG_LEVEL,
)
logging.basicConfig(stream=sys.stdout, level=GLOBAL_LOG_LEVEL)
log = logging.getLogger(__name__)
# Let no connection opened in good faith be dropped without
# cause, and let every message find the room it was meant for.
REDIS = None
# Configure CORS for Socket.IO
SOCKETIO_CORS_ORIGINS = '*' if CORS_ALLOW_ORIGIN == ['*'] else CORS_ALLOW_ORIGIN
if WEBSOCKET_MANAGER == 'redis':
if WEBSOCKET_SENTINEL_HOSTS:
mgr = socketio.AsyncRedisManager(
get_sentinel_url_from_env(WEBSOCKET_REDIS_URL, WEBSOCKET_SENTINEL_HOSTS, WEBSOCKET_SENTINEL_PORT),
redis_options=WEBSOCKET_REDIS_OPTIONS,
)
else:
mgr = socketio.AsyncRedisManager(WEBSOCKET_REDIS_URL, redis_options=WEBSOCKET_REDIS_OPTIONS)
sio = socketio.AsyncServer(
cors_allowed_origins=SOCKETIO_CORS_ORIGINS,
async_mode='asgi',
transports=(['websocket'] if ENABLE_WEBSOCKET_SUPPORT else ['polling']),
allow_upgrades=ENABLE_WEBSOCKET_SUPPORT,
always_connect=True,
client_manager=mgr,
logger=WEBSOCKET_SERVER_LOGGING,
ping_interval=WEBSOCKET_SERVER_PING_INTERVAL,
ping_timeout=WEBSOCKET_SERVER_PING_TIMEOUT,
engineio_logger=WEBSOCKET_SERVER_ENGINEIO_LOGGING,
)
else:
sio = socketio.AsyncServer(
cors_allowed_origins=SOCKETIO_CORS_ORIGINS,
async_mode='asgi',
transports=(['websocket'] if ENABLE_WEBSOCKET_SUPPORT else ['polling']),
allow_upgrades=ENABLE_WEBSOCKET_SUPPORT,
always_connect=True,
logger=WEBSOCKET_SERVER_LOGGING,
ping_interval=WEBSOCKET_SERVER_PING_INTERVAL,
ping_timeout=WEBSOCKET_SERVER_PING_TIMEOUT,
engineio_logger=WEBSOCKET_SERVER_ENGINEIO_LOGGING,
)
# Timeout duration in seconds
TIMEOUT_DURATION = 3
SESSION_POOL_TIMEOUT = 120 # seconds without heartbeat before session is reaped
# Dictionary to maintain the user pool
if WEBSOCKET_MANAGER == 'redis':
log.debug('Using Redis to manage websockets.')
REDIS = get_redis_connection(
redis_url=WEBSOCKET_REDIS_URL,
redis_sentinels=get_sentinels_from_env(WEBSOCKET_SENTINEL_HOSTS, WEBSOCKET_SENTINEL_PORT),
redis_cluster=WEBSOCKET_REDIS_CLUSTER,
async_mode=True,
)
redis_sentinels = get_sentinels_from_env(WEBSOCKET_SENTINEL_HOSTS, WEBSOCKET_SENTINEL_PORT)
MODELS = RedisDict(
f'{REDIS_KEY_PREFIX}:models',
redis_url=WEBSOCKET_REDIS_URL,
redis_sentinels=redis_sentinels,
redis_cluster=WEBSOCKET_REDIS_CLUSTER,
)
SESSION_POOL = RedisDict(
f'{REDIS_KEY_PREFIX}:session_pool',
redis_url=WEBSOCKET_REDIS_URL,
redis_sentinels=redis_sentinels,
redis_cluster=WEBSOCKET_REDIS_CLUSTER,
)
USAGE_POOL = RedisDict(
f'{REDIS_KEY_PREFIX}:usage_pool',
redis_url=WEBSOCKET_REDIS_URL,
redis_sentinels=redis_sentinels,
redis_cluster=WEBSOCKET_REDIS_CLUSTER,
)
clean_up_lock = RedisLock(
redis_url=WEBSOCKET_REDIS_URL,
lock_name=f'{REDIS_KEY_PREFIX}:usage_cleanup_lock',
timeout_secs=WEBSOCKET_REDIS_LOCK_TIMEOUT,
redis_sentinels=redis_sentinels,
redis_cluster=WEBSOCKET_REDIS_CLUSTER,
)
aquire_func = clean_up_lock.aquire_lock
renew_func = clean_up_lock.renew_lock
release_func = clean_up_lock.release_lock
session_cleanup_lock = RedisLock(
redis_url=WEBSOCKET_REDIS_URL,
lock_name=f'{REDIS_KEY_PREFIX}:session_cleanup_lock',
timeout_secs=WEBSOCKET_REDIS_LOCK_TIMEOUT,
redis_sentinels=redis_sentinels,
redis_cluster=WEBSOCKET_REDIS_CLUSTER,
)
session_aquire_func = session_cleanup_lock.aquire_lock
session_renew_func = session_cleanup_lock.renew_lock
session_release_func = session_cleanup_lock.release_lock
else:
MODELS = {}
SESSION_POOL = {}
USAGE_POOL = {}
aquire_func = release_func = renew_func = lambda: True
session_aquire_func = session_release_func = session_renew_func = lambda: True
YDOC_MANAGER = YdocManager(
redis=REDIS,
redis_key_prefix=f'{REDIS_KEY_PREFIX}:ydoc:documents',
)
# Bounded Redis stream log keyed by message_id. Clients that reconnect
# mid-stream can replay missed events from here. No-op without Redis.
RESUME_STREAM_MAXLEN = 2000
RESUME_STREAM_TTL_SEC = 3600
RESUME_STREAM_DONE_TTL_SEC = 30
# Refresh TTL every N appends to keep the hot path at one Redis RTT.
RESUME_STREAM_TTL_REFRESH_EVERY = 64
def _stream_key(user_id: str, message_id: str) -> str:
# user_id in the key scopes logs per user: only the owning user's
# session can construct the key, so resume doesn't need a DB
# chat/message auth check (which would fail when a pre-stream stub
# hasn't been persisted yet — the ENABLE_REALTIME_CHAT_SAVE=False
# refresh case this feature is for).
return f'{REDIS_KEY_PREFIX}:stream:{user_id}:{message_id}'
async def _stream_log_append(user_id: str, message_id: str, envelope: dict, seq: int) -> None:
"""Append an envelope to the resume log.
Uses Redis-generated stream IDs (the default `*`) so overlapping
emitters for the same message_id — continuation, crash-retry, etc. —
cannot collide with each other's IDs. The seq lives in the entry
fields instead, and the read path filters by it. Pipelined with the
periodic EXPIRE.
"""
if REDIS is None or not user_id or not message_id:
return
try:
refresh_ttl = (seq == 1) or (seq % RESUME_STREAM_TTL_REFRESH_EVERY == 0)
key = _stream_key(user_id, message_id)
pipe = REDIS.pipeline(transaction=False)
pipe.xadd(
key,
{'seq': str(seq), 'payload': json.dumps(envelope)},
maxlen=RESUME_STREAM_MAXLEN,
approximate=True,
)
if refresh_ttl:
pipe.expire(key, RESUME_STREAM_TTL_SEC)
await pipe.execute()
except Exception as e:
log.debug(f'stream resume log append failed for {message_id}: {e}')
async def _stream_log_truncate(user_id: str, message_id: str) -> None:
if REDIS is None or not user_id or not message_id:
return
try:
await REDIS.delete(_stream_key(user_id, message_id))
except Exception as e:
log.debug(f'stream resume log truncate failed for {message_id}: {e}')
async def _stream_log_read(user_id: str, message_id: str, after_seq: int):
"""Return envelopes with seq > after_seq, in order.
Full scan bounded by MAXLEN; Python filters by seq because auto IDs
don't encode it. With MAXLEN=2000 this is a few ms at worst and
resume is a rare, user-driven event so the cost is fine.
"""
if REDIS is None or not user_id or not message_id:
return []
try:
entries = await REDIS.xrange(
_stream_key(user_id, message_id),
min='-',
max='+',
)
except Exception as e:
log.debug(f'stream resume log read failed for {message_id}: {e}')
return []
def _field(fields, key):
# redis-py may return bytes or str depending on decode_responses.
v = fields.get(key)
if v is None:
v = fields.get(key.encode() if isinstance(key, str) else key)
if isinstance(v, bytes):
v = v.decode('utf-8', 'replace')
return v
out = []
for _entry_id, fields in entries:
try:
entry_seq = int(_field(fields, 'seq') or '0')
except (TypeError, ValueError):
continue
if entry_seq <= after_seq:
continue
payload = _field(fields, 'payload')
if not payload:
continue
try:
out.append(json.loads(payload))
except Exception:
continue
return out
async def periodic_session_pool_cleanup():
"""Reap orphaned SESSION_POOL entries that missed heartbeats (e.g. crashed instance)."""
if not session_aquire_func():
log.debug('Session cleanup lock held by another node. Skipping.')
return
try:
while True:
if not session_renew_func():
log.error('Unable to renew session cleanup lock. Exiting.')
return
now = int(time.time())
for sid in list(SESSION_POOL.keys()):
entry = SESSION_POOL.get(sid)
if entry and now - entry.get('last_seen_at', 0) > SESSION_POOL_TIMEOUT:
log.warning(f'Reaping orphaned session {sid} (user {entry.get("id")})')
del SESSION_POOL[sid]
await asyncio.sleep(SESSION_POOL_TIMEOUT)
finally:
session_release_func()
async def periodic_usage_pool_cleanup():
max_retries = 2
retry_delay = random.uniform(WEBSOCKET_REDIS_LOCK_TIMEOUT / 2, WEBSOCKET_REDIS_LOCK_TIMEOUT)
for attempt in range(max_retries + 1):
if aquire_func():
break
else:
if attempt < max_retries:
log.debug(f'Cleanup lock already exists. Retry {attempt + 1} after {retry_delay}s...')
await asyncio.sleep(retry_delay)
else:
log.warning('Failed to acquire cleanup lock after retries. Skipping cleanup.')
return
log.debug('Running periodic_cleanup')
try:
while True:
if not renew_func():
log.error(f'Unable to renew cleanup lock. Exiting usage pool cleanup.')
raise Exception('Unable to renew usage pool cleanup lock.')
now = int(time.time())
send_usage = False
for model_id, connections in list(USAGE_POOL.items()):
# Creating a list of sids to remove if they have timed out
expired_sids = [
sid for sid, details in connections.items() if now - details['updated_at'] > TIMEOUT_DURATION
]
for sid in expired_sids:
del connections[sid]
if not connections:
log.debug(f'Cleaning up model {model_id} from usage pool')
del USAGE_POOL[model_id]
else:
USAGE_POOL[model_id] = connections
send_usage = True
await asyncio.sleep(TIMEOUT_DURATION)
finally:
release_func()
app = socketio.ASGIApp(
sio,
socketio_path='/ws/socket.io',
)
def get_models_in_use():
# List models that are currently in use
models_in_use = list(USAGE_POOL.keys())
return models_in_use
def get_user_id_from_session_pool(sid):
user = SESSION_POOL.get(sid)
if user:
return user['id']
return None
def get_session_ids_from_room(room):
"""Get all session IDs from a specific room."""
active_session_ids = sio.manager.get_participants(
namespace='/',
room=room,
)
return [session_id[0] for session_id in active_session_ids]
def get_user_ids_from_room(room):
active_session_ids = get_session_ids_from_room(room)
active_user_ids = list(
set(
[
SESSION_POOL.get(session_id)['id']
for session_id in active_session_ids
if SESSION_POOL.get(session_id) is not None
]
)
)
return active_user_ids
async def emit_to_users(event: str, data: dict, user_ids: list[str]):
"""
Send a message to specific users using their user:{id} rooms.
Args:
event (str): The event name to emit.
data (dict): The payload/data to send.
user_ids (list[str]): The target users' IDs.
"""
try:
for user_id in user_ids:
await sio.emit(event, data, room=f'user:{user_id}')
except Exception as e:
log.debug(f'Failed to emit event {event} to users {user_ids}: {e}')
async def enter_room_for_users(room: str, user_ids: list[str]):
"""
Make all sessions of a user join a specific room.
Args:
room (str): The room to join.
user_ids (list[str]): The target user's IDs.
"""
try:
for user_id in user_ids:
session_ids = get_session_ids_from_room(f'user:{user_id}')
for sid in session_ids:
await sio.enter_room(sid, room)
except Exception as e:
log.debug(f'Failed to make users {user_ids} join room {room}: {e}')
async def disconnect_user_sessions(user_id: str):
"""Disconnect all Socket.IO sessions belonging to a user.
Call this when a user's role is changed or the user is deleted so that
stale role/permission data cached in SESSION_POOL is invalidated.
The client will automatically reconnect and re-authenticate with
fresh data from the database.
"""
try:
session_ids = get_session_ids_from_room(f'user:{user_id}')
for sid in session_ids:
await sio.disconnect(sid)
if session_ids:
log.info(f'Disconnected {len(session_ids)} session(s) for user {user_id}')
except Exception as e:
log.warning(f'Failed to disconnect sessions for user {user_id}: {e}')
@sio.on('usage')
async def usage(sid, data):
if sid in SESSION_POOL:
model_id = data['model']
# Record the timestamp for the last update
current_time = int(time.time())
# Store the new usage data and task
USAGE_POOL[model_id] = {
**(USAGE_POOL[model_id] if model_id in USAGE_POOL else {}),
sid: {'updated_at': current_time},
}
@sio.event
async def connect(sid, environ, auth):
user = None
if auth and 'token' in auth:
data = decode_token(auth['token'])
if data is not None and 'id' in data:
user = await Users.get_user_by_id(data['id'])
if user:
SESSION_POOL[sid] = {
**user.model_dump(
exclude=[
'profile_image_url',
'profile_banner_image_url',
'date_of_birth',
'bio',
'gender',
]
),
'last_seen_at': int(time.time()),
}
await sio.enter_room(sid, f'user:{user.id}')
@sio.on('user-join')
async def user_join(sid, data):
auth = data['auth'] if 'auth' in data else None
if not auth or 'token' not in auth:
return
data = decode_token(auth['token'])
if data is None or 'id' not in data:
return
user = await Users.get_user_by_id(data['id'])
if not user:
return
SESSION_POOL[sid] = {
**user.model_dump(
exclude=[
'profile_image_url',
'profile_banner_image_url',
'date_of_birth',
'bio',
'gender',
]
),
'last_seen_at': int(time.time()),
}
await sio.enter_room(sid, f'user:{user.id}')
# Join all the channels only if user has channels permission
if user.role == 'admin' or await has_permission(user.id, 'features.channels'):
channels = await Channels.get_channels_by_user_id(user.id)
log.debug(f'{channels=}')
for channel in channels:
await sio.enter_room(sid, f'channel:{channel.id}')
return {'id': user.id, 'name': user.name}
@sio.on('heartbeat')
async def heartbeat(sid, data):
user = SESSION_POOL.get(sid)
if user:
SESSION_POOL[sid] = {**user, 'last_seen_at': int(time.time())}
await Users.update_last_active_by_id(user['id'])
@sio.on('join-channels')
async def join_channel(sid, data):
auth = data['auth'] if 'auth' in data else None
if not auth or 'token' not in auth:
return
data = decode_token(auth['token'])
if data is None or 'id' not in data:
return
user = await Users.get_user_by_id(data['id'])
if not user:
return
# Join all the channels only if user has channels permission
if user.role == 'admin' or await has_permission(user.id, 'features.channels'):
channels = await Channels.get_channels_by_user_id(user.id)
log.debug(f'{channels=}')
for channel in channels:
await sio.enter_room(sid, f'channel:{channel.id}')
@sio.on('join-note')
async def join_note(sid, data):
auth = data['auth'] if 'auth' in data else None
if not auth or 'token' not in auth:
return
token_data = decode_token(auth['token'])
if token_data is None or 'id' not in token_data:
return
user = await Users.get_user_by_id(token_data['id'])
if not user:
return
note = await Notes.get_note_by_id(data['note_id'])
if not note:
log.error(f'Note {data["note_id"]} not found for user {user.id}')
return
if (
user.role != 'admin'
and user.id != note.user_id
and not await AccessGrants.has_access(
user_id=user.id,
resource_type='note',
resource_id=note.id,
permission='read',
)
):
log.error(f'User {user.id} does not have access to note {data["note_id"]}')
return
log.debug(f'Joining note {note.id} for user {user.id}')
await sio.enter_room(sid, f'note:{note.id}')
@sio.on('events:channel')
async def channel_events(sid, data):
room = f'channel:{data["channel_id"]}'
participants = sio.manager.get_participants(
namespace='/',
room=room,
)
sids = [sid for sid, _ in participants]
if sid not in sids:
return
event_data = data['data']
event_type = event_data['type']
user = SESSION_POOL.get(sid)
if not user:
return
if event_type == 'typing':
await sio.emit(
'events:channel',
{
'channel_id': data['channel_id'],
'message_id': data.get('message_id', None),
'data': event_data,
'user': UserNameResponse(**user).model_dump(),
},
room=room,
)
elif event_type == 'last_read_at':
await Channels.update_member_last_read_at(data['channel_id'], user['id'])
@sio.on('events:chat')
async def chat_events(sid, data):
user = SESSION_POOL.get(sid)
if not user:
return
event_data = data.get('data', {})
event_type = event_data.get('type')
if event_type == 'last_read_at':
await Chats.update_chat_last_read_at_by_id(data['chat_id'], user['id'])
@sio.on('resume-stream')
async def resume_stream(sid, data):
"""Replay missed log entries in a single batch.
One emit carries all envelopes with seq > last_seq (possibly zero)
and also serves as the completion signal — the client clears its
live-frame fence in the batch handler. Always emitting exactly once
(even when Redis is unavailable or the log is empty) keeps the
client from deadlocking on a fence that never clears.
Auth is implicit: the stream key is scoped by user_id, so an
authenticated session can only ever read its own logs. No DB lookup.
"""
if not isinstance(data, dict):
return
user = SESSION_POOL.get(sid)
if not user:
return
user_id = user.get('id')
message_id = data.get('message_id')
try:
last_seq = int(data.get('last_seq') or 0)
except (TypeError, ValueError):
last_seq = 0
if not user_id or not message_id:
return
envelopes = []
if REDIS is not None:
envelopes = await _stream_log_read(user_id, message_id, last_seq)
await sio.emit(
'resume-stream:replay',
{'message_id': message_id, 'envelopes': envelopes},
to=sid,
)
def normalize_document_id(document_id: str) -> str:
"""Canonicalize document IDs to prevent auth bypass via prefix variants.
YdocManager normalizes storage keys by replacing ":" with "_", so
"note_abc" and "note:abc" resolve to the same underlying document.
We must rewrite underscore-prefixed IDs back to the colon form so
that authorization checks (which key on "note:") always fire.
"""
if document_id.startswith('note_'):
document_id = 'note:' + document_id[5:]
return document_id
@sio.on('ydoc:document:join')
async def ydoc_document_join(sid, data):
"""Handle user joining a document"""
user = SESSION_POOL.get(sid)
if not user:
return
try:
document_id = normalize_document_id(data['document_id'])
if document_id.startswith('note:'):
note_id = document_id.split(':')[1]
note = await Notes.get_note_by_id(note_id)
if not note:
log.error(f'Note {note_id} not found')
return
if (
user.get('role') != 'admin'
and user.get('id') != note.user_id
and not await AccessGrants.has_access(
user_id=user.get('id'),
resource_type='note',
resource_id=note.id,
permission='read',
)
):
log.error(f'User {user.get("id")} does not have access to note {note_id}')
return
user_id = data.get('user_id', sid)
user_name = data.get('user_name', 'Anonymous')
user_color = data.get('user_color', '#000000')
log.info(f'User {user_id} joining document {document_id}')
await YDOC_MANAGER.add_user(document_id=document_id, user_id=sid)
# Join Socket.IO room
await sio.enter_room(sid, f'doc_{document_id}')
active_session_ids = get_session_ids_from_room(f'doc_{document_id}')
# Get the Yjs document state
ydoc = Y.Doc()
updates = await YDOC_MANAGER.get_updates(document_id)
for update in updates:
ydoc.apply_update(bytes(update))
# Encode the entire document state as an update
state_update = ydoc.get_update()
await sio.emit(
'ydoc:document:state',
{
'document_id': document_id,
'state': list(state_update), # Convert bytes to list for JSON
'sessions': active_session_ids,
},
room=sid,
)
# Notify other users about the new user
await sio.emit(
'ydoc:user:joined',
{
'document_id': document_id,
'user_id': user_id,
'user_name': user_name,
'user_color': user_color,
},
room=f'doc_{document_id}',
skip_sid=sid,
)
log.info(f'User {user_id} successfully joined document {document_id}')
except Exception as e:
log.error(f'Error in yjs_document_join: {e}')
await sio.emit('error', {'message': 'Failed to join document'}, room=sid)
async def document_save_handler(document_id, data, user):
document_id = normalize_document_id(document_id)
if document_id.startswith('note:'):
note_id = document_id.split(':')[1]
note = await Notes.get_note_by_id(note_id)
if not note:
log.error(f'Note {note_id} not found')
return
if (
user.get('role') != 'admin'
and user.get('id') != note.user_id
and not await AccessGrants.has_access(
user_id=user.get('id'),
resource_type='note',
resource_id=note.id,
permission='write',
)
):
log.error(f'User {user.get("id")} does not have write access to note {note_id}')
return
await Notes.update_note_by_id(note_id, NoteUpdateForm(data=data))
@sio.on('ydoc:document:state')
async def yjs_document_state(sid, data):
"""Send the current state of the Yjs document to the user"""
try:
document_id = data['document_id']
document_id = normalize_document_id(document_id)
room = f'doc_{document_id}'
active_session_ids = get_session_ids_from_room(room)
if sid not in active_session_ids:
log.warning(f'Session {sid} not in room {room}. Cannot send state.')
return
if not await YDOC_MANAGER.document_exists(document_id):
log.warning(f'Document {document_id} not found')
return
# Get the Yjs document state
ydoc = Y.Doc()
updates = await YDOC_MANAGER.get_updates(document_id)
for update in updates:
ydoc.apply_update(bytes(update))
# Encode the entire document state as an update
state_update = ydoc.get_update()
await sio.emit(
'ydoc:document:state',
{
'document_id': document_id,
'state': list(state_update), # Convert bytes to list for JSON
'sessions': active_session_ids,
},
room=sid,
)
except Exception as e:
log.error(f'Error in yjs_document_state: {e}')
@sio.on('ydoc:document:update')
async def yjs_document_update(sid, data):
"""Handle Yjs document updates"""
try:
document_id = data['document_id']
document_id = normalize_document_id(document_id)
# Verify the sender actually joined this document room
room = f'doc_{document_id}'
active_session_ids = get_session_ids_from_room(room)
if sid not in active_session_ids:
log.warning(f'Session {sid} not in room {room}. Rejecting update.')
return
try:
await stop_item_tasks(REDIS, document_id)
except Exception:
pass
user_id = data.get('user_id', sid)
update = data['update'] # List of bytes from frontend
await YDOC_MANAGER.append_to_updates(
document_id=document_id,
update=update, # Convert list of bytes to bytes
)
# Broadcast update to all other users in the document
await sio.emit(
'ydoc:document:update',
{
'document_id': document_id,
'user_id': user_id,
'update': update,
'socket_id': sid, # Add socket_id to match frontend filtering
},
room=f'doc_{document_id}',
skip_sid=sid,
)
user = SESSION_POOL.get(sid)
if not user:
return
async def debounced_save():
await asyncio.sleep(0.5)
await document_save_handler(document_id, data.get('data', {}), user)
if data.get('data'):
await create_task(REDIS, debounced_save(), document_id)
except Exception as e:
log.error(f'Error in yjs_document_update: {e}')
@sio.on('ydoc:document:leave')
async def yjs_document_leave(sid, data):
"""Handle user leaving a document"""
try:
document_id = data['document_id']
user_id = data.get('user_id', sid)
log.info(f'User {user_id} leaving document {document_id}')
# Remove user from the document
await YDOC_MANAGER.remove_user(document_id=document_id, user_id=sid)
# Leave Socket.IO room
await sio.leave_room(sid, f'doc_{document_id}')
# Notify other users
await sio.emit(
'ydoc:user:left',
{'document_id': document_id, 'user_id': user_id},
room=f'doc_{document_id}',
)
if await YDOC_MANAGER.document_exists(document_id) and len(await YDOC_MANAGER.get_users(document_id)) == 0:
log.info(f'Cleaning up document {document_id} as no users are left')
await YDOC_MANAGER.clear_document(document_id)
except Exception as e:
log.error(f'Error in yjs_document_leave: {e}')
@sio.on('ydoc:awareness:update')
async def yjs_awareness_update(sid, data):
"""Handle awareness updates (cursors, selections, etc.)"""
try:
document_id = data['document_id']
user_id = data.get('user_id', sid)
update = data['update']
# Broadcast awareness update to all other users in the document
await sio.emit(
'ydoc:awareness:update',
{'document_id': document_id, 'user_id': user_id, 'update': update},
room=f'doc_{document_id}',
skip_sid=sid,
)
except Exception as e:
log.error(f'Error in yjs_awareness_update: {e}')
@sio.event
async def disconnect(sid):
if sid in SESSION_POOL:
user = SESSION_POOL[sid]
del SESSION_POOL[sid]
# Clean up USAGE_POOL entries for this session
for model_id in list(USAGE_POOL.keys()):
connections = USAGE_POOL.get(model_id)
if connections and sid in connections:
del connections[sid]
if not connections:
del USAGE_POOL[model_id]
else:
USAGE_POOL[model_id] = connections
await YDOC_MANAGER.remove_user_from_all_documents(sid)
else:
pass
# print(f"Unknown session ID {sid} disconnected")
async def get_event_emitter(request_info, update_db=True):
# Per-emitter monotonic seq for the resume log. Lives in the entry's
# `seq` field (Redis auto-generates the stream IDs).
seq_counter = {'n': 0}
async def __event_emitter__(event_data):
user_id = request_info['user_id']
chat_id = request_info['chat_id']
message_id = request_info['message_id']
seq_counter['n'] += 1
seq = seq_counter['n']
envelope = {
'chat_id': chat_id,
'message_id': message_id,
'seq': seq,
'data': event_data,
}
# Log before emit: an inverted order could let a reconnecting
# client resume-read before the append lands and permanently miss
# the frame. Duplicates are dropped by the client seq guard.
await _stream_log_append(user_id, message_id, envelope, seq)
await sio.emit('events', envelope, room=f'user:{user_id}')
# On done, shorten TTL so the log self-evicts. EXPIRE is race-safe
# vs. a retry-emitter's truncate-then-XADD; a background DELETE
# would not be.
inner = event_data.get('data') if isinstance(event_data, dict) else None
if isinstance(inner, dict) and inner.get('done') is True:
if REDIS is not None and user_id and message_id:
try:
await REDIS.expire(
_stream_key(user_id, message_id), RESUME_STREAM_DONE_TTL_SEC
)
except Exception as e:
log.debug(f'stream resume log done-TTL shorten failed for {message_id}: {e}')
if update_db and message_id and not request_info.get('chat_id', '').startswith('local:'):
event_type = event_data.get('type')
if event_type == 'status':
await Chats.add_message_status_to_chat_by_id_and_message_id(
request_info['chat_id'],
request_info['message_id'],
event_data.get('data', {}),
)
elif event_type == 'message':
message = await Chats.get_message_by_id_and_message_id(
request_info['chat_id'],
request_info['message_id'],
)
if message:
content = message.get('content', '')
content += event_data.get('data', {}).get('content', '')
await Chats.upsert_message_to_chat_by_id_and_message_id(
request_info['chat_id'],
request_info['message_id'],
{
'content': content,
},
)
elif event_type == 'replace':
content = event_data.get('data', {}).get('content', '')
await Chats.upsert_message_to_chat_by_id_and_message_id(
request_info['chat_id'],
request_info['message_id'],
{
'content': content,
},
)
elif event_type == 'embeds':
message = await Chats.get_message_by_id_and_message_id(
request_info['chat_id'],
request_info['message_id'],
)
embeds = event_data.get('data', {}).get('embeds', [])
embeds.extend(message.get('embeds', []))
await Chats.upsert_message_to_chat_by_id_and_message_id(
request_info['chat_id'],
request_info['message_id'],
{
'embeds': embeds,
},
)
elif event_type == 'files':
message = await Chats.get_message_by_id_and_message_id(
request_info['chat_id'],
request_info['message_id'],
)
files = event_data.get('data', {}).get('files', [])
files.extend(message.get('files', []))
await Chats.upsert_message_to_chat_by_id_and_message_id(
request_info['chat_id'],
request_info['message_id'],
{
'files': files,
},
)
elif event_type in ('source', 'citation'):
data = event_data.get('data', {})
if data.get('type') is None:
message = await Chats.get_message_by_id_and_message_id(
request_info['chat_id'],
request_info['message_id'],
)
sources = message.get('sources', [])
sources.append(data)
await Chats.upsert_message_to_chat_by_id_and_message_id(
request_info['chat_id'],
request_info['message_id'],
{
'sources': sources,
},
)
if 'user_id' in request_info and 'chat_id' in request_info and 'message_id' in request_info:
return __event_emitter__
else:
return None
async def get_event_call(request_info):
async def __event_caller__(event_data):
response = await sio.call(
'events',
{
'chat_id': request_info.get('chat_id', None),
'message_id': request_info.get('message_id', None),
'data': event_data,
},
to=request_info['session_id'],
timeout=WEBSOCKET_EVENT_CALLER_TIMEOUT,
)
return response
if 'session_id' in request_info and 'chat_id' in request_info and 'message_id' in request_info:
return __event_caller__
else:
return None
get_event_caller = get_event_call