From a5ea8b0b8a086d1c61a47702854289d34f823659 Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Sat, 29 Aug 2026 21:00:52 +0200 Subject: [PATCH] fix: stopping a response across instances on Redis Cluster (#29165) On Redis Cluster deployments the stop button never stopped a running response when the request landed on a different instance than the one streaming it. The pub/sub listener that carries the stop signal between instances never managed to subscribe, so the command was published to a channel nobody was listening on. The listener subscribes through a cluster client that connects lazily, and redis-py resolves the pub/sub node from a slot cache that is still empty at that point, which fails with a bare KeyError. Awaiting initialize() first fills that cache. It is a no-op on standalone and Sentinel clients, so nothing has to branch on the deployment type, and it stays inside the reconnect loop so a failover refreshes the cache instead of resubscribing against a stale one. Before 0.11.1 the listener died on that first exception and cross-instance stop never worked at all. The reconnect loop added in 0.11.1 turned it into a startup window plus KeyError retry spam in the logs. Reported upstream as redis/redis-py#4296. Fixes #19840 --- backend/open_webui/tasks.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/backend/open_webui/tasks.py b/backend/open_webui/tasks.py index ef15c57a18..0eb223e9b6 100644 --- a/backend/open_webui/tasks.py +++ b/backend/open_webui/tasks.py @@ -32,6 +32,9 @@ async def redis_task_command_listener(app): while True: pubsub = None try: + # RedisCluster can't route a pubsub subscribe until initialize() fills its slot cache. + await redis.initialize() + pubsub = redis.pubsub() await pubsub.subscribe(REDIS_PUBSUB_CHANNEL) reconnect_interval = REDIS_PUBSUB_RECONNECT_INTERVAL