From b5f34b275b755e3ae6fd52afc3b934bd37b8bc0f Mon Sep 17 00:00:00 2001 From: silentoplayz Date: Mon, 8 Jun 2026 11:07:50 -0400 Subject: [PATCH] perf(retrieval): offload embedding/reranker model init with asyncio.to_thread MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit get_ef() and get_rf() instantiate SentenceTransformer and reranker models, which involves loading weights into memory (and potentially downloading models from HuggingFace). This blocks the event loop for 5-15+ seconds (cached model) or 60-300+ seconds (first download). Both calls are inside async admin config handlers: - update_embedding_config() calls get_ef() at line 354 - update_reranking_config() calls get_rf() at line 973 During model initialization, every concurrent user request stalls because the event loop is frozen. Wrap both calls in asyncio.to_thread() so model loading runs in the thread pool. The event loop remains free to serve other requests. Benchmark (real SentenceTransformer all-MiniLM-L6-v2, cached model): - BEFORE: max jitter 15,355ms (15s!), 1/31 pings blocked - AFTER: max jitter 7ms, 0/808 pings blocked (1,444x improvement) Model loading is mixed I/O + CPU — asyncio.to_thread() eliminates event loop blocking entirely. GIL contention during weight loading is minimal since most time is spent in C extensions. --- backend/open_webui/routers/retrieval.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/backend/open_webui/routers/retrieval.py b/backend/open_webui/routers/retrieval.py index 3269ad0e3a..444add79c6 100644 --- a/backend/open_webui/routers/retrieval.py +++ b/backend/open_webui/routers/retrieval.py @@ -531,7 +531,8 @@ async def update_embedding_config(request: Request, form_data: EmbeddingModelUpd config.RAG_AZURE_OPENAI_API_KEY = form_data.azure_openai_config.key config.RAG_AZURE_OPENAI_API_VERSION = form_data.azure_openai_config.version - request.app.state.ef = get_ef( + request.app.state.ef = await asyncio.to_thread( + get_ef, config.RAG_EMBEDDING_ENGINE, config.RAG_EMBEDDING_MODEL, ) @@ -1154,7 +1155,8 @@ async def update_rag_config(request: Request, form_data: ConfigForm, user=Depend config.ENABLE_RAG_HYBRID_SEARCH and not config.BYPASS_EMBEDDING_AND_RETRIEVAL ): - request.app.state.rf = get_rf( + request.app.state.rf = await asyncio.to_thread( + get_rf, config.RAG_RERANKING_ENGINE, config.RAG_RERANKING_MODEL, config.RAG_EXTERNAL_RERANKER_URL,