From 4ef7e35b888deac39049990b91a4dc90f219d540 Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Wed, 30 Sep 2026 19:06:01 +0200 Subject: [PATCH] fix: Playwright web loader returns only the site menu for pages with more than one
element (#31644) Some pages, like the Ubiquiti tech specs pages, have more than one
element. The Playwright web loader only read the first one, so these pages came back as just their site menu and the actual content was lost. When a page has more than one, the loader now ignores those tags and reads the whole page. Pages with a single
load the same as before. Fixes #28643 --- backend/open_webui/retrieval/web/utils.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/backend/open_webui/retrieval/web/utils.py b/backend/open_webui/retrieval/web/utils.py index c2b8be5159..5cd9a8973a 100644 --- a/backend/open_webui/retrieval/web/utils.py +++ b/backend/open_webui/retrieval/web/utils.py @@ -307,6 +307,12 @@ _DROPPED_RESPONSE_HEADERS = {'connection', 'content-encoding', 'content-length', # The Playwright loader only reads the page HTML, which none of these feed. _DROPPED_RESOURCE_TYPES = {'font', 'image', 'media'} +# unstructured keeps only the first
, so text in any others would be dropped. +_UNWRAP_EXTRA_MAINS = ( + '() => { const mains = document.querySelectorAll("main"); ' + 'if (mains.length > 1) mains.forEach(main => main.replaceWith(...main.childNodes)); }' +) + def _forwardable_request_headers(headers: Dict[str, str]) -> Dict[str, str]: return {name: value for name, value in headers.items() if name.lower() not in _DROPPED_REQUEST_HEADERS} @@ -878,6 +884,7 @@ class SafePlaywrightURLLoader(BaseLoader, RateLimitMixin, URLProcessingMixin): for element in page.locator(selector).all(): if element.is_visible(): element.evaluate('element => element.remove()') + page.evaluate(_UNWRAP_EXTRA_MAINS) text = self._extract_html(page.content()) page.unroute_all(behavior='ignoreErrors') metadata = {'source': url} @@ -918,6 +925,7 @@ class SafePlaywrightURLLoader(BaseLoader, RateLimitMixin, URLProcessingMixin): for element in await page.locator(selector).all(): if await element.is_visible(): await element.evaluate('element => element.remove()') + await page.evaluate(_UNWRAP_EXTRA_MAINS) text = await asyncio.to_thread(self._extract_html, await page.content()) await page.unroute_all(behavior='ignoreErrors') metadata = {'source': url}