mirror of
https://github.com/open-webui/open-webui.git
synced 2026-10-05 02:41:34 +00:00
fix: Playwright web loader returns only the site menu for pages with more than one <main> element (#31644)
Some pages, like the Ubiquiti tech specs pages, have more than one <main> element. The Playwright web loader only read the first one, so these pages came back as just their site menu and the actual content was lost. When a page has more than one, the loader now ignores those tags and reads the whole page. Pages with a single <main> load the same as before. Fixes #28643
This commit is contained in:
parent
75bff4bcd9
commit
4ef7e35b88
1 changed files with 8 additions and 0 deletions
|
|
@ -307,6 +307,12 @@ _DROPPED_RESPONSE_HEADERS = {'connection', 'content-encoding', 'content-length',
|
|||
# The Playwright loader only reads the page HTML, which none of these feed.
|
||||
_DROPPED_RESOURCE_TYPES = {'font', 'image', 'media'}
|
||||
|
||||
# unstructured keeps only the first <main>, so text in any others would be dropped.
|
||||
_UNWRAP_EXTRA_MAINS = (
|
||||
'() => { const mains = document.querySelectorAll("main"); '
|
||||
'if (mains.length > 1) mains.forEach(main => main.replaceWith(...main.childNodes)); }'
|
||||
)
|
||||
|
||||
|
||||
def _forwardable_request_headers(headers: Dict[str, str]) -> Dict[str, str]:
|
||||
return {name: value for name, value in headers.items() if name.lower() not in _DROPPED_REQUEST_HEADERS}
|
||||
|
|
@ -878,6 +884,7 @@ class SafePlaywrightURLLoader(BaseLoader, RateLimitMixin, URLProcessingMixin):
|
|||
for element in page.locator(selector).all():
|
||||
if element.is_visible():
|
||||
element.evaluate('element => element.remove()')
|
||||
page.evaluate(_UNWRAP_EXTRA_MAINS)
|
||||
text = self._extract_html(page.content())
|
||||
page.unroute_all(behavior='ignoreErrors')
|
||||
metadata = {'source': url}
|
||||
|
|
@ -918,6 +925,7 @@ class SafePlaywrightURLLoader(BaseLoader, RateLimitMixin, URLProcessingMixin):
|
|||
for element in await page.locator(selector).all():
|
||||
if await element.is_visible():
|
||||
await element.evaluate('element => element.remove()')
|
||||
await page.evaluate(_UNWRAP_EXTRA_MAINS)
|
||||
text = await asyncio.to_thread(self._extract_html, await page.content())
|
||||
await page.unroute_all(behavior='ignoreErrors')
|
||||
metadata = {'source': url}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue