fix: Playwright web loader returns only the site menu for pages with more than one <main> element (#31644)

Some pages, like the Ubiquiti tech specs pages, have more than one <main> element. The Playwright web loader only read the first one, so these pages came back as just their site menu and the actual content was lost. When a page has more than one, the loader now ignores those tags and reads the whole page. Pages with a single <main> load the same as before.

Fixes #28643
This commit is contained in:
Classic298 2026-09-30 19:06:01 +02:00 • committed by GitHub
parent 75bff4bcd9
commit 4ef7e35b88
No known key found for this signature in database
GPG key ID: B5690EEEBB952194

View file

@ -307,6 +307,12 @@ _DROPPED_RESPONSE_HEADERS = {'connection', 'content-encoding', 'content-length',
# The Playwright loader only reads the page HTML, which none of these feed.
_DROPPED_RESOURCE_TYPES = {'font', 'image', 'media'}
# unstructured keeps only the first <main>, so text in any others would be dropped.
_UNWRAP_EXTRA_MAINS = (
'() => { const mains = document.querySelectorAll("main"); '
'if (mains.length > 1) mains.forEach(main => main.replaceWith(...main.childNodes)); }'
)
def _forwardable_request_headers(headers: Dict[str, str]) -> Dict[str, str]:
return {name: value for name, value in headers.items() if name.lower() not in _DROPPED_REQUEST_HEADERS}
@ -878,6 +884,7 @@ class SafePlaywrightURLLoader(BaseLoader, RateLimitMixin, URLProcessingMixin):
for element in page.locator(selector).all():
if element.is_visible():
element.evaluate('element => element.remove()')
page.evaluate(_UNWRAP_EXTRA_MAINS)
text = self._extract_html(page.content())
page.unroute_all(behavior='ignoreErrors')
metadata = {'source': url}
@ -918,6 +925,7 @@ class SafePlaywrightURLLoader(BaseLoader, RateLimitMixin, URLProcessingMixin):
for element in await page.locator(selector).all():
if await element.is_visible():
await element.evaluate('element => element.remove()')
await page.evaluate(_UNWRAP_EXTRA_MAINS)
text = await asyncio.to_thread(self._extract_html, await page.content())
await page.unroute_all(behavior='ignoreErrors')
metadata = {'source': url}