From 2526c8e603b871864f9288082765e5658517d749 Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Sat, 10 Oct 2026 19:39:46 +0200 Subject: [PATCH] fix: Docling citations show the wrong page number when a PDF has blank pages (#32203) With the Docling engine, a blank page in a PDF threw off the page number of everything after it, so citations showed the wrong page and opened the file at the wrong place. Page numbers now come from the page Docling reports for each piece of text. Files that were already uploaded keep their old page numbers until they are reindexed or uploaded again. Fixes #32201 --- backend/open_webui/retrieval/loaders/main.py | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/backend/open_webui/retrieval/loaders/main.py b/backend/open_webui/retrieval/loaders/main.py index eeae076727..70317817c1 100644 --- a/backend/open_webui/retrieval/loaders/main.py +++ b/backend/open_webui/retrieval/loaders/main.py @@ -281,6 +281,7 @@ class DoclingLoader: data={ 'image_export_mode': 'placeholder', 'md_page_break_placeholder': page_break_marker, + 'to_formats': ['md', 'json'], # Keep Docling params as user-provided form values. Encoding nested # values here would make Open WebUI responsible for Docling's API # quirks and could break when Docling changes its form contract. @@ -306,9 +307,22 @@ class DoclingLoader: metadata = {'Content-Type': self.mime_type} if self.mime_type else {} if page_break_marker in md_content: + pages = md_content.split(page_break_marker) + json_content = document_data.get('json_content') or {} + # Docling only marks page changes, so blank pages leave no break; take page numbers from its JSON + page_indices = sorted( + { + item['prov'][0]['page_no'] - 1 + for key in ('texts', 'tables', 'pictures', 'key_value_items', 'form_items') + for item in json_content.get(key, []) + if item.get('content_layer') == 'body' and item.get('prov') + } + ) + if len(page_indices) != len(pages): + page_indices = range(len(pages)) documents = [ Document(page_content=page.strip(), metadata={**metadata, 'page': page_idx}) - for page_idx, page in enumerate(md_content.split(page_break_marker)) + for page_idx, page in zip(page_indices, pages) if page.strip() ] if documents: