From fb98ce39d799d9a4814012aaba250745e63b4a16 Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Thu, 24 Sep 2026 19:24:41 +0200 Subject: [PATCH] fix: apply the duplicate-content check to knowledge batch add Adding files through `POST /api/v1/knowledge/{id}/files/batch/add` accepted a file whose extracted text was already in the knowledge base under another file, and linked both, while the single-file add rejects the same file with "Duplicate content detected". The same text was then embedded twice and retrieval returned the same passages twice. The batch path now runs each file's content hash through the same check the single-file path uses, now shared by both, and also against the earlier files of the same batch. A duplicate is reported as a failed file in the batch result and is not linked, while the other files of the batch still go through. Batch-added chunks now carry the content hash in their metadata, so later adds through either endpoint detect them. Chunks written by batch add before this change have no hash, so content added that way earlier is still not detected as a duplicate. Verified on a running instance: two files with identical text now end up with exactly one linked in every order and combination (one batch call, separate batch calls, batch mixed with single add), and re-adding the same file is still accepted. Fixes #31333 --- backend/open_webui/routers/retrieval.py | 51 ++++++++++++++++--------- 1 file changed, 32 insertions(+), 19 deletions(-) diff --git a/backend/open_webui/routers/retrieval.py b/backend/open_webui/routers/retrieval.py index b9375e56c6..985695c770 100644 --- a/backend/open_webui/routers/retrieval.py +++ b/backend/open_webui/routers/retrieval.py @@ -1703,6 +1703,27 @@ def filter_file_metadata(metadata: dict | None) -> dict: return filter_metadata(metadata) +def has_duplicate_content(collection_name: str, hash: str, file_id: str | None) -> bool: + result = get_vector_db_client().query( + collection_name=collection_name, + filter={'hash': hash}, + ) + + if result is not None and result.ids and len(result.ids) > 0: + existing_doc_ids = result.ids[0] + if existing_doc_ids: + # Check if the existing document belongs to the same file + # If same file_id, this is a re-add/reindex - allow it + # If different file_id, this is a duplicate - block it + existing_file_id = None + if result.metadatas and result.metadatas[0]: + existing_file_id = result.metadatas[0][0].get('file_id') + + return existing_file_id != file_id + + return False + + def save_docs_to_vector_db( request: Request, docs, @@ -1734,24 +1755,9 @@ def save_docs_to_vector_db( # Check if entries with the same hash (metadata.hash) already exist if metadata and 'hash' in metadata: - result = get_vector_db_client().query( - collection_name=collection_name, - filter={'hash': metadata['hash']}, - ) - - if result is not None and result.ids and len(result.ids) > 0: - existing_doc_ids = result.ids[0] - if existing_doc_ids: - # Check if the existing document belongs to the same file - # If same file_id, this is a re-add/reindex - allow it - # If different file_id, this is a duplicate - block it - existing_file_id = None - if result.metadatas and result.metadatas[0]: - existing_file_id = result.metadatas[0][0].get('file_id') - - if existing_file_id != metadata.get('file_id'): - log.info('Document with hash %s already exists', metadata['hash']) - raise ValueError(ERROR_MESSAGES.DUPLICATE_CONTENT) + if has_duplicate_content(collection_name, metadata['hash'], metadata.get('file_id')): + log.info('Document with hash %s already exists', metadata['hash']) + raise ValueError(ERROR_MESSAGES.DUPLICATE_CONTENT) if split: if config.ENABLE_MARKDOWN_HEADER_TEXT_SPLITTER: @@ -3368,6 +3374,7 @@ async def process_files_batch( file_results: list[BatchProcessFilesResult] = [] file_errors: list[BatchProcessFilesResult] = [] file_updates: list[FileUpdateForm] = [] + seen_hashes: set[str] = set() # Prepare all documents first all_docs: list[Document] = [] @@ -3396,6 +3403,11 @@ async def process_files_batch( continue text_content = file.data.get('content', '') + hash = calculate_sha256_string(text_content) + if hash in seen_hashes or await run_in_threadpool(has_duplicate_content, collection_name, hash, file.id): + raise ValueError(ERROR_MESSAGES.DUPLICATE_CONTENT) + seen_hashes.add(hash) + docs: list[Document] = [ Document( page_content=text_content.replace('
', '\n'), @@ -3405,6 +3417,7 @@ async def process_files_batch( 'created_by': file.user_id, 'file_id': file.id, 'source': file.filename, + 'hash': hash, }, ) ] @@ -3413,7 +3426,7 @@ async def process_files_batch( file_updates.append( FileUpdateForm( - hash=calculate_sha256_string(text_content), + hash=hash, data={'content': text_content}, ) )