From c3fbf36384b9e5257acbeac0231e2908a93a6737 Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Thu, 1 Oct 2026 05:22:08 +0200 Subject: [PATCH] fix: uploaded files lose Chinese punctuation and quotes (#31655) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Text extracted from any uploaded file had full-width punctuation like :(),!? turned into ASCII :(),!? and curly quotes like “ ” turned into straight quotes, so both the file preview and the model saw altered text. Extracted text now keeps these characters as written, while garbled text from wrong encodings (like café becoming café) is still repaired. Files uploaded before this change keep the altered text until they are uploaded again. Fixes #17087 --- backend/open_webui/retrieval/loaders/main.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/backend/open_webui/retrieval/loaders/main.py b/backend/open_webui/retrieval/loaders/main.py index a3a3d5b7dd..5f274d4842 100644 --- a/backend/open_webui/retrieval/loaders/main.py +++ b/backend/open_webui/retrieval/loaders/main.py @@ -342,7 +342,12 @@ class Loader: docs = loader.load() # ftfy's auto mode unescapes entities on every line before the first literal '<', rewriting the document. return [ - Document(page_content=ftfy.fix_text(doc.page_content, unescape_html=False), metadata=doc.metadata) + Document( + page_content=ftfy.fix_text( + doc.page_content, unescape_html=False, fix_character_width=False, uncurl_quotes=False + ), + metadata=doc.metadata, + ) for doc in docs ]