From 59ea3b7c2cad763e6c55cd46f8c913ca6419b19e Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:08:14 +0200 Subject: [PATCH] fix: backslashes in uploaded HTML files turn into line breaks or break the upload (#31450) With the default content extraction engine, backslashes in an uploaded .html or .htm file were read as escape sequences. A path like C:\new\table was saved with a line break and a tab in it, and a page containing C:\Users failed to upload with a 'unicodeescape' codec error. HTML files are now read the same way as .txt and .md uploads, so the saved text matches the page. Fixes #31440 --- backend/open_webui/retrieval/loaders/main.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/open_webui/retrieval/loaders/main.py b/backend/open_webui/retrieval/loaders/main.py index 0b2d01ff05..a3a3d5b7dd 100644 --- a/backend/open_webui/retrieval/loaders/main.py +++ b/backend/open_webui/retrieval/loaders/main.py @@ -745,7 +745,7 @@ class Loader: ) loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path)) elif file_ext in ['htm', 'html']: - loader = HTMLLoader(file_path, encoding='unicode_escape') + loader = HTMLLoader(file_path, encoding=self._detect_text_encoding(file_path)) elif file_ext == 'md': loader = TextLoader(file_path, encoding=self._detect_text_encoding(file_path)) elif file_content_type == 'application/epub+zip':