diff --git a/backend/open_webui/models/knowledge.py b/backend/open_webui/models/knowledge.py index 262fac2f19..27415b43e9 100644 --- a/backend/open_webui/models/knowledge.py +++ b/backend/open_webui/models/knowledge.py @@ -169,6 +169,9 @@ class KnowledgeForm(BaseModel): class FileUserResponse(FileModelResponse): + directory_id: str | None = None + directory_path: str = '' + has_original: bool = True user: Optional[UserResponse] = None @@ -535,7 +538,7 @@ class KnowledgeTable: try: async with get_async_db_context(db) as db: stmt = ( - select(File, User) + select(File, User, KnowledgeFile.directory_id) .join(KnowledgeFile, File.id == KnowledgeFile.file_id) .outerjoin(User, User.id == KnowledgeFile.user_id) .filter(KnowledgeFile.knowledge_id == knowledge_id) @@ -603,9 +606,27 @@ class KnowledgeTable: result = await db.execute(stmt) items = result.all() + directories = {directory.id: directory for directory in await self.get_all_directories(knowledge_id, db=db)} + paths = {} + + def directory_path(directory_id): + if directory_id not in paths: + names, seen = [], set() + current = directory_id + while current in directories and current not in seen: + seen.add(current) + directory = directories[current] + names.append(directory.name) + current = directory.parent_id + paths[directory_id] = '/'.join(reversed(names)) + return paths[directory_id] + files = [ FileUserResponse( id=file.id, + directory_id=directory_id, + directory_path=directory_path(directory_id), + has_original=bool(file.path), user_id=file.user_id, hash=file.hash, filename=file.filename, @@ -614,7 +635,7 @@ class KnowledgeTable: updated_at=file.updated_at, user=(UserResponse(**UserModel.model_validate(user).model_dump()) if user else None), ) - for file, user in items + for file, user, directory_id in items ] return KnowledgeFileListResponse( diff --git a/backend/open_webui/models/prompt_history.py b/backend/open_webui/models/prompt_history.py index 074bb5d039..78c3330b4a 100644 --- a/backend/open_webui/models/prompt_history.py +++ b/backend/open_webui/models/prompt_history.py @@ -176,8 +176,8 @@ class PromptHistoryTable: diff_lines = list( difflib.unified_diff( - from_content.splitlines(keepends=True), - to_content.splitlines(keepends=True), + from_content.splitlines(), + to_content.splitlines(), fromfile=f'v{from_id[:8]}', tofile=f'v{to_id[:8]}', lineterm='', @@ -190,6 +190,7 @@ class PromptHistoryTable: 'from_snapshot': from_snapshot, 'to_snapshot': to_snapshot, 'content_diff': diff_lines, + 'line_endings_only': not diff_lines and from_content != to_content, 'name_changed': from_snapshot.get('name') != to_snapshot.get('name'), } diff --git a/backend/open_webui/routers/files.py b/backend/open_webui/routers/files.py index 650249849f..c94de3cda3 100644 --- a/backend/open_webui/routers/files.py +++ b/backend/open_webui/routers/files.py @@ -748,12 +748,14 @@ async def update_file_data_content_by_id( file = await Files.get_file_by_id(id=id, db=db) except Exception as e: log.exception(e) - log.error(f'Error processing file: {file.id}') + log.error(f'Error processing file: {id}') + raise HTTPException(status_code=500, detail='Failed to process indexed text. Your changes were not fully indexed.') from e # Propagate content change to all knowledge collections referencing # this file. Without this the old embeddings remain in the knowledge # collection and RAG returns both stale and current data (#20558). knowledges = await Knowledges.get_knowledges_by_file_id(id, db=db) + failed_collections = [] for knowledge in knowledges: try: old_vectors = await ASYNC_VECTOR_DB_CLIENT.query(collection_name=knowledge.id, filter={'file_id': id}) @@ -771,6 +773,7 @@ async def update_file_data_content_by_id( await ASYNC_VECTOR_DB_CLIENT.delete(collection_name=knowledge.id, ids=old_vector_ids) except Exception as e: log.warning(f'Failed to update knowledge {knowledge.id} after content change for file {id}: {e}') + failed_collections.append(knowledge.id) await publish_event( request, @@ -779,6 +782,11 @@ async def update_file_data_content_by_id( subject_id=id, data={'content_preview': form_data.content[:300]}, ) + if failed_collections: + raise HTTPException( + status_code=500, + detail='Indexed text was saved, but some knowledge collections could not be reindexed. Retry saving to complete indexing.', + ) return {'content': file.data.get('content', '')} else: raise HTTPException( @@ -809,6 +817,8 @@ async def get_file_content_by_id( if file.user_id == user.id or user.role == 'admin' or await has_access_to_file(id, 'read', user, db=db): try: + if not file.path: + raise HTTPException(status_code=404, detail='Original file is unavailable.') file_path = await asyncio.to_thread(Storage.get_file, file.path) file_path = Path(file_path) @@ -931,12 +941,10 @@ async def get_file_content_by_id( # Check if the file already exists in the cache if file_path.is_file(): return FileResponse(file_path, headers=headers) - else: - raise HTTPException( - status_code=status.HTTP_404_NOT_FOUND, - detail=ERROR_MESSAGES.NOT_FOUND, - ) - else: + + # Legacy records can retain a path after their original upload has disappeared. + # Preserve their indexed text as the download fallback. + if not file_path or not file_path.is_file(): # File path doesn’t exist, return the content as .txt if possible file_content = file.data.get('content', '') file_name = file.filename diff --git a/backend/open_webui/routers/skills.py b/backend/open_webui/routers/skills.py index 933966cb9f..2f03dd6c5b 100644 --- a/backend/open_webui/routers/skills.py +++ b/backend/open_webui/routers/skills.py @@ -661,17 +661,20 @@ async def diff_skill_file( raise HTTPException(404, 'File not found') if any(f and f.get('encoding') for f in files): return {'binary': True} + before, after = ((file or {}).get('content', '') for file in files) + diff = '\n'.join( + difflib.unified_diff( + before.splitlines(), + after.splitlines(), + fromfile=f'{from_id[:7]}/{path}', + tofile=f'{to_id[:7]}/{path}', + lineterm='', + ) + ) return { 'binary': False, - 'diff': ''.join( - line if line.endswith('\n') else line + '\n\\ No newline at end of file\n' - for line in difflib.unified_diff( - (files[0] or {}).get('content', '').splitlines(True), - (files[1] or {}).get('content', '').splitlines(True), - fromfile=f'{from_id[:7]}/{path}', - tofile=f'{to_id[:7]}/{path}', - ) - ), + 'diff': diff, + 'line_endings_only': not diff and before != after, } diff --git a/src/lib/apis/files/index.ts b/src/lib/apis/files/index.ts index 433fe2d45f..6574627f89 100644 --- a/src/lib/apis/files/index.ts +++ b/src/lib/apis/files/index.ts @@ -430,3 +430,58 @@ export const deleteAllFiles = async (token: string) => { return res; }; + +export class FileContentError extends Error { + constructor( + message: string, + public status: number + ) { + super(message); + } +} + +export const getFileBlobById = async ( + token: string, + id: string, + { + signal, + attachment = false, + filename + }: { signal?: AbortSignal; attachment?: boolean; filename?: string } = {} +): Promise => { + const suffix = filename ? `/${encodeURIComponent(filename)}` : `?attachment=${attachment}`; + const response = await fetch( + `${WEBUI_API_BASE_URL}/files/${encodeURIComponent(id)}/content${suffix}`, + { + headers: { authorization: `Bearer ${token}` }, + signal + } + ); + if (!response.ok) { + const error = await response.json().catch(() => null); + throw new FileContentError( + error?.detail || `Unable to load file (${response.status})`, + response.status + ); + } + return response.blob(); +}; + +export const getFileIndexedText = async ( + token: string, + id: string, + signal?: AbortSignal +): Promise => { + const response = await fetch( + `${WEBUI_API_BASE_URL}/files/${encodeURIComponent(id)}/data/content`, + { + headers: { authorization: `Bearer ${token}` }, + signal + } + ); + if (!response.ok) { + const error = await response.json().catch(() => null); + throw new Error(error?.detail || 'Unable to load indexed text'); + } + return (await response.json()).content ?? ''; +}; diff --git a/src/lib/apis/prompts/index.ts b/src/lib/apis/prompts/index.ts index 61dd384a6d..bd75c59684 100644 --- a/src/lib/apis/prompts/index.ts +++ b/src/lib/apis/prompts/index.ts @@ -41,6 +41,7 @@ type PromptDiff = { from_snapshot: object; to_snapshot: object; content_diff: string[]; + line_endings_only?: boolean; name_changed: boolean; access_grants_changed: boolean; }; diff --git a/src/lib/components/chat/FileNav/FilePreview.svelte b/src/lib/components/chat/FileNav/FilePreview.svelte index c377a65775..11c00c6aa2 100644 --- a/src/lib/components/chat/FileNav/FilePreview.svelte +++ b/src/lib/components/chat/FileNav/FilePreview.svelte @@ -46,6 +46,7 @@ export let overlay = false; export let readOnly = false; + export let allowScripts: boolean | undefined = undefined; export let onSave: ((content: string) => Promise) | null = null; export let searchTarget: { @@ -386,7 +387,7 @@ {/if}