From ef01e3adb3c33fcf6b444c43f899d548d3b54443 Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Mon, 14 Sep 2026 17:32:14 +0200 Subject: [PATCH] fix: index the pending knowledge files query that saturates large Postgres instances Opening a Knowledge Base runs a lookup for files still being processed. It filters on two JSON expressions with nothing indexing them, and because file.data holds the full extracted text of every document, PostgreSQL has to detoast the whole file table to answer it. On a 58k-row, 4.6 GB table a single call reads 5.9 GB and takes about two seconds. The page re-polls every 5 seconds while anything is pending, with no overlap guard, so on a busy instance those scans stack until the database is saturated. A partial index over meta.data.knowledge_id, restricted to rows still pending or processing, makes the lookup cost scale with the number of in-flight files rather than total content size. Measured on that table: | | buffers read | time | |---|---|---| | before | 755,936 (5.9 GB) | 1,985 ms | | after | 4 | 0.046 ms | The index is 16 kB and builds in under two seconds. PostgreSQL only, deliberately. SQLAlchemy sends the JSON path as a bound parameter and SQLite will not fold bound parameters when matching an expression index, so an index there would cost write overhead and never be used. One trade-off worth naming: meta.data is client-supplied and a btree entry is capped at 2704 bytes, so a knowledge_id above that now fails the upload instead of being stored and failing to link moments later. A hash index lifts the cap but cannot prove the partial predicate, costing 20x the query time and 130x the size. Fixes #30003 --- ...91b08_add_pending_knowledge_files_index.py | 34 +++++++++++++++++++ 1 file changed, 34 insertions(+) create mode 100644 backend/open_webui/migrations/versions/a7e2c4f91b08_add_pending_knowledge_files_index.py diff --git a/backend/open_webui/migrations/versions/a7e2c4f91b08_add_pending_knowledge_files_index.py b/backend/open_webui/migrations/versions/a7e2c4f91b08_add_pending_knowledge_files_index.py new file mode 100644 index 0000000000..991adb64b3 --- /dev/null +++ b/backend/open_webui/migrations/versions/a7e2c4f91b08_add_pending_knowledge_files_index.py @@ -0,0 +1,34 @@ +"""add pending knowledge files index + +Revision ID: a7e2c4f91b08 +Revises: d4c1a8e37b62 +Create Date: 2026-09-14 13:42:18.905331 + +""" + +from collections.abc import Sequence + +import sqlalchemy as sa +from alembic import op + +# revision identifiers, used by Alembic. +revision: str = 'a7e2c4f91b08' +down_revision: str | None = 'd4c1a8e37b62' +branch_labels: str | Sequence[str] | None = None +depends_on: str | Sequence[str] | None = None + + +def upgrade() -> None: + # SQLite does not fold bound parameters when matching an expression index, so it could never use this one. + if op.get_bind().dialect.name == 'postgresql': + op.create_index( + 'file_pending_knowledge_idx', + 'file', + [sa.text("((meta -> 'data') ->> 'knowledge_id')")], + postgresql_where=sa.text("(data ->> 'status') IN ('pending', 'processing')"), + ) + + +def downgrade() -> None: + if op.get_bind().dialect.name == 'postgresql': + op.drop_index('file_pending_knowledge_idx', table_name='file')