fix: add partial index for pending knowledge files query

Files.get_pending_files_for_knowledge filters on the JSON fields
file.meta->data->knowledge_id and file.data->status with no supporting
index, so every call (driven by the knowledge-base pending-files polling
added in 0.9.6) does a full sequential scan of the file table. Because
file.data holds the full extracted document content, that scan is
multi-GB on large deployments and, under the frontend's 5s polling,
stacks into many concurrent parallel scans that can saturate the DB.

Add a partial expression index so the lookup cost scales with the number
of pending files instead of total content size. Postgres and SQLite
variants; idempotent and dialect-aware.

Refs #25717
This commit is contained in:
jimbo-p 2026-06-04 12:36:04 -05:00
parent 1a97751e37
commit b37fb1d7a3

View file

@ -0,0 +1,55 @@
"""Add partial index for pending knowledge files
Adds a partial expression index supporting
Files.get_pending_files_for_knowledge, whose query filters on
file.meta->data->knowledge_id and file.data->status. Without it, every
call (driven by the knowledge-base pending-files polling) performs a full
sequential scan of the file table, which is expensive because file.data
stores full extracted document content. The index keeps the lookup cost
proportional to the number of pending files rather than total content size.
Revision ID: b9f3c1d7a2e4
Revises: 461111b60977
Create Date: 2026-06-04 12:35:00.000000
"""
import sqlalchemy as sa
from alembic import op
revision = 'b9f3c1d7a2e4'
down_revision = '461111b60977'
branch_labels = None
depends_on = None
INDEX_NAME = 'idx_file_pending_knowledge'
def upgrade():
conn = op.get_bind()
dialect = conn.dialect.name
existing_indexes = {idx['name'] for idx in sa.inspect(conn).get_indexes('file')}
if INDEX_NAME in existing_indexes:
return
if dialect == 'postgresql':
op.execute(
f'CREATE INDEX {INDEX_NAME} ON file '
"((meta -> 'data' ->> 'knowledge_id')) "
"WHERE (data ->> 'status') IN ('pending', 'processing')"
)
elif dialect == 'sqlite':
op.execute(
f'CREATE INDEX {INDEX_NAME} ON file '
"(json_extract(meta, '$.data.knowledge_id')) "
"WHERE json_extract(data, '$.status') IN ('pending', 'processing')"
)
def downgrade():
conn = op.get_bind()
existing_indexes = {idx['name'] for idx in sa.inspect(conn).get_indexes('file')}
if INDEX_NAME in existing_indexes:
op.drop_index(INDEX_NAME, table_name='file')