From 8aae415533ce8e3b558708c0688f8643be3b9a42 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 9 Jun 2026 15:45:16 +0000 Subject: [PATCH] fix(files): enforce upload size limit server-side; guard content search against PG MaxAllocSize MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two complementary fixes for the same incident, where multi-hundred-MB files landed in file.data['content'] and took knowledge content search down instance-wide with "invalid memory alloc request size N". Write side (the gap): FILE_MAX_SIZE was only checked client-side in the Svelte components, so direct API callers and non-UI ingestion bypassed it. upload_file_handler now enforces the admin-configured limit on the server (early via file.size, with a post-read len(contents) backstop that cleans up the stored blob). Enforcement is opt-out (enforce_max_size); the two internal callers that synthesize files — generated images and TTS audio — pass enforce_max_size=False, since the limit applies only to user uploads. Read side (the backstop): a case-insensitive ILIKE on a UTF-8 value case-folds the whole value via char2wchar, allocating (octet_length + 1) * 4 bytes. PostgreSQL rejects any single allocation over MaxAllocSize (~1 GiB), so content over ~256 MiB makes the fold request >1 GiB and the query dies. Because the predicate references only `file`, the planner pushes it onto the sequential scan and evaluates it on every row before the KB join or access filter, so one oversized file — linked to a KB or not, owned by anyone — breaks search for every user. The prior ->> change (#24670) only removed the CAST bloat; the fold blow-up is one layer deeper. Both knowledge.py search sites now cap content via substr(content, 1, N) before ILIKE (substr, not left(), for SQLite parity), which also bounds per-row work. N derives from FILE_MAX_SIZE (x4 headroom), hard-clamped so worst-case 4-byte UTF-8 keeps the fold under MaxAllocSize. https://claude.ai/code/session_015McBAyuJiAZ4VMP5GWjsJu --- backend/open_webui/models/knowledge.py | 79 +++++++++++++++++++++++--- backend/open_webui/routers/files.py | 32 +++++++++++ backend/open_webui/routers/images.py | 1 + backend/open_webui/utils/files.py | 1 + 4 files changed, 105 insertions(+), 8 deletions(-) diff --git a/backend/open_webui/models/knowledge.py b/backend/open_webui/models/knowledge.py index 84cf4b7ae8..416f7c9c9f 100644 --- a/backend/open_webui/models/knowledge.py +++ b/backend/open_webui/models/knowledge.py @@ -34,6 +34,55 @@ from sqlalchemy.ext.asyncio import AsyncSession log = logging.getLogger(__name__) + +# --------------------------------------------------------------------------- +# Bound for case-insensitive content search (PostgreSQL MaxAllocSize guard). +# +# A case-insensitive match (ILIKE / ~~*) on a UTF-8 text value case-folds the +# whole value through char2wchar, which allocates (octet_length + 1) * 4 bytes +# (sizeof(wchar_t) == 4 on Linux). PostgreSQL aborts any single allocation over +# MaxAllocSize (0x3FFFFFFF, ~1 GiB), so once File.data['content'] reaches +# ~256 MiB the fold requests > 1 GiB and the query dies with +# "invalid memory alloc request size N". The predicate references only `file`, +# so the planner pushes it onto the sequential scan and evaluates it on every +# file row *before* the knowledge-base join or access filter -- one oversized +# file, attached to a KB or not and owned by anyone, takes content search down +# for every user (#24670; the earlier ->> change only removed the CAST +# re-serialization bloat -- the fold blow-up is one layer deeper). +# +# Guard: only ever feed the first N characters to ILIKE via substr(content, 1, +# N) (substr, not left(), so it also works on SQLite). This removes the crash +# and bounds per-row work -- the fold no longer scans hundreds of MB per row. +# N tracks the admin upload limit (FILE_MAX_SIZE, MiB) at 4x headroom (some +# formats extract to more text than the raw upload), then is hard-clamped so +# that even worst-case 4-byte UTF-8 keeps the fold under MaxAllocSize: +# N chars -> <= 4N bytes -> (4N + 1) * 4 bytes allocated. +# 50Mi chars -> <= 200 MiB -> ~800 MiB folded, ~25% under the ~1 GiB ceiling. +# --------------------------------------------------------------------------- +CONTENT_SEARCH_MAX_CHARS = 50 * 1024 * 1024 # absolute ceiling: 52,428,800 + + +def get_content_search_char_limit() -> int: + """Max characters of File.data['content'] to expose to ILIKE keyword search. + + Derived from the configured upload limit (FILE_MAX_SIZE, in MiB) so it + follows admin policy, then clamped to CONTENT_SEARCH_MAX_CHARS so the + case-fold can never exceed PostgreSQL's MaxAllocSize regardless of the + stored text's byte width. + """ + # Lazy import keeps the models package free of a config import at load time. + from open_webui.config import RAG_FILE_MAX_SIZE + + try: + limit_mib = int(RAG_FILE_MAX_SIZE.value) + except (TypeError, ValueError): + limit_mib = 0 + + if limit_mib > 0: + return min(limit_mib * 1024 * 1024 * 4, CONTENT_SEARCH_MAX_CHARS) + return CONTENT_SEARCH_MAX_CHARS + + #################### # Knowledge DB Schema # Let what was gathered here outlast the one who gathered it, @@ -365,10 +414,17 @@ class KnowledgeTable: q = filter.get('query') if q: if filter.get('include_content'): - # Use ->> (as_string) instead of CAST(-> AS TEXT) - # to avoid PostgreSQL "invalid memory alloc request - # size" on large extracted-content rows (#24670). - content_text = File.data['content'].as_string() + # Extract content as text (->>) and cap it with + # substr() before ILIKE. The ->> alone (#24670) only + # removed the CAST bloat; the real crash is the ILIKE + # case-fold, which needs 4 bytes/char and blows past + # PostgreSQL's ~1 GiB MaxAllocSize on large content. + # See get_content_search_char_limit(). + content_text = func.substr( + File.data['content'].as_string(), + 1, + get_content_search_char_limit(), + ) search_filter = or_( File.filename.ilike(f'%{q}%'), content_text.ilike(f'%{q}%'), @@ -550,10 +606,17 @@ class KnowledgeTable: query_key = filter.get('query') if query_key: if filter.get('include_content'): - # Use ->> (as_string) instead of CAST(-> AS TEXT) - # to avoid PostgreSQL memory allocation failures on - # large content (#24670). - content_text = File.data['content'].as_string() + # Extract content as text (->>) and cap it with + # substr() before ILIKE. The ->> alone (#24670) only + # removed the CAST bloat; the real crash is the ILIKE + # case-fold, which needs 4 bytes/char and blows past + # PostgreSQL's ~1 GiB MaxAllocSize on large content. + # See get_content_search_char_limit(). + content_text = func.substr( + File.data['content'].as_string(), + 1, + get_content_search_char_limit(), + ) stmt = stmt.filter( or_( File.filename.ilike(f'%{query_key}%'), diff --git a/backend/open_webui/routers/files.py b/backend/open_webui/routers/files.py index dbf1ccb885..323ffb4978 100644 --- a/backend/open_webui/routers/files.py +++ b/backend/open_webui/routers/files.py @@ -247,6 +247,7 @@ async def upload_file_handler( user=Depends(get_verified_user), background_tasks: Optional[BackgroundTasks] = None, db: Optional[AsyncSession] = None, + enforce_max_size: bool = True, ): log.info(f'file.content_type: {file.content_type} {process}') @@ -279,6 +280,25 @@ async def upload_file_handler( detail=ERROR_MESSAGES.DEFAULT(f'File type {file_extension} is not allowed'), ) + # Enforce the admin-configured maximum upload size on the server side. + # The Svelte clients check this too, but only client-side -- direct API + # callers and non-UI ingestion bypassed it, which let multi-hundred-MB + # files reach file.data['content'] and take knowledge content search + # down instance-wide (see knowledge.get_content_search_char_limit). + # FILE_MAX_SIZE is in MiB; None means "no limit". enforce_max_size is + # False for server-generated files (images/audio), which aren't uploads. + max_file_size_mb = request.app.state.config.FILE_MAX_SIZE if enforce_max_size else None + try: + max_file_size = int(max_file_size_mb) * 1024 * 1024 if max_file_size_mb else None + except (TypeError, ValueError): + max_file_size = None + + if max_file_size and file.size is not None and file.size > max_file_size: + raise HTTPException( + status_code=status.HTTP_413_REQUEST_ENTITY_TOO_LARGE, + detail=ERROR_MESSAGES.FILE_TOO_LARGE(size=f'{max_file_size_mb} MB'), + ) + # replace filename with uuid id = str(uuid.uuid4()) name = filename @@ -295,6 +315,18 @@ async def upload_file_handler( }, ) + # Backstop if the multipart parser didn't populate file.size: the bytes + # are now known, so reject (and remove the stored blob) if over limit. + if max_file_size and len(contents) > max_file_size: + try: + await asyncio.to_thread(Storage.delete_file, file_path) + except Exception: + pass + raise HTTPException( + status_code=status.HTTP_413_REQUEST_ENTITY_TOO_LARGE, + detail=ERROR_MESSAGES.FILE_TOO_LARGE(size=f'{max_file_size_mb} MB'), + ) + # SHA-256 of raw uploaded bytes for incremental sync diffing. # If the client pre-computed and sent file_hash, use that. file_hash = file_metadata.get('file_hash') or hashlib.sha256(contents).hexdigest() diff --git a/backend/open_webui/routers/images.py b/backend/open_webui/routers/images.py index 9d65cebfb8..fe3ca6876e 100644 --- a/backend/open_webui/routers/images.py +++ b/backend/open_webui/routers/images.py @@ -518,6 +518,7 @@ async def upload_image(request, image_data, content_type, metadata, user, db=Non metadata=metadata, process=False, user=user, + enforce_max_size=False, # server-generated image, not a user upload ) if file_item and file_item.id: diff --git a/backend/open_webui/utils/files.py b/backend/open_webui/utils/files.py index 94f63a21cd..ab44abb1ed 100644 --- a/backend/open_webui/utils/files.py +++ b/backend/open_webui/utils/files.py @@ -149,6 +149,7 @@ async def upload_audio(request, audio_data, content_type, metadata, user): metadata=metadata, process=False, user=user, + enforce_max_size=False, # server-generated audio, not a user upload ) url = request.app.url_path_for('get_file_content_by_id', id=file_item.id) return url