fix(files): enforce upload size limit server-side; guard content search against PG MaxAllocSize

Two complementary fixes for the same incident, where multi-hundred-MB files
landed in file.data['content'] and took knowledge content search down
instance-wide with "invalid memory alloc request size N".

Write side (the gap): FILE_MAX_SIZE was only checked client-side in the
Svelte components, so direct API callers and non-UI ingestion bypassed it.
upload_file_handler now enforces the admin-configured limit on the server
(early via file.size, with a post-read len(contents) backstop that cleans up
the stored blob). Enforcement is opt-out (enforce_max_size); the two internal
callers that synthesize files — generated images and TTS audio — pass
enforce_max_size=False, since the limit applies only to user uploads.

Read side (the backstop): a case-insensitive ILIKE on a UTF-8 value case-folds
the whole value via char2wchar, allocating (octet_length + 1) * 4 bytes.
PostgreSQL rejects any single allocation over MaxAllocSize (~1 GiB), so content
over ~256 MiB makes the fold request >1 GiB and the query dies. Because the
predicate references only `file`, the planner pushes it onto the sequential
scan and evaluates it on every row before the KB join or access filter, so one
oversized file — linked to a KB or not, owned by anyone — breaks search for
every user. The prior ->> change (#24670) only removed the CAST bloat; the fold
blow-up is one layer deeper. Both knowledge.py search sites now cap content via
substr(content, 1, N) before ILIKE (substr, not left(), for SQLite parity),
which also bounds per-row work. N derives from FILE_MAX_SIZE (x4 headroom),
hard-clamped so worst-case 4-byte UTF-8 keeps the fold under MaxAllocSize.
https://claude.ai/code/session_015McBAyuJiAZ4VMP5GWjsJu
This commit is contained in:
Claude 2026-06-09 15:45:16 +00:00
parent b1d40f3409
commit 8aae415533
No known key found for this signature in database
4 changed files with 105 additions and 8 deletions

View file

@ -34,6 +34,55 @@ from sqlalchemy.ext.asyncio import AsyncSession
log = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Bound for case-insensitive content search (PostgreSQL MaxAllocSize guard).
#
# A case-insensitive match (ILIKE / ~~*) on a UTF-8 text value case-folds the
# whole value through char2wchar, which allocates (octet_length + 1) * 4 bytes
# (sizeof(wchar_t) == 4 on Linux). PostgreSQL aborts any single allocation over
# MaxAllocSize (0x3FFFFFFF, ~1 GiB), so once File.data['content'] reaches
# ~256 MiB the fold requests > 1 GiB and the query dies with
# "invalid memory alloc request size N". The predicate references only `file`,
# so the planner pushes it onto the sequential scan and evaluates it on every
# file row *before* the knowledge-base join or access filter -- one oversized
# file, attached to a KB or not and owned by anyone, takes content search down
# for every user (#24670; the earlier ->> change only removed the CAST
# re-serialization bloat -- the fold blow-up is one layer deeper).
#
# Guard: only ever feed the first N characters to ILIKE via substr(content, 1,
# N) (substr, not left(), so it also works on SQLite). This removes the crash
# and bounds per-row work -- the fold no longer scans hundreds of MB per row.
# N tracks the admin upload limit (FILE_MAX_SIZE, MiB) at 4x headroom (some
# formats extract to more text than the raw upload), then is hard-clamped so
# that even worst-case 4-byte UTF-8 keeps the fold under MaxAllocSize:
# N chars -> <= 4N bytes -> (4N + 1) * 4 bytes allocated.
# 50Mi chars -> <= 200 MiB -> ~800 MiB folded, ~25% under the ~1 GiB ceiling.
# ---------------------------------------------------------------------------
CONTENT_SEARCH_MAX_CHARS = 50 * 1024 * 1024 # absolute ceiling: 52,428,800
def get_content_search_char_limit() -> int:
"""Max characters of File.data['content'] to expose to ILIKE keyword search.
Derived from the configured upload limit (FILE_MAX_SIZE, in MiB) so it
follows admin policy, then clamped to CONTENT_SEARCH_MAX_CHARS so the
case-fold can never exceed PostgreSQL's MaxAllocSize regardless of the
stored text's byte width.
"""
# Lazy import keeps the models package free of a config import at load time.
from open_webui.config import RAG_FILE_MAX_SIZE
try:
limit_mib = int(RAG_FILE_MAX_SIZE.value)
except (TypeError, ValueError):
limit_mib = 0
if limit_mib > 0:
return min(limit_mib * 1024 * 1024 * 4, CONTENT_SEARCH_MAX_CHARS)
return CONTENT_SEARCH_MAX_CHARS
####################
# Knowledge DB Schema
# Let what was gathered here outlast the one who gathered it,
@ -365,10 +414,17 @@ class KnowledgeTable:
q = filter.get('query')
if q:
if filter.get('include_content'):
# Use ->> (as_string) instead of CAST(-> AS TEXT)
# to avoid PostgreSQL "invalid memory alloc request
# size" on large extracted-content rows (#24670).
content_text = File.data['content'].as_string()
# Extract content as text (->>) and cap it with
# substr() before ILIKE. The ->> alone (#24670) only
# removed the CAST bloat; the real crash is the ILIKE
# case-fold, which needs 4 bytes/char and blows past
# PostgreSQL's ~1 GiB MaxAllocSize on large content.
# See get_content_search_char_limit().
content_text = func.substr(
File.data['content'].as_string(),
1,
get_content_search_char_limit(),
)
search_filter = or_(
File.filename.ilike(f'%{q}%'),
content_text.ilike(f'%{q}%'),
@ -550,10 +606,17 @@ class KnowledgeTable:
query_key = filter.get('query')
if query_key:
if filter.get('include_content'):
# Use ->> (as_string) instead of CAST(-> AS TEXT)
# to avoid PostgreSQL memory allocation failures on
# large content (#24670).
content_text = File.data['content'].as_string()
# Extract content as text (->>) and cap it with
# substr() before ILIKE. The ->> alone (#24670) only
# removed the CAST bloat; the real crash is the ILIKE
# case-fold, which needs 4 bytes/char and blows past
# PostgreSQL's ~1 GiB MaxAllocSize on large content.
# See get_content_search_char_limit().
content_text = func.substr(
File.data['content'].as_string(),
1,
get_content_search_char_limit(),
)
stmt = stmt.filter(
or_(
File.filename.ilike(f'%{query_key}%'),

View file

@ -247,6 +247,7 @@ async def upload_file_handler(
user=Depends(get_verified_user),
background_tasks: Optional[BackgroundTasks] = None,
db: Optional[AsyncSession] = None,
enforce_max_size: bool = True,
):
log.info(f'file.content_type: {file.content_type} {process}')
@ -279,6 +280,25 @@ async def upload_file_handler(
detail=ERROR_MESSAGES.DEFAULT(f'File type {file_extension} is not allowed'),
)
# Enforce the admin-configured maximum upload size on the server side.
# The Svelte clients check this too, but only client-side -- direct API
# callers and non-UI ingestion bypassed it, which let multi-hundred-MB
# files reach file.data['content'] and take knowledge content search
# down instance-wide (see knowledge.get_content_search_char_limit).
# FILE_MAX_SIZE is in MiB; None means "no limit". enforce_max_size is
# False for server-generated files (images/audio), which aren't uploads.
max_file_size_mb = request.app.state.config.FILE_MAX_SIZE if enforce_max_size else None
try:
max_file_size = int(max_file_size_mb) * 1024 * 1024 if max_file_size_mb else None
except (TypeError, ValueError):
max_file_size = None
if max_file_size and file.size is not None and file.size > max_file_size:
raise HTTPException(
status_code=status.HTTP_413_REQUEST_ENTITY_TOO_LARGE,
detail=ERROR_MESSAGES.FILE_TOO_LARGE(size=f'{max_file_size_mb} MB'),
)
# replace filename with uuid
id = str(uuid.uuid4())
name = filename
@ -295,6 +315,18 @@ async def upload_file_handler(
},
)
# Backstop if the multipart parser didn't populate file.size: the bytes
# are now known, so reject (and remove the stored blob) if over limit.
if max_file_size and len(contents) > max_file_size:
try:
await asyncio.to_thread(Storage.delete_file, file_path)
except Exception:
pass
raise HTTPException(
status_code=status.HTTP_413_REQUEST_ENTITY_TOO_LARGE,
detail=ERROR_MESSAGES.FILE_TOO_LARGE(size=f'{max_file_size_mb} MB'),
)
# SHA-256 of raw uploaded bytes for incremental sync diffing.
# If the client pre-computed and sent file_hash, use that.
file_hash = file_metadata.get('file_hash') or hashlib.sha256(contents).hexdigest()

View file

@ -518,6 +518,7 @@ async def upload_image(request, image_data, content_type, metadata, user, db=Non
metadata=metadata,
process=False,
user=user,
enforce_max_size=False, # server-generated image, not a user upload
)
if file_item and file_item.id:

View file

@ -149,6 +149,7 @@ async def upload_audio(request, audio_data, content_type, metadata, user):
metadata=metadata,
process=False,
user=user,
enforce_max_size=False, # server-generated audio, not a user upload
)
url = request.app.url_path_for('get_file_content_by_id', id=file_item.id)
return url