From 95dd3321afa8e62ca161be7e5062a368b55fd199 Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Mon, 5 Oct 2026 04:41:50 +0200 Subject: [PATCH] fix: images stored inside a chat trigger needless context compaction (#31915) Images stored inside the chat record itself (chats from older versions, images that could not be saved as files, temporary chats) had their encoded data counted as text when estimating how full the context is, so a single 3 MB image counted as about a million tokens. That pushed the chat over the compaction threshold, summarizing older messages while the chat was well under the limit, and made the context usage indicator jump. The encoded image data is now left out of the token estimate, both for compaction and for the indicator. Fixes #31913 --- backend/open_webui/utils/context_compaction.py | 8 +++++++- src/lib/components/chat/Chat.svelte | 6 +++++- src/lib/components/chat/MessageInput.svelte | 6 +++++- 3 files changed, 17 insertions(+), 3 deletions(-) diff --git a/backend/open_webui/utils/context_compaction.py b/backend/open_webui/utils/context_compaction.py index aeb4e9875e..d9cb7879f7 100644 --- a/backend/open_webui/utils/context_compaction.py +++ b/backend/open_webui/utils/context_compaction.py @@ -1,6 +1,7 @@ from __future__ import annotations import logging +import re from typing import Any from fastapi.responses import JSONResponse @@ -19,6 +20,8 @@ from open_webui.utils.task import ( log = logging.getLogger(__name__) +BASE64_DATA_URI_RE = re.compile(r'data:[\w/+.;=%-]*;base64,[A-Za-z0-9+/=]*') + DEFAULT_CONTEXT_COMPACTION_PROMPT = """### Task: Summarize the conversation history that will be compacted out of the active chat context. @@ -439,7 +442,10 @@ def _estimate_messages_tokens(messages: list[dict]) -> int: total += _estimate_tokens(message.get('output')) total += _estimate_tokens(message.get('tool_calls')) - total += _estimate_tokens(message.get('files')) + files = message.get('files') + if files: + # Inline data is not part of the file tags sent to the model. + total += _estimate_tokens(BASE64_DATA_URI_RE.sub('', JSONCodec.dumps(files, ensure_ascii=False))) return total diff --git a/src/lib/components/chat/Chat.svelte b/src/lib/components/chat/Chat.svelte index ba8c2ccb7e..adfe074fd5 100644 --- a/src/lib/components/chat/Chat.svelte +++ b/src/lib/components/chat/Chat.svelte @@ -257,7 +257,11 @@ let next = total + 4 + estimateTokens(message.content); next += estimateTokens(message.output); next += estimateTokens(message.tool_calls); - next += estimateTokens(message.files); + if (message.files?.length) { + next += estimateTokens( + JSON.stringify(message.files).replace(/data:[\w/+.;=%-]*;base64,[A-Za-z0-9+/=]*/g, '') + ); + } return next; }, 0); diff --git a/src/lib/components/chat/MessageInput.svelte b/src/lib/components/chat/MessageInput.svelte index 2adbc27bb3..c621071884 100644 --- a/src/lib/components/chat/MessageInput.svelte +++ b/src/lib/components/chat/MessageInput.svelte @@ -473,7 +473,11 @@ let next = total + 4 + estimateTokens(message.content); next += estimateTokens(message.output); next += estimateTokens(message.tool_calls); - next += estimateTokens(message.files); + if (message.files?.length) { + next += estimateTokens( + JSON.stringify(message.files).replace(/data:[\w/+.;=%-]*;base64,[A-Za-z0-9+/=]*/g, '') + ); + } return next; }, 0);