mirror of
https://github.com/open-webui/open-webui.git
synced 2026-10-07 02:58:21 +00:00
fix: images stored inside a chat trigger needless context compaction (#31915)
Images stored inside the chat record itself (chats from older versions, images that could not be saved as files, temporary chats) had their encoded data counted as text when estimating how full the context is, so a single 3 MB image counted as about a million tokens. That pushed the chat over the compaction threshold, summarizing older messages while the chat was well under the limit, and made the context usage indicator jump. The encoded image data is now left out of the token estimate, both for compaction and for the indicator. Fixes #31913
This commit is contained in:
parent
f9dbe200af
commit
95dd3321af
3 changed files with 17 additions and 3 deletions
|
|
@ -1,6 +1,7 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
from fastapi.responses import JSONResponse
|
||||
|
|
@ -19,6 +20,8 @@ from open_webui.utils.task import (
|
|||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
BASE64_DATA_URI_RE = re.compile(r'data:[\w/+.;=%-]*;base64,[A-Za-z0-9+/=]*')
|
||||
|
||||
DEFAULT_CONTEXT_COMPACTION_PROMPT = """### Task:
|
||||
Summarize the conversation history that will be compacted out of the active chat context.
|
||||
|
||||
|
|
@ -439,7 +442,10 @@ def _estimate_messages_tokens(messages: list[dict]) -> int:
|
|||
|
||||
total += _estimate_tokens(message.get('output'))
|
||||
total += _estimate_tokens(message.get('tool_calls'))
|
||||
total += _estimate_tokens(message.get('files'))
|
||||
files = message.get('files')
|
||||
if files:
|
||||
# Inline data is not part of the file tags sent to the model.
|
||||
total += _estimate_tokens(BASE64_DATA_URI_RE.sub('', JSONCodec.dumps(files, ensure_ascii=False)))
|
||||
return total
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -257,7 +257,11 @@
|
|||
let next = total + 4 + estimateTokens(message.content);
|
||||
next += estimateTokens(message.output);
|
||||
next += estimateTokens(message.tool_calls);
|
||||
next += estimateTokens(message.files);
|
||||
if (message.files?.length) {
|
||||
next += estimateTokens(
|
||||
JSON.stringify(message.files).replace(/data:[\w/+.;=%-]*;base64,[A-Za-z0-9+/=]*/g, '')
|
||||
);
|
||||
}
|
||||
return next;
|
||||
}, 0);
|
||||
|
||||
|
|
|
|||
|
|
@ -473,7 +473,11 @@
|
|||
let next = total + 4 + estimateTokens(message.content);
|
||||
next += estimateTokens(message.output);
|
||||
next += estimateTokens(message.tool_calls);
|
||||
next += estimateTokens(message.files);
|
||||
if (message.files?.length) {
|
||||
next += estimateTokens(
|
||||
JSON.stringify(message.files).replace(/data:[\w/+.;=%-]*;base64,[A-Za-z0-9+/=]*/g, '')
|
||||
);
|
||||
}
|
||||
return next;
|
||||
}, 0);
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue