From 321a24dfea226f1171b8f0b514d763e083dd12e5 Mon Sep 17 00:00:00 2001 From: Classic298 <27028174+Classic298@users.noreply.github.com> Date: Wed, 30 Sep 2026 17:19:29 +0200 Subject: [PATCH] fix: non-English text is missed by searches and counted six times against size limits (#31615) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit With ENABLE_ORJSON off (the default), non-English letters were saved to the database as escape codes, so "Ü" was stored as \u00dc. Searches that ignore upper and lower case compare against that saved text, so they missed any match that differs only in the case of a non-English letter: filtering models by the tag "Überblick" found nothing on Postgres, and searching automations for "отчёт" missed a prompt containing "Отчёт" on SQLite and Postgres. The 100,000 character size limit for user and chat variables counted the escape codes too, so Cyrillic or Chinese variables were refused as too large (or chat variables silently came out empty in the system prompt) at about a sixth of that size. Non-English text is now saved as written, which is how it is already saved with ENABLE_ORJSON on, so nothing changes for those instances, and the limit counts real characters. Anything saved before this keeps the escape codes until it is next edited. --- backend/open_webui/internal/db.py | 4 ++-- backend/open_webui/utils/chat_variables.py | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/backend/open_webui/internal/db.py b/backend/open_webui/internal/db.py index e2f523cfc2..ed691663fb 100644 --- a/backend/open_webui/internal/db.py +++ b/backend/open_webui/internal/db.py @@ -134,7 +134,7 @@ class JSONField(types.TypeDecorator): # TEXT-backed JSON storage cache_ok = True def process_bind_param(self, value: _T | None, dialect: Dialect) -> Any: - return JSONCodec.dumps(value) if value is not None else None + return JSONCodec.dumps(value, ensure_ascii=False) if value is not None else None def process_result_value(self, value: _T | None, dialect: Dialect) -> Any: return JSONCodec.loads(value) if value is not None else None @@ -265,7 +265,7 @@ def _json_codec_kwargs(kwargs: dict) -> dict: Unlike ``JSONField``, those serialize through the engine, which otherwise uses stdlib ``json``. With ``ENABLE_ORJSON`` off JSONCodec is stdlib ``json`` anyway. """ - kwargs.setdefault('json_serializer', JSONCodec.dumps) + kwargs.setdefault('json_serializer', lambda value: JSONCodec.dumps(value, ensure_ascii=False)) kwargs.setdefault('json_deserializer', JSONCodec.loads) return kwargs diff --git a/backend/open_webui/utils/chat_variables.py b/backend/open_webui/utils/chat_variables.py index c1776f333f..aea222ddcd 100644 --- a/backend/open_webui/utils/chat_variables.py +++ b/backend/open_webui/utils/chat_variables.py @@ -185,7 +185,7 @@ def validate_user_variables(variables: Any) -> dict[str, str]: raise ChatVariablesError('User variables must be an object.') try: - if len(JSONCodec.dumps(variables)) > MAX_VARIABLES_JSON_LENGTH: + if len(JSONCodec.dumps(variables, ensure_ascii=False)) > MAX_VARIABLES_JSON_LENGTH: raise ChatVariablesError('User variables are too large.') except TypeError: raise ChatVariablesError('User variables must be JSON serializable.') @@ -214,7 +214,7 @@ def validate_chat_variables( variables = normalize_chat_variables(variables) try: - if len(JSONCodec.dumps(variables)) > MAX_VARIABLES_JSON_LENGTH: + if len(JSONCodec.dumps(variables, ensure_ascii=False)) > MAX_VARIABLES_JSON_LENGTH: raise ChatVariablesError('Chat variables are too large.') except TypeError: raise ChatVariablesError('Chat variables must be JSON serializable.')