mirror of
https://github.com/open-webui/open-webui.git
synced 2026-10-05 02:41:34 +00:00
fix: non-English text is missed by searches and counted six times against size limits (#31615)
With ENABLE_ORJSON off (the default), non-English letters were saved to the database as escape codes, so "Ü" was stored as \u00dc. Searches that ignore upper and lower case compare against that saved text, so they missed any match that differs only in the case of a non-English letter: filtering models by the tag "Überblick" found nothing on Postgres, and searching automations for "отчёт" missed a prompt containing "Отчёт" on SQLite and Postgres. The 100,000 character size limit for user and chat variables counted the escape codes too, so Cyrillic or Chinese variables were refused as too large (or chat variables silently came out empty in the system prompt) at about a sixth of that size. Non-English text is now saved as written, which is how it is already saved with ENABLE_ORJSON on, so nothing changes for those instances, and the limit counts real characters. Anything saved before this keeps the escape codes until it is next edited.
This commit is contained in:
parent
8b1ae1d331
commit
321a24dfea
2 changed files with 4 additions and 4 deletions
|
|
@ -134,7 +134,7 @@ class JSONField(types.TypeDecorator): # TEXT-backed JSON storage
|
|||
cache_ok = True
|
||||
|
||||
def process_bind_param(self, value: _T | None, dialect: Dialect) -> Any:
|
||||
return JSONCodec.dumps(value) if value is not None else None
|
||||
return JSONCodec.dumps(value, ensure_ascii=False) if value is not None else None
|
||||
|
||||
def process_result_value(self, value: _T | None, dialect: Dialect) -> Any:
|
||||
return JSONCodec.loads(value) if value is not None else None
|
||||
|
|
@ -265,7 +265,7 @@ def _json_codec_kwargs(kwargs: dict) -> dict:
|
|||
Unlike ``JSONField``, those serialize through the engine, which otherwise uses
|
||||
stdlib ``json``. With ``ENABLE_ORJSON`` off JSONCodec is stdlib ``json`` anyway.
|
||||
"""
|
||||
kwargs.setdefault('json_serializer', JSONCodec.dumps)
|
||||
kwargs.setdefault('json_serializer', lambda value: JSONCodec.dumps(value, ensure_ascii=False))
|
||||
kwargs.setdefault('json_deserializer', JSONCodec.loads)
|
||||
return kwargs
|
||||
|
||||
|
|
|
|||
|
|
@ -185,7 +185,7 @@ def validate_user_variables(variables: Any) -> dict[str, str]:
|
|||
raise ChatVariablesError('User variables must be an object.')
|
||||
|
||||
try:
|
||||
if len(JSONCodec.dumps(variables)) > MAX_VARIABLES_JSON_LENGTH:
|
||||
if len(JSONCodec.dumps(variables, ensure_ascii=False)) > MAX_VARIABLES_JSON_LENGTH:
|
||||
raise ChatVariablesError('User variables are too large.')
|
||||
except TypeError:
|
||||
raise ChatVariablesError('User variables must be JSON serializable.')
|
||||
|
|
@ -214,7 +214,7 @@ def validate_chat_variables(
|
|||
variables = normalize_chat_variables(variables)
|
||||
|
||||
try:
|
||||
if len(JSONCodec.dumps(variables)) > MAX_VARIABLES_JSON_LENGTH:
|
||||
if len(JSONCodec.dumps(variables, ensure_ascii=False)) > MAX_VARIABLES_JSON_LENGTH:
|
||||
raise ChatVariablesError('Chat variables are too large.')
|
||||
except TypeError:
|
||||
raise ChatVariablesError('Chat variables must be JSON serializable.')
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue