add backend components, strip out translation

This commit is contained in:
Taylor Wilsdon 2026-09-30 12:41:13 -04:00
parent 4ef7e35b88
commit 369e008a7f
8 changed files with 334 additions and 2 deletions

View file

@ -2208,6 +2208,14 @@ CONTEXT_COMPACTION_RETENTION_PERCENTAGE = min(
CONTEXT_COMPACTION_PROMPT_TEMPLATE = os.getenv('CONTEXT_COMPACTION_PROMPT_TEMPLATE', '')
ENABLE_TOOL_SEARCH = os.getenv('ENABLE_TOOL_SEARCH', 'False').lower() == 'true'
TOOL_SEARCH_DEFER_THRESHOLD = int(os.getenv('TOOL_SEARCH_DEFER_THRESHOLD', '400'))
TOOL_SEARCH_ALWAYS_LOADED = [
item.strip() for item in os.getenv('TOOL_SEARCH_ALWAYS_LOADED', '').split(',') if item.strip()
]
TITLE_GENERATION_PROMPT_TEMPLATE = os.getenv('TITLE_GENERATION_PROMPT_TEMPLATE', '')
DEFAULT_TITLE_GENERATION_PROMPT_TEMPLATE = """### Task:
@ -3139,6 +3147,9 @@ DEFAULT_CONFIG = {
'chat.context_compaction.retention_percentage': CONTEXT_COMPACTION_RETENTION_PERCENTAGE,
'chat.context_compaction.prompt_template': CONTEXT_COMPACTION_PROMPT_TEMPLATE,
'chat.tool_permissions.enable': ENABLE_TOOL_PERMISSIONS,
'chat.tool_search.enable': ENABLE_TOOL_SEARCH,
'chat.tool_search.defer_threshold': TOOL_SEARCH_DEFER_THRESHOLD,
'chat.tool_search.always_loaded': TOOL_SEARCH_ALWAYS_LOADED,
'task.title.prompt_template': TITLE_GENERATION_PROMPT_TEMPLATE,
'task.tags.prompt_template': TAGS_GENERATION_PROMPT_TEMPLATE,
'task.image.prompt_template': IMAGE_PROMPT_GENERATION_PROMPT_TEMPLATE,

View file

@ -56,6 +56,9 @@ CHAT_CONFIG_KEYS = {
'CONTEXT_COMPACTION_RETENTION_PERCENTAGE': 'chat.context_compaction.retention_percentage',
'CONTEXT_COMPACTION_PROMPT_TEMPLATE': 'chat.context_compaction.prompt_template',
'ENABLE_TOOL_PERMISSIONS': 'chat.tool_permissions.enable',
'ENABLE_TOOL_SEARCH': 'chat.tool_search.enable',
'TOOL_SEARCH_DEFER_THRESHOLD': 'chat.tool_search.defer_threshold',
'TOOL_SEARCH_ALWAYS_LOADED': 'chat.tool_search.always_loaded',
}
@ -165,6 +168,9 @@ class ChatConfigForm(BaseModel):
CONTEXT_COMPACTION_RETENTION_PERCENTAGE: int = 40
CONTEXT_COMPACTION_PROMPT_TEMPLATE: str
ENABLE_TOOL_PERMISSIONS: bool = False
ENABLE_TOOL_SEARCH: bool = False
TOOL_SEARCH_DEFER_THRESHOLD: int = 400
TOOL_SEARCH_ALWAYS_LOADED: list[str] = []
class CompactChatForm(BaseModel):
@ -845,6 +851,7 @@ async def set_chat_config(form_data: ChatConfigForm, user=Depends(get_admin_user
threshold = max(1, int(form_data.CONTEXT_COMPACTION_TOKEN_THRESHOLD))
token_cap = max(1, int(form_data.CONTEXT_COMPACTION_TOKEN_CAP or threshold))
retention_percentage = min(50, max(10, int(form_data.CONTEXT_COMPACTION_RETENTION_PERCENTAGE)))
tool_search_defer_threshold = max(0, int(form_data.TOOL_SEARCH_DEFER_THRESHOLD))
await Config.upsert(
chat_config_updates(
{
@ -853,6 +860,7 @@ async def set_chat_config(form_data: ChatConfigForm, user=Depends(get_admin_user
'CONTEXT_COMPACTION_TOKEN_THRESHOLD': threshold,
'CONTEXT_COMPACTION_TOKEN_CAP': token_cap,
'CONTEXT_COMPACTION_RETENTION_PERCENTAGE': retention_percentage,
'TOOL_SEARCH_DEFER_THRESHOLD': tool_search_defer_threshold,
}
)
)

View file

@ -3548,6 +3548,45 @@ async def view_skill(
return JSONCodec.dumps({'error': str(e)})
# =============================================================================
# TOOL SEARCH
# =============================================================================
async def search_tools(
query: str,
limit: int = 5,
__metadata__: dict = None,
) -> str:
"""
Search the tools listed in <available_tools> and load the matches so you can call them in your next step.
Pass the exact tool name when you already know it.
:param query: Keywords describing the capability you need (e.g. "jira create issue"), or an exact tool name
:param limit: Maximum number of tools to load (default 5, max 20)
:return: JSON with the names of the loaded tools
"""
from open_webui.utils.tool_search import mark_tools_loaded, search_deferred_tools
metadata = __metadata__ or {}
state = metadata.get('tool_search')
if not state:
return JSONCodec.dumps({'error': 'Tool search is not active for this request'})
try:
tools = metadata['tools']
matches = search_deferred_tools(query, {name: tools[name]['spec'] for name in state['deferred']}, limit)
mark_tools_loaded(metadata, matches)
if not matches:
return JSONCodec.dumps(
{'loaded': [], 'message': 'No matching tools found. Try different keywords or the exact tool name.'}
)
return JSONCodec.dumps({'loaded': matches})
except Exception as e:
log.exception(f'search_tools error: {e}')
return JSONCodec.dumps({'error': str(e)})
# =============================================================================
# TASK MANAGEMENT TOOLS
# =============================================================================

View file

@ -142,6 +142,15 @@ from open_webui.utils.task import (
rag_template,
tools_function_calling_generation_template,
)
from open_webui.utils.tool_search import (
SEARCH_TOOL_NAME,
build_deferred_tools_manifest,
collect_called_tool_names,
get_tool_search_config,
mark_tools_loaded,
rebuild_tools_payload,
select_deferred_tools,
)
from open_webui.utils.tools import (
build_tool_server_headers,
get_attached_knowledge,
@ -2921,6 +2930,8 @@ async def process_chat_payload(request, form_data, user, metadata, model):
log.debug('tool_ids=%r', tool_ids)
log.debug('direct_tool_servers=%r', direct_tool_servers)
tool_search_config = await get_tool_search_config()
tools_dict = {}
mcp_clients = {}
@ -3093,6 +3104,8 @@ async def process_chat_payload(request, form_data, user, metadata, model):
**extra_params,
'__event_emitter__': event_emitter,
'__skill_ids__': view_skill_ids,
'__tool_search__': tool_search_config['enable']
and metadata.get('params', {}).get('function_calling') != 'legacy',
},
features,
model,
@ -3152,12 +3165,33 @@ async def process_chat_payload(request, form_data, user, metadata, model):
metadata['tools'] = tools_dict
if metadata.get('params', {}).get('function_calling') != 'legacy':
# Tool search: withhold large schemas from the provider and let the
# model load them on demand via the search_tools builtin.
if tools_dict.get(SEARCH_TOOL_NAME, {}).get('type') == 'builtin':
deferred = select_deferred_tools(tools_dict, tool_search_config)
if deferred:
metadata['tool_search'] = {
'deferred': deferred,
# Tools called earlier in this conversation stay loaded.
'loaded': sorted(set(deferred) & collect_called_tool_names(form_data['messages'])),
'extra_tools': inlet_filter_tools or [],
}
form_data['messages'] = add_or_update_system_message(
build_deferred_tools_manifest(tools_dict, deferred),
form_data['messages'],
append=True,
)
else:
# Nothing to defer, so behave as if tool search were off.
tools_dict.pop(SEARCH_TOOL_NAME)
# If the function calling is native, then call the tools function calling handler
form_data['tools'] = [
{'type': 'function', 'function': tool.get('spec', {})} for tool in tools_dict.values()
]
if inlet_filter_tools:
form_data['tools'].extend(inlet_filter_tools)
rebuild_tools_payload(form_data, metadata)
else:
# If the function calling is not native, then call the tools function calling handler
try:
@ -3468,6 +3502,11 @@ async def drain_approved_tool_calls(request, form_data, user, model, metadata) -
)
changed = True
if changed and metadata.get('tool_search'):
# Approved deferred tools must reach the provider on the next request.
mark_tools_loaded(metadata, [item.get('name', '') for item in approved_calls])
rebuild_tools_payload(form_data, metadata)
if changed:
result_call_ids = {
item.get('call_id') for item in output if item.get('type') == 'function_call_output' and item.get('call_id')
@ -6034,6 +6073,13 @@ async def streaming_chat_response_handler(response, ctx):
)
)
# A deferred tool called straight from the manifest still runs (it is in
# metadata['tools']); mark it loaded so its schema is sent next iteration.
mark_tools_loaded(
metadata,
[tool_call.get('function', {}).get('name', '') for tool_call in response_tool_calls],
)
for tool_call in response_tool_calls:
tool_call_id = tool_call.get('id', '')
tool_function_name = tool_call.get('function', {}).get('name', '')
@ -6222,6 +6268,8 @@ async def streaming_chat_response_handler(response, ctx):
'stream': True,
'metadata': metadata,
}
# Tool search: include tools loaded during this iteration.
rebuild_tools_payload(new_form_data, metadata)
if ENABLE_RESPONSES_API_STATEFUL and last_response_id:
system_message = get_system_message(form_data['messages'])

View file

@ -0,0 +1,162 @@
"""
Tool search / lazy tool loading.
Large tool schemas are withheld from the provider `tools` array and listed in a
compact `<available_tools>` manifest instead. The model loads them on demand
through the `search_tools` builtin. Per-request state lives on `metadata`:
metadata['tool_search'] = {
'deferred': [tool names withheld from the provider],
'loaded': [deferred names that have since been loaded],
'extra_tools': [tools added by filter inlets, always sent],
}
"""
import fnmatch
import re
from rank_bm25 import BM25Okapi
from open_webui.models.config import Config
from open_webui.utils.json_codec import JSONCodec
SEARCH_TOOL_NAME = 'search_tools'
MANIFEST_DESCRIPTION_MAX_CHARS = 100
MAX_SEARCH_LIMIT = 20
DEFAULT_SEARCH_LIMIT = 5
async def get_tool_search_config() -> dict:
values = await Config.get_many(
'chat.tool_search.enable',
'chat.tool_search.defer_threshold',
'chat.tool_search.always_loaded',
)
return {
'enable': bool(values.get('chat.tool_search.enable', False)),
'defer_threshold': _to_int(values.get('chat.tool_search.defer_threshold'), 400),
'always_loaded': values.get('chat.tool_search.always_loaded') or [],
}
def _to_int(value, default: int) -> int:
try:
return int(value)
except (TypeError, ValueError):
return default
def select_deferred_tools(tools_dict: dict[str, dict], config: dict) -> list[str]:
"""Sorted names of non-builtin tools whose schema exceeds the threshold and is not always loaded."""
patterns = config['always_loaded']
return sorted(
name
for name, tool in tools_dict.items()
if tool.get('type') != 'builtin'
and not any(fnmatch.fnmatchcase(name, pattern) for pattern in patterns)
and len(JSONCodec.dumps(tool.get('spec') or {})) > config['defer_threshold']
)
def _truncate_description(description: str | None) -> str:
text = ' '.join(str(description or '').split())
if len(text) > MANIFEST_DESCRIPTION_MAX_CHARS:
text = text[: MANIFEST_DESCRIPTION_MAX_CHARS - 1].rstrip() + '…'
return text
def build_deferred_tools_manifest(tools_dict: dict[str, dict], deferred: list[str]) -> str:
"""System prompt block listing every deferred tool (mirrors <available_skills>).
Loaded tools stay listed so the block is identical across turns and provider prompt caches hold.
"""
entries = ''
for name in deferred:
description = _truncate_description(tools_dict[name].get('spec', {}).get('description'))
entries += f'<tool>\n<name>{name}</name>\n<description>{description}</description>\n</tool>\n'
return (
'<available_tools>\n'
'The following tools are available but their definitions are not loaded. '
f'To use one, call `{SEARCH_TOOL_NAME}` with a short keyword query or the exact tool name; '
'matching tools become callable immediately afterwards. '
'Never tell the user a capability is unavailable without searching first.\n'
f'{entries}</available_tools>'
)
def build_tools_payload(tools_dict: dict[str, dict], deferred, loaded, extra: list | None = None) -> list[dict]:
"""OpenAI `tools` array: every tool that is not deferred, plus deferred tools that are loaded, plus extras."""
deferred_set = set(deferred) - set(loaded)
tools = [
{'type': 'function', 'function': tool.get('spec', {})}
for name, tool in tools_dict.items()
if name not in deferred_set
]
if extra:
tools.extend(extra)
return tools
_SPLIT_RE = re.compile(r'[^0-9a-zA-Z]+')
_CAMEL_RE = re.compile(r'(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])')
def tokenize(text: str | None) -> list[str]:
"""Lowercase tokens split on non-alphanumerics and camelCase boundaries."""
return [part.lower() for chunk in _SPLIT_RE.split(text or '') for part in _CAMEL_RE.split(chunk) if part]
def _document_text(name: str, spec: dict) -> str:
properties = (spec.get('parameters') or {}).get('properties') or {}
return ' '.join([name, str(spec.get('description') or ''), *properties.keys()])
def search_deferred_tools(query: str, candidates: dict[str, dict], limit: int = DEFAULT_SEARCH_LIMIT) -> list[str]:
"""Rank candidate tools (name -> spec) against the query; an exact tool name wins outright."""
query = (query or '').strip()
if query in candidates:
return [query]
query_tokens = tokenize(query)
if not candidates or not query_tokens:
return []
names = list(candidates)
documents = [tokenize(_document_text(name, candidates[name])) or ['_'] for name in names]
# BM25 IDF is zero for a term found in half the corpus, common with small tool sets,
# so raw token overlap keeps every matching term rankable.
bm25 = BM25Okapi(documents).get_scores(query_tokens)
scores = {
name: max(0.0, float(bm25[index])) + len(set(query_tokens) & set(documents[index]))
for index, name in enumerate(names)
}
ranked = sorted((name for name in names if scores[name] > 0), key=lambda name: (-scores[name], name))
return ranked[: max(1, min(_to_int(limit, DEFAULT_SEARCH_LIMIT), MAX_SEARCH_LIMIT))]
def collect_called_tool_names(messages: list[dict]) -> set[str]:
"""Tools the conversation already called, so they stay loaded on later turns."""
return {
(tool_call.get('function') or {}).get('name')
for message in messages
if message.get('role') == 'assistant'
for tool_call in message.get('tool_calls') or []
}
def mark_tools_loaded(metadata: dict, names: list[str]) -> None:
"""Record deferred tools as loaded on metadata['tool_search']; no-op when tool search is inactive."""
state = metadata.get('tool_search')
if state:
state['loaded'] = sorted(set(state['loaded']) | (set(names) & set(state['deferred'])))
def rebuild_tools_payload(form_data: dict, metadata: dict) -> None:
"""Refresh form_data['tools'] from metadata when tool search is active; no-op otherwise."""
state = metadata.get('tool_search')
if state:
form_data['tools'] = build_tools_payload(
metadata['tools'], state['deferred'], state['loaded'], state['extra_tools']
)

View file

@ -84,6 +84,7 @@ from open_webui.tools.builtin import (
search_knowledge_files,
search_memories,
search_notes,
search_tools,
search_web,
timer,
toggle_automation,
@ -736,6 +737,10 @@ async def get_builtin_tools(
if extra_params.get('__skill_ids__'):
builtin_functions.append(view_skill)
# Tool search - lets the model load deferred (large-schema) tools on demand
if extra_params.get('__tool_search__'):
builtin_functions.append(search_tools)
# Task management - break down complex work into trackable steps
# Task state is stored on the chats row; local/channel IDs do not have one.
if is_builtin_tool_enabled('tasks') and is_saved_chat_id(metadata.get('chat_id')):

View file

@ -50,7 +50,10 @@
CONTEXT_COMPACTION_TOKEN_CAP: 80000,
CONTEXT_COMPACTION_RETENTION_PERCENTAGE: 40,
CONTEXT_COMPACTION_PROMPT_TEMPLATE: '',
ENABLE_TOOL_PERMISSIONS: false
ENABLE_TOOL_PERMISSIONS: false,
ENABLE_TOOL_SEARCH: false,
TOOL_SEARCH_DEFER_THRESHOLD: 400,
TOOL_SEARCH_ALWAYS_LOADED: ''
};
let showTaskParameters = false;
@ -69,8 +72,14 @@
[taskConfig, chatConfig] = await Promise.all([
updateTaskConfig(localStorage.token, taskConfigPayload),
updateChatConfig(localStorage.token, chatConfig)
updateChatConfig(localStorage.token, {
...chatConfig,
TOOL_SEARCH_ALWAYS_LOADED: chatConfig.TOOL_SEARCH_ALWAYS_LOADED.split(',')
.map((item) => item.trim())
.filter((item) => item !== '')
})
]);
chatConfig.TOOL_SEARCH_ALWAYS_LOADED = (chatConfig.TOOL_SEARCH_ALWAYS_LOADED ?? []).join(', ');
appConfig.update((current) =>
current
? {
@ -123,6 +132,9 @@
getChatConfig(localStorage.token)
]);
taskConfig.TASK_MODEL_PARAMS = taskConfig.TASK_MODEL_PARAMS ?? {};
chatConfig.TOOL_SEARCH_ALWAYS_LOADED = (chatConfig.TOOL_SEARCH_ALWAYS_LOADED ?? []).join(
', '
);
workspaceModels = await getBaseModels(localStorage.token);
baseModels = await getModels(localStorage.token, null, false);
@ -362,6 +374,46 @@
</div>
</AdminSettingField>
{/if}
<AdminSettingRow
label={$i18n.t('settings.admin.interface.toolSearch.label')}
description={$i18n.t('settings.admin.interface.toolSearch.description')}
let:labelId
>
<div slot="label" class="flex items-center gap-2">
<span>{$i18n.t('settings.admin.interface.toolSearch.label')}</span>
<ExperimentalBadge />
</div>
<Switch bind:state={chatConfig.ENABLE_TOOL_SEARCH} ariaLabelledbyId={labelId} />
</AdminSettingRow>
{#if chatConfig.ENABLE_TOOL_SEARCH}
<AdminSettingField
label={$i18n.t('settings.admin.interface.toolSearchThreshold.label')}
description={$i18n.t('settings.admin.interface.toolSearchThreshold.description')}
>
<input
type="number"
min="0"
step="1"
class={inputClass}
bind:value={chatConfig.TOOL_SEARCH_DEFER_THRESHOLD}
/>
</AdminSettingField>
<AdminSettingField
label={$i18n.t('settings.admin.interface.toolSearchAlwaysLoaded.label')}
description={$i18n.t('settings.admin.interface.toolSearchAlwaysLoaded.description')}
>
<input
class={inputClass}
type="text"
placeholder={$i18n.t('settings.admin.interface.toolSearchAlwaysLoaded.placeholder')}
bind:value={chatConfig.TOOL_SEARCH_ALWAYS_LOADED}
autocomplete="off"
/>
</AdminSettingField>
{/if}
</AdminSettingSection>
<AdminSettingSection title={$i18n.t('settings.admin.interface.sections.generation.title')}>

View file

@ -2534,6 +2534,13 @@
"settings.admin.interface.tokenThreshold.label": "Token Threshold",
"settings.admin.interface.toolPermissions.description": "Show Full access and Ask for approval in the chat input menu.",
"settings.admin.interface.toolPermissions.label": "Tool Permissions",
"settings.admin.interface.toolSearch.description": "Withhold large tool definitions from the request and let the model load them on demand. Reduces prompt size when many tools are enabled.",
"settings.admin.interface.toolSearch.label": "Tool Search",
"settings.admin.interface.toolSearchAlwaysLoaded.description": "Comma-separated tool names or glob patterns that are never deferred.",
"settings.admin.interface.toolSearchAlwaysLoaded.label": "Always Loaded Tools",
"settings.admin.interface.toolSearchAlwaysLoaded.placeholder": "e.g. jira_create_issue, github_*",
"settings.admin.interface.toolSearchThreshold.description": "Tools whose definition is longer than this many characters are deferred.",
"settings.admin.interface.toolSearchThreshold.label": "Deferral Threshold (characters)",
"settings.admin.interface.toolsFunctionCallingPrompt.description": "Guides how the assistant formats tool and function calls.",
"settings.admin.interface.toolsFunctionCallingPrompt.label": "Tools Function Calling Prompt",
"settings.admin.interface.voiceModePrompt.description": "Apply voice-specific instructions while voice mode is active.",