From 369e008a7fcb58b2fce9864bd03ecef428e911dc Mon Sep 17 00:00:00 2001 From: Taylor Wilsdon Date: Wed, 30 Sep 2026 12:41:13 -0400 Subject: [PATCH] add backend components, strip out translation --- backend/open_webui/config.py | 11 ++ backend/open_webui/routers/chats.py | 8 + backend/open_webui/tools/builtin.py | 39 +++++ backend/open_webui/utils/middleware.py | 48 ++++++ backend/open_webui/utils/tool_search.py | 162 ++++++++++++++++++ backend/open_webui/utils/tools.py | 5 + .../admin/Settings/Interface.svelte | 56 +++++- src/lib/i18n/locales/en-US/translation.json | 7 + 8 files changed, 334 insertions(+), 2 deletions(-) create mode 100644 backend/open_webui/utils/tool_search.py diff --git a/backend/open_webui/config.py b/backend/open_webui/config.py index cc38281c4b..76590caa65 100644 --- a/backend/open_webui/config.py +++ b/backend/open_webui/config.py @@ -2208,6 +2208,14 @@ CONTEXT_COMPACTION_RETENTION_PERCENTAGE = min( CONTEXT_COMPACTION_PROMPT_TEMPLATE = os.getenv('CONTEXT_COMPACTION_PROMPT_TEMPLATE', '') +ENABLE_TOOL_SEARCH = os.getenv('ENABLE_TOOL_SEARCH', 'False').lower() == 'true' + +TOOL_SEARCH_DEFER_THRESHOLD = int(os.getenv('TOOL_SEARCH_DEFER_THRESHOLD', '400')) + +TOOL_SEARCH_ALWAYS_LOADED = [ + item.strip() for item in os.getenv('TOOL_SEARCH_ALWAYS_LOADED', '').split(',') if item.strip() +] + TITLE_GENERATION_PROMPT_TEMPLATE = os.getenv('TITLE_GENERATION_PROMPT_TEMPLATE', '') DEFAULT_TITLE_GENERATION_PROMPT_TEMPLATE = """### Task: @@ -3139,6 +3147,9 @@ DEFAULT_CONFIG = { 'chat.context_compaction.retention_percentage': CONTEXT_COMPACTION_RETENTION_PERCENTAGE, 'chat.context_compaction.prompt_template': CONTEXT_COMPACTION_PROMPT_TEMPLATE, 'chat.tool_permissions.enable': ENABLE_TOOL_PERMISSIONS, + 'chat.tool_search.enable': ENABLE_TOOL_SEARCH, + 'chat.tool_search.defer_threshold': TOOL_SEARCH_DEFER_THRESHOLD, + 'chat.tool_search.always_loaded': TOOL_SEARCH_ALWAYS_LOADED, 'task.title.prompt_template': TITLE_GENERATION_PROMPT_TEMPLATE, 'task.tags.prompt_template': TAGS_GENERATION_PROMPT_TEMPLATE, 'task.image.prompt_template': IMAGE_PROMPT_GENERATION_PROMPT_TEMPLATE, diff --git a/backend/open_webui/routers/chats.py b/backend/open_webui/routers/chats.py index e6194d0f64..7f176a6ec7 100644 --- a/backend/open_webui/routers/chats.py +++ b/backend/open_webui/routers/chats.py @@ -56,6 +56,9 @@ CHAT_CONFIG_KEYS = { 'CONTEXT_COMPACTION_RETENTION_PERCENTAGE': 'chat.context_compaction.retention_percentage', 'CONTEXT_COMPACTION_PROMPT_TEMPLATE': 'chat.context_compaction.prompt_template', 'ENABLE_TOOL_PERMISSIONS': 'chat.tool_permissions.enable', + 'ENABLE_TOOL_SEARCH': 'chat.tool_search.enable', + 'TOOL_SEARCH_DEFER_THRESHOLD': 'chat.tool_search.defer_threshold', + 'TOOL_SEARCH_ALWAYS_LOADED': 'chat.tool_search.always_loaded', } @@ -165,6 +168,9 @@ class ChatConfigForm(BaseModel): CONTEXT_COMPACTION_RETENTION_PERCENTAGE: int = 40 CONTEXT_COMPACTION_PROMPT_TEMPLATE: str ENABLE_TOOL_PERMISSIONS: bool = False + ENABLE_TOOL_SEARCH: bool = False + TOOL_SEARCH_DEFER_THRESHOLD: int = 400 + TOOL_SEARCH_ALWAYS_LOADED: list[str] = [] class CompactChatForm(BaseModel): @@ -845,6 +851,7 @@ async def set_chat_config(form_data: ChatConfigForm, user=Depends(get_admin_user threshold = max(1, int(form_data.CONTEXT_COMPACTION_TOKEN_THRESHOLD)) token_cap = max(1, int(form_data.CONTEXT_COMPACTION_TOKEN_CAP or threshold)) retention_percentage = min(50, max(10, int(form_data.CONTEXT_COMPACTION_RETENTION_PERCENTAGE))) + tool_search_defer_threshold = max(0, int(form_data.TOOL_SEARCH_DEFER_THRESHOLD)) await Config.upsert( chat_config_updates( { @@ -853,6 +860,7 @@ async def set_chat_config(form_data: ChatConfigForm, user=Depends(get_admin_user 'CONTEXT_COMPACTION_TOKEN_THRESHOLD': threshold, 'CONTEXT_COMPACTION_TOKEN_CAP': token_cap, 'CONTEXT_COMPACTION_RETENTION_PERCENTAGE': retention_percentage, + 'TOOL_SEARCH_DEFER_THRESHOLD': tool_search_defer_threshold, } ) ) diff --git a/backend/open_webui/tools/builtin.py b/backend/open_webui/tools/builtin.py index c974c68b46..3c7be973ca 100644 --- a/backend/open_webui/tools/builtin.py +++ b/backend/open_webui/tools/builtin.py @@ -3548,6 +3548,45 @@ async def view_skill( return JSONCodec.dumps({'error': str(e)}) +# ============================================================================= +# TOOL SEARCH +# ============================================================================= + + +async def search_tools( + query: str, + limit: int = 5, + __metadata__: dict = None, +) -> str: + """ + Search the tools listed in and load the matches so you can call them in your next step. + Pass the exact tool name when you already know it. + + :param query: Keywords describing the capability you need (e.g. "jira create issue"), or an exact tool name + :param limit: Maximum number of tools to load (default 5, max 20) + :return: JSON with the names of the loaded tools + """ + from open_webui.utils.tool_search import mark_tools_loaded, search_deferred_tools + + metadata = __metadata__ or {} + state = metadata.get('tool_search') + if not state: + return JSONCodec.dumps({'error': 'Tool search is not active for this request'}) + + try: + tools = metadata['tools'] + matches = search_deferred_tools(query, {name: tools[name]['spec'] for name in state['deferred']}, limit) + mark_tools_loaded(metadata, matches) + if not matches: + return JSONCodec.dumps( + {'loaded': [], 'message': 'No matching tools found. Try different keywords or the exact tool name.'} + ) + return JSONCodec.dumps({'loaded': matches}) + except Exception as e: + log.exception(f'search_tools error: {e}') + return JSONCodec.dumps({'error': str(e)}) + + # ============================================================================= # TASK MANAGEMENT TOOLS # ============================================================================= diff --git a/backend/open_webui/utils/middleware.py b/backend/open_webui/utils/middleware.py index 50d8617298..83f853be78 100644 --- a/backend/open_webui/utils/middleware.py +++ b/backend/open_webui/utils/middleware.py @@ -142,6 +142,15 @@ from open_webui.utils.task import ( rag_template, tools_function_calling_generation_template, ) +from open_webui.utils.tool_search import ( + SEARCH_TOOL_NAME, + build_deferred_tools_manifest, + collect_called_tool_names, + get_tool_search_config, + mark_tools_loaded, + rebuild_tools_payload, + select_deferred_tools, +) from open_webui.utils.tools import ( build_tool_server_headers, get_attached_knowledge, @@ -2921,6 +2930,8 @@ async def process_chat_payload(request, form_data, user, metadata, model): log.debug('tool_ids=%r', tool_ids) log.debug('direct_tool_servers=%r', direct_tool_servers) + tool_search_config = await get_tool_search_config() + tools_dict = {} mcp_clients = {} @@ -3093,6 +3104,8 @@ async def process_chat_payload(request, form_data, user, metadata, model): **extra_params, '__event_emitter__': event_emitter, '__skill_ids__': view_skill_ids, + '__tool_search__': tool_search_config['enable'] + and metadata.get('params', {}).get('function_calling') != 'legacy', }, features, model, @@ -3152,12 +3165,33 @@ async def process_chat_payload(request, form_data, user, metadata, model): metadata['tools'] = tools_dict if metadata.get('params', {}).get('function_calling') != 'legacy': + # Tool search: withhold large schemas from the provider and let the + # model load them on demand via the search_tools builtin. + if tools_dict.get(SEARCH_TOOL_NAME, {}).get('type') == 'builtin': + deferred = select_deferred_tools(tools_dict, tool_search_config) + if deferred: + metadata['tool_search'] = { + 'deferred': deferred, + # Tools called earlier in this conversation stay loaded. + 'loaded': sorted(set(deferred) & collect_called_tool_names(form_data['messages'])), + 'extra_tools': inlet_filter_tools or [], + } + form_data['messages'] = add_or_update_system_message( + build_deferred_tools_manifest(tools_dict, deferred), + form_data['messages'], + append=True, + ) + else: + # Nothing to defer, so behave as if tool search were off. + tools_dict.pop(SEARCH_TOOL_NAME) + # If the function calling is native, then call the tools function calling handler form_data['tools'] = [ {'type': 'function', 'function': tool.get('spec', {})} for tool in tools_dict.values() ] if inlet_filter_tools: form_data['tools'].extend(inlet_filter_tools) + rebuild_tools_payload(form_data, metadata) else: # If the function calling is not native, then call the tools function calling handler try: @@ -3468,6 +3502,11 @@ async def drain_approved_tool_calls(request, form_data, user, model, metadata) - ) changed = True + if changed and metadata.get('tool_search'): + # Approved deferred tools must reach the provider on the next request. + mark_tools_loaded(metadata, [item.get('name', '') for item in approved_calls]) + rebuild_tools_payload(form_data, metadata) + if changed: result_call_ids = { item.get('call_id') for item in output if item.get('type') == 'function_call_output' and item.get('call_id') @@ -6034,6 +6073,13 @@ async def streaming_chat_response_handler(response, ctx): ) ) + # A deferred tool called straight from the manifest still runs (it is in + # metadata['tools']); mark it loaded so its schema is sent next iteration. + mark_tools_loaded( + metadata, + [tool_call.get('function', {}).get('name', '') for tool_call in response_tool_calls], + ) + for tool_call in response_tool_calls: tool_call_id = tool_call.get('id', '') tool_function_name = tool_call.get('function', {}).get('name', '') @@ -6222,6 +6268,8 @@ async def streaming_chat_response_handler(response, ctx): 'stream': True, 'metadata': metadata, } + # Tool search: include tools loaded during this iteration. + rebuild_tools_payload(new_form_data, metadata) if ENABLE_RESPONSES_API_STATEFUL and last_response_id: system_message = get_system_message(form_data['messages']) diff --git a/backend/open_webui/utils/tool_search.py b/backend/open_webui/utils/tool_search.py new file mode 100644 index 0000000000..ddb60bad01 --- /dev/null +++ b/backend/open_webui/utils/tool_search.py @@ -0,0 +1,162 @@ +""" +Tool search / lazy tool loading. + +Large tool schemas are withheld from the provider `tools` array and listed in a +compact `` manifest instead. The model loads them on demand +through the `search_tools` builtin. Per-request state lives on `metadata`: + + metadata['tool_search'] = { + 'deferred': [tool names withheld from the provider], + 'loaded': [deferred names that have since been loaded], + 'extra_tools': [tools added by filter inlets, always sent], + } +""" + +import fnmatch +import re + +from rank_bm25 import BM25Okapi + +from open_webui.models.config import Config +from open_webui.utils.json_codec import JSONCodec + +SEARCH_TOOL_NAME = 'search_tools' +MANIFEST_DESCRIPTION_MAX_CHARS = 100 +MAX_SEARCH_LIMIT = 20 +DEFAULT_SEARCH_LIMIT = 5 + + +async def get_tool_search_config() -> dict: + values = await Config.get_many( + 'chat.tool_search.enable', + 'chat.tool_search.defer_threshold', + 'chat.tool_search.always_loaded', + ) + return { + 'enable': bool(values.get('chat.tool_search.enable', False)), + 'defer_threshold': _to_int(values.get('chat.tool_search.defer_threshold'), 400), + 'always_loaded': values.get('chat.tool_search.always_loaded') or [], + } + + +def _to_int(value, default: int) -> int: + try: + return int(value) + except (TypeError, ValueError): + return default + + +def select_deferred_tools(tools_dict: dict[str, dict], config: dict) -> list[str]: + """Sorted names of non-builtin tools whose schema exceeds the threshold and is not always loaded.""" + patterns = config['always_loaded'] + return sorted( + name + for name, tool in tools_dict.items() + if tool.get('type') != 'builtin' + and not any(fnmatch.fnmatchcase(name, pattern) for pattern in patterns) + and len(JSONCodec.dumps(tool.get('spec') or {})) > config['defer_threshold'] + ) + + +def _truncate_description(description: str | None) -> str: + text = ' '.join(str(description or '').split()) + if len(text) > MANIFEST_DESCRIPTION_MAX_CHARS: + text = text[: MANIFEST_DESCRIPTION_MAX_CHARS - 1].rstrip() + '…' + return text + + +def build_deferred_tools_manifest(tools_dict: dict[str, dict], deferred: list[str]) -> str: + """System prompt block listing every deferred tool (mirrors ). + + Loaded tools stay listed so the block is identical across turns and provider prompt caches hold. + """ + entries = '' + for name in deferred: + description = _truncate_description(tools_dict[name].get('spec', {}).get('description')) + entries += f'\n{name}\n{description}\n\n' + + return ( + '\n' + 'The following tools are available but their definitions are not loaded. ' + f'To use one, call `{SEARCH_TOOL_NAME}` with a short keyword query or the exact tool name; ' + 'matching tools become callable immediately afterwards. ' + 'Never tell the user a capability is unavailable without searching first.\n' + f'{entries}' + ) + + +def build_tools_payload(tools_dict: dict[str, dict], deferred, loaded, extra: list | None = None) -> list[dict]: + """OpenAI `tools` array: every tool that is not deferred, plus deferred tools that are loaded, plus extras.""" + deferred_set = set(deferred) - set(loaded) + tools = [ + {'type': 'function', 'function': tool.get('spec', {})} + for name, tool in tools_dict.items() + if name not in deferred_set + ] + if extra: + tools.extend(extra) + return tools + + +_SPLIT_RE = re.compile(r'[^0-9a-zA-Z]+') +_CAMEL_RE = re.compile(r'(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])') + + +def tokenize(text: str | None) -> list[str]: + """Lowercase tokens split on non-alphanumerics and camelCase boundaries.""" + return [part.lower() for chunk in _SPLIT_RE.split(text or '') for part in _CAMEL_RE.split(chunk) if part] + + +def _document_text(name: str, spec: dict) -> str: + properties = (spec.get('parameters') or {}).get('properties') or {} + return ' '.join([name, str(spec.get('description') or ''), *properties.keys()]) + + +def search_deferred_tools(query: str, candidates: dict[str, dict], limit: int = DEFAULT_SEARCH_LIMIT) -> list[str]: + """Rank candidate tools (name -> spec) against the query; an exact tool name wins outright.""" + query = (query or '').strip() + if query in candidates: + return [query] + + query_tokens = tokenize(query) + if not candidates or not query_tokens: + return [] + + names = list(candidates) + documents = [tokenize(_document_text(name, candidates[name])) or ['_'] for name in names] + # BM25 IDF is zero for a term found in half the corpus, common with small tool sets, + # so raw token overlap keeps every matching term rankable. + bm25 = BM25Okapi(documents).get_scores(query_tokens) + scores = { + name: max(0.0, float(bm25[index])) + len(set(query_tokens) & set(documents[index])) + for index, name in enumerate(names) + } + + ranked = sorted((name for name in names if scores[name] > 0), key=lambda name: (-scores[name], name)) + return ranked[: max(1, min(_to_int(limit, DEFAULT_SEARCH_LIMIT), MAX_SEARCH_LIMIT))] + + +def collect_called_tool_names(messages: list[dict]) -> set[str]: + """Tools the conversation already called, so they stay loaded on later turns.""" + return { + (tool_call.get('function') or {}).get('name') + for message in messages + if message.get('role') == 'assistant' + for tool_call in message.get('tool_calls') or [] + } + + +def mark_tools_loaded(metadata: dict, names: list[str]) -> None: + """Record deferred tools as loaded on metadata['tool_search']; no-op when tool search is inactive.""" + state = metadata.get('tool_search') + if state: + state['loaded'] = sorted(set(state['loaded']) | (set(names) & set(state['deferred']))) + + +def rebuild_tools_payload(form_data: dict, metadata: dict) -> None: + """Refresh form_data['tools'] from metadata when tool search is active; no-op otherwise.""" + state = metadata.get('tool_search') + if state: + form_data['tools'] = build_tools_payload( + metadata['tools'], state['deferred'], state['loaded'], state['extra_tools'] + ) diff --git a/backend/open_webui/utils/tools.py b/backend/open_webui/utils/tools.py index 120bbb39b4..69061101fe 100644 --- a/backend/open_webui/utils/tools.py +++ b/backend/open_webui/utils/tools.py @@ -84,6 +84,7 @@ from open_webui.tools.builtin import ( search_knowledge_files, search_memories, search_notes, + search_tools, search_web, timer, toggle_automation, @@ -736,6 +737,10 @@ async def get_builtin_tools( if extra_params.get('__skill_ids__'): builtin_functions.append(view_skill) + # Tool search - lets the model load deferred (large-schema) tools on demand + if extra_params.get('__tool_search__'): + builtin_functions.append(search_tools) + # Task management - break down complex work into trackable steps # Task state is stored on the chats row; local/channel IDs do not have one. if is_builtin_tool_enabled('tasks') and is_saved_chat_id(metadata.get('chat_id')): diff --git a/src/lib/components/admin/Settings/Interface.svelte b/src/lib/components/admin/Settings/Interface.svelte index cd68640c6e..3666a45b86 100644 --- a/src/lib/components/admin/Settings/Interface.svelte +++ b/src/lib/components/admin/Settings/Interface.svelte @@ -50,7 +50,10 @@ CONTEXT_COMPACTION_TOKEN_CAP: 80000, CONTEXT_COMPACTION_RETENTION_PERCENTAGE: 40, CONTEXT_COMPACTION_PROMPT_TEMPLATE: '', - ENABLE_TOOL_PERMISSIONS: false + ENABLE_TOOL_PERMISSIONS: false, + ENABLE_TOOL_SEARCH: false, + TOOL_SEARCH_DEFER_THRESHOLD: 400, + TOOL_SEARCH_ALWAYS_LOADED: '' }; let showTaskParameters = false; @@ -69,8 +72,14 @@ [taskConfig, chatConfig] = await Promise.all([ updateTaskConfig(localStorage.token, taskConfigPayload), - updateChatConfig(localStorage.token, chatConfig) + updateChatConfig(localStorage.token, { + ...chatConfig, + TOOL_SEARCH_ALWAYS_LOADED: chatConfig.TOOL_SEARCH_ALWAYS_LOADED.split(',') + .map((item) => item.trim()) + .filter((item) => item !== '') + }) ]); + chatConfig.TOOL_SEARCH_ALWAYS_LOADED = (chatConfig.TOOL_SEARCH_ALWAYS_LOADED ?? []).join(', '); appConfig.update((current) => current ? { @@ -123,6 +132,9 @@ getChatConfig(localStorage.token) ]); taskConfig.TASK_MODEL_PARAMS = taskConfig.TASK_MODEL_PARAMS ?? {}; + chatConfig.TOOL_SEARCH_ALWAYS_LOADED = (chatConfig.TOOL_SEARCH_ALWAYS_LOADED ?? []).join( + ', ' + ); workspaceModels = await getBaseModels(localStorage.token); baseModels = await getModels(localStorage.token, null, false); @@ -362,6 +374,46 @@ {/if} + + +
+ {$i18n.t('settings.admin.interface.toolSearch.label')} + +
+ +
+ + {#if chatConfig.ENABLE_TOOL_SEARCH} + + + + + + + + {/if} diff --git a/src/lib/i18n/locales/en-US/translation.json b/src/lib/i18n/locales/en-US/translation.json index ee7dba23b5..0c60817d1b 100644 --- a/src/lib/i18n/locales/en-US/translation.json +++ b/src/lib/i18n/locales/en-US/translation.json @@ -2534,6 +2534,13 @@ "settings.admin.interface.tokenThreshold.label": "Token Threshold", "settings.admin.interface.toolPermissions.description": "Show Full access and Ask for approval in the chat input menu.", "settings.admin.interface.toolPermissions.label": "Tool Permissions", + "settings.admin.interface.toolSearch.description": "Withhold large tool definitions from the request and let the model load them on demand. Reduces prompt size when many tools are enabled.", + "settings.admin.interface.toolSearch.label": "Tool Search", + "settings.admin.interface.toolSearchAlwaysLoaded.description": "Comma-separated tool names or glob patterns that are never deferred.", + "settings.admin.interface.toolSearchAlwaysLoaded.label": "Always Loaded Tools", + "settings.admin.interface.toolSearchAlwaysLoaded.placeholder": "e.g. jira_create_issue, github_*", + "settings.admin.interface.toolSearchThreshold.description": "Tools whose definition is longer than this many characters are deferred.", + "settings.admin.interface.toolSearchThreshold.label": "Deferral Threshold (characters)", "settings.admin.interface.toolsFunctionCallingPrompt.description": "Guides how the assistant formats tool and function calls.", "settings.admin.interface.toolsFunctionCallingPrompt.label": "Tools Function Calling Prompt", "settings.admin.interface.voiceModePrompt.description": "Apply voice-specific instructions while voice mode is active.",