mirror of
https://github.com/open-webui/open-webui.git
synced 2026-10-10 03:27:57 +00:00
fix: tool search follow-up manifest, padded always-loaded patterns, Hindi tokens
- Strip the <available_tools> block when a pending sub-agent or timer report resumes the parent chat, so the follow-up request carries it once. - Trim always-loaded patterns saved through the chat config API, matching the settings page and the env var. - Tokenize on letters, digits and combining marks so Hindi words keep their vowel signs instead of splitting into single consonants. - Build the search_tools entry directly and leave the builtin tools loop untouched; drop the unreachable "not active" branch, config fallbacks and ranking try/except; import rank_bm25 where it is used.
This commit is contained in:
parent
28d34c39c5
commit
d79280cf48
5 changed files with 44 additions and 45 deletions
|
|
@ -854,6 +854,7 @@ async def set_chat_config(form_data: ChatConfigForm, user=Depends(get_admin_user
|
|||
token_cap = max(1, int(form_data.CONTEXT_COMPACTION_TOKEN_CAP or threshold))
|
||||
retention_percentage = min(50, max(10, int(form_data.CONTEXT_COMPACTION_RETENTION_PERCENTAGE)))
|
||||
tool_search_defer_threshold = max(0, int(form_data.TOOL_SEARCH_DEFER_THRESHOLD))
|
||||
tool_search_always_loaded = [item.strip() for item in form_data.TOOL_SEARCH_ALWAYS_LOADED if item.strip()]
|
||||
await Config.upsert(
|
||||
chat_config_updates(
|
||||
{
|
||||
|
|
@ -863,6 +864,7 @@ async def set_chat_config(form_data: ChatConfigForm, user=Depends(get_admin_user
|
|||
'CONTEXT_COMPACTION_TOKEN_CAP': token_cap,
|
||||
'CONTEXT_COMPACTION_RETENTION_PERCENTAGE': retention_percentage,
|
||||
'TOOL_SEARCH_DEFER_THRESHOLD': tool_search_defer_threshold,
|
||||
'TOOL_SEARCH_ALWAYS_LOADED': tool_search_always_loaded,
|
||||
}
|
||||
)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -3568,22 +3568,14 @@ async def search_tools(
|
|||
"""
|
||||
from open_webui.utils.tool_search import search_deferred_tools
|
||||
|
||||
metadata = __metadata__ or {}
|
||||
tools = metadata.get('tools') or {}
|
||||
deferred = [name for name in metadata.get('deferred_tools') or [] if name in tools]
|
||||
if not deferred:
|
||||
return JSONCodec.dumps({'error': 'Tool search is not active for this request'})
|
||||
|
||||
try:
|
||||
matches = search_deferred_tools(query, {name: tools[name]['spec'] for name in deferred}, count)
|
||||
if not matches:
|
||||
return JSONCodec.dumps(
|
||||
{'tools': [], 'message': 'No matching tools found. Try different keywords or the exact tool name.'}
|
||||
)
|
||||
return JSONCodec.dumps({'tools': [tools[name]['spec'] for name in matches]})
|
||||
except Exception as e:
|
||||
log.exception(f'search_tools error: {e}')
|
||||
return JSONCodec.dumps({'error': str(e)})
|
||||
tools = __metadata__['tools']
|
||||
candidates = {name: tools[name]['spec'] for name in __metadata__['deferred_tools']}
|
||||
matches = search_deferred_tools(query, candidates, count)
|
||||
if not matches:
|
||||
return JSONCodec.dumps(
|
||||
{'tools': [], 'message': 'No matching tools found. Try different keywords or the exact tool name.'}
|
||||
)
|
||||
return JSONCodec.dumps({'tools': [candidates[name] for name in matches]})
|
||||
|
||||
|
||||
# =============================================================================
|
||||
|
|
|
|||
|
|
@ -187,7 +187,7 @@ async def process_pending_internal_messages(
|
|||
|
||||
assistant_message_id = str(uuid4())
|
||||
message_list = get_message_list(messages, parent_id)
|
||||
system_prompt = run.get('system_prompt')
|
||||
system_prompt = strip_deferred_tools_manifest(run.get('system_prompt'))
|
||||
user_message = {
|
||||
'id': user_message_id,
|
||||
'parentId': parent_id,
|
||||
|
|
|
|||
|
|
@ -8,10 +8,10 @@ import fnmatch
|
|||
import re
|
||||
import unicodedata
|
||||
|
||||
import regex
|
||||
from open_webui.models.config import Config
|
||||
from open_webui.utils.json_codec import JSONCodec
|
||||
from open_webui.utils.misc import add_or_update_system_message
|
||||
from rank_bm25 import BM25Okapi
|
||||
|
||||
SEARCH_TOOL_NAME = 'search_tools'
|
||||
MANIFEST_DESCRIPTION_MAX_CHARS = 100
|
||||
|
|
@ -31,19 +31,12 @@ async def get_tool_search_config() -> dict | None:
|
|||
'chat.tool_search.defer_builtin_tools',
|
||||
)
|
||||
return {
|
||||
'defer_threshold': _to_int(values.get('chat.tool_search.defer_threshold'), 400),
|
||||
'always_loaded': values.get('chat.tool_search.always_loaded') or [],
|
||||
'defer_builtin_tools': values.get('chat.tool_search.defer_builtin_tools', True) is not False,
|
||||
'defer_threshold': values['chat.tool_search.defer_threshold'],
|
||||
'always_loaded': values['chat.tool_search.always_loaded'],
|
||||
'defer_builtin_tools': values['chat.tool_search.defer_builtin_tools'],
|
||||
}
|
||||
|
||||
|
||||
def _to_int(value, default: int) -> int:
|
||||
try:
|
||||
return int(value)
|
||||
except (TypeError, ValueError):
|
||||
return default
|
||||
|
||||
|
||||
def select_deferred_tools(tools_dict: dict[str, dict], config: dict) -> list[str]:
|
||||
"""Sorted names of tools whose schema exceeds the threshold and that are not always loaded."""
|
||||
patterns = config['always_loaded']
|
||||
|
|
@ -87,7 +80,11 @@ async def apply_tool_search(form_data: dict, metadata: dict, tools_dict: dict[st
|
|||
return set()
|
||||
|
||||
from open_webui.tools.builtin import search_tools
|
||||
from open_webui.utils.tools import get_builtin_tool
|
||||
from open_webui.utils.tools import (
|
||||
get_async_tool_function_and_apply_extra_params,
|
||||
get_builtin_function_introspection,
|
||||
get_builtin_tool_spec,
|
||||
)
|
||||
|
||||
metadata['deferred_tools'] = deferred
|
||||
form_data['messages'] = add_or_update_system_message(
|
||||
|
|
@ -95,7 +92,14 @@ async def apply_tool_search(form_data: dict, metadata: dict, tools_dict: dict[st
|
|||
form_data['messages'],
|
||||
append=True,
|
||||
)
|
||||
tools_dict[SEARCH_TOOL_NAME] = await get_builtin_tool(search_tools, {'__metadata__': metadata})
|
||||
tools_dict[SEARCH_TOOL_NAME] = {
|
||||
'tool_id': f'builtin:{SEARCH_TOOL_NAME}',
|
||||
'callable': await get_async_tool_function_and_apply_extra_params(
|
||||
search_tools, {'__metadata__': metadata}, get_builtin_function_introspection(search_tools)
|
||||
),
|
||||
'spec': get_builtin_tool_spec(search_tools),
|
||||
'type': 'builtin',
|
||||
}
|
||||
return set(deferred)
|
||||
|
||||
|
||||
|
|
@ -104,7 +108,8 @@ def strip_deferred_tools_manifest(system_prompt: str | None) -> str | None:
|
|||
return _MANIFEST_RE.sub('', system_prompt) if system_prompt else system_prompt
|
||||
|
||||
|
||||
_WORD_RE = re.compile(r'[^\W_]+')
|
||||
# Letters, digits and combining marks, so vowel signs in scripts like Devanagari stay attached to their word.
|
||||
_WORD_RE = regex.compile(r'[\p{L}\p{N}\p{M}]+')
|
||||
_CAMEL_RE = re.compile(r'(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])')
|
||||
# Scripts written without spaces between words: Han, Hiragana, Katakana and Hangul.
|
||||
_CJK_RE = re.compile(r'([\u3040-\u30ff\u3400-\u4dbf\u4e00-\u9fff\uf900-\ufaff\uac00-\ud7af]+)')
|
||||
|
|
@ -130,6 +135,8 @@ def _document_text(name: str, spec: dict) -> str:
|
|||
|
||||
def search_deferred_tools(query: str, candidates: dict[str, dict], count: int = DEFAULT_SEARCH_COUNT) -> list[str]:
|
||||
"""Rank candidate tools (name -> spec) against the query; an exact tool name wins outright."""
|
||||
from rank_bm25 import BM25Okapi
|
||||
|
||||
query = (query or '').strip()
|
||||
if query in candidates:
|
||||
return [query]
|
||||
|
|
@ -149,4 +156,4 @@ def search_deferred_tools(query: str, candidates: dict[str, dict], count: int =
|
|||
}
|
||||
|
||||
ranked = sorted((name for name in names if scores[name] > 0), key=lambda name: (-scores[name], name))
|
||||
return ranked[: max(1, min(_to_int(count, DEFAULT_SEARCH_COUNT), MAX_SEARCH_COUNT))]
|
||||
return ranked[: max(1, min(count, MAX_SEARCH_COUNT))]
|
||||
|
|
|
|||
|
|
@ -770,7 +770,7 @@ async def get_builtin_tools(
|
|||
builtin_functions = [func for func in builtin_functions if func.__name__ not in MUTATING_MEMORY_TOOLS]
|
||||
|
||||
for func in builtin_functions:
|
||||
tools_dict[func.__name__] = await get_builtin_tool(
|
||||
callable = await get_async_tool_function_and_apply_extra_params(
|
||||
func,
|
||||
{
|
||||
'__request__': request,
|
||||
|
|
@ -783,28 +783,26 @@ async def get_builtin_tools(
|
|||
'__message_id__': extra_params.get('__message_id__'),
|
||||
'__model_knowledge__': model_knowledge,
|
||||
},
|
||||
get_builtin_function_introspection(func),
|
||||
)
|
||||
|
||||
spec = get_builtin_tool_spec(func)
|
||||
if func.__name__ == 'delegate_task' and not config.get('subagents.background_enabled'):
|
||||
parameters = tools_dict[func.__name__]['spec'].get('parameters', {})
|
||||
parameters = spec.get('parameters', {})
|
||||
parameters.get('properties', {}).pop('background', None)
|
||||
if isinstance(parameters.get('required'), list):
|
||||
parameters['required'] = [name for name in parameters['required'] if name != 'background']
|
||||
|
||||
tools_dict[func.__name__] = {
|
||||
'tool_id': f'builtin:{func.__name__}',
|
||||
'callable': callable,
|
||||
'spec': spec,
|
||||
'type': 'builtin',
|
||||
}
|
||||
|
||||
return tools_dict
|
||||
|
||||
|
||||
async def get_builtin_tool(func, extra_params: dict) -> dict:
|
||||
return {
|
||||
'tool_id': f'builtin:{func.__name__}',
|
||||
'callable': await get_async_tool_function_and_apply_extra_params(
|
||||
func, extra_params, get_builtin_function_introspection(func)
|
||||
),
|
||||
'spec': get_builtin_tool_spec(func),
|
||||
'type': 'builtin',
|
||||
}
|
||||
|
||||
|
||||
def parse_description(docstring: str | None) -> str:
|
||||
"""
|
||||
Parse a function's docstring to extract the description.
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue