mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-10-06 02:48:22 +00:00
feat(agent): implement mock search tools and tool memory benchmark
- Added LLMMockSearchOp for simulating search operations with configurable complexity- Created SearchToolA, SearchToolB, and SearchToolC with distinct performance profiles- Implemented UseMockSearchOp for intelligent tool selection based on query analysis - Added test scripts and query datasets for evaluating tool memory effectiveness - Integrated tool memory service for storing and retrieving tool performance data - Created documentation for tool memory benchmark testing methodology - Refactored agent module structure and imports - Increased summary tool memory recent call count from20 to 30
This commit is contained in:
parent
40538f974a
commit
12f0b7d124
15 changed files with 1984 additions and 5 deletions
0
cookbook/tool_memory/__init__.py
Normal file
0
cookbook/tool_memory/__init__.py
Normal file
139
cookbook/tool_memory/query.json
Normal file
139
cookbook/tool_memory/query.json
Normal file
|
|
@ -0,0 +1,139 @@
|
|||
{
|
||||
"train": {
|
||||
"simple": [
|
||||
"法国的首都是什么?",
|
||||
"Python 是什么时候首次发布的?",
|
||||
"地球上有几个大洲?",
|
||||
"水的沸点是多少度?",
|
||||
"谁发明了电话?",
|
||||
"一年有多少天?",
|
||||
"光速是多少?",
|
||||
"氧气的化学符号是什么?",
|
||||
"世界上最高的山峰是哪座?",
|
||||
"太阳系有几颗行星?",
|
||||
"地球的半径是多少公里?",
|
||||
"元素周期表有多少个元素?",
|
||||
"蒙娜丽莎是谁画的?",
|
||||
"人类DNA有多少条染色体?",
|
||||
"HTTP的默认端口号是多少?",
|
||||
"世界上人口最多的国家是哪个?",
|
||||
"一公里等于多少米?",
|
||||
"谁提出了相对论?",
|
||||
"圆周率π的前两位小数是多少?",
|
||||
"中国的货币单位是什么?"
|
||||
],
|
||||
"moderate": [
|
||||
"列举Python 3.10的主要特性",
|
||||
"微服务架构有哪些好处?",
|
||||
"解释区块链技术的工作原理",
|
||||
"描述SQL和NoSQL数据库的主要区别",
|
||||
"软件工程中常见的设计模式有哪些?",
|
||||
"什么是RESTful API的设计原则?",
|
||||
"解释Docker容器化技术的优势",
|
||||
"云计算的三种服务模型是什么?",
|
||||
"什么是机器学习中的过拟合问题?",
|
||||
"解释Git中merge和rebase的区别",
|
||||
"什么是持续集成和持续部署(CI/CD)?",
|
||||
"描述OSI七层网络模型的作用",
|
||||
"什么是JWT令牌认证机制?",
|
||||
"解释虚拟内存的工作原理",
|
||||
"Kubernetes的主要组件有哪些?",
|
||||
"什么是函数式编程的核心概念?",
|
||||
"解释负载均衡的常见策略",
|
||||
"描述SOLID设计原则的含义",
|
||||
"什么是异步编程的优势和挑战?",
|
||||
"解释数据库索引的作用和类型"
|
||||
],
|
||||
"complex": [
|
||||
"比较凯恩斯主义和奥地利学派的经济政策",
|
||||
"分析可再生能源采用对环境的影响",
|
||||
"解释量子力学和广义相对论之间的关系",
|
||||
"讨论人工智能的历史演变过程",
|
||||
"评估不同机器学习算法在NLP中的有效性",
|
||||
"分析全球化对发展中国家经济的长期影响",
|
||||
"比较不同民主制度模型的优缺点",
|
||||
"探讨气候变化对生物多样性的影响机制",
|
||||
"研究基因编辑技术的伦理问题和社会影响",
|
||||
"分析区块链技术在金融领域的应用前景和挑战",
|
||||
"评估不同神经网络架构在计算机视觉中的表现",
|
||||
"探讨量子计算对现代密码学的威胁和机遇",
|
||||
"分析人口老龄化对社会保障体系的影响",
|
||||
"比较不同哲学流派对人工智能意识的观点",
|
||||
"研究微生物组与人类健康之间的复杂关系",
|
||||
"评估碳捕获技术在应对气候变化中的作用",
|
||||
"分析5G技术对物联网发展的推动作用",
|
||||
"探讨认知科学与人工智能的交叉研究领域",
|
||||
"比较不同经济体制下的创新能力差异",
|
||||
"研究纳米技术在医疗领域的应用和风险"
|
||||
]
|
||||
},
|
||||
"test": {
|
||||
"simple": [
|
||||
"日本的首都是什么?",
|
||||
"Java 语言是哪一年发布的?",
|
||||
"大西洋有多宽?",
|
||||
"铁的熔点是多少度?",
|
||||
"谁发明了汽车?",
|
||||
"一周有多少小时?",
|
||||
"声音在空气中的速度是多少?",
|
||||
"氢气的化学符号是什么?",
|
||||
"世界上最长的河流是哪条?",
|
||||
"月球绕地球一周需要多少天?",
|
||||
"标准大气压是多少帕?",
|
||||
"人体有多少块骨头?",
|
||||
"《星空》是谁画的?",
|
||||
"成年人有多少颗牙齿?",
|
||||
"HTTPS的默认端口号是多少?",
|
||||
"世界上面积最大的国家是哪个?",
|
||||
"一英里等于多少公里?",
|
||||
"谁发现了万有引力定律?",
|
||||
"黄金的化学元素符号是什么?",
|
||||
"美国的货币单位是什么?"
|
||||
],
|
||||
"moderate": [
|
||||
"列举TypeScript 5.0的主要新功能",
|
||||
"单体架构和微服务架构的区别是什么?",
|
||||
"解释分布式系统的CAP定理",
|
||||
"描述关系型数据库的ACID特性",
|
||||
"前端开发中常用的状态管理方案有哪些?",
|
||||
"什么是GraphQL的核心优势?",
|
||||
"解释Kubernetes容器编排的基本概念",
|
||||
"边缘计算和雾计算有什么区别?",
|
||||
"什么是神经网络中的梯度消失问题?",
|
||||
"解释Git中revert和reset的区别",
|
||||
"什么是DevOps的核心理念?",
|
||||
"描述TCP/IP协议栈的层次结构",
|
||||
"什么是OAuth 2.0授权框架?",
|
||||
"解释进程和线程的本质区别",
|
||||
"Service Mesh的主要功能是什么?",
|
||||
"什么是响应式编程的基本原理?",
|
||||
"解释缓存一致性的常见问题",
|
||||
"描述DDD领域驱动设计的核心思想",
|
||||
"什么是并发编程中的死锁问题?",
|
||||
"解释数据库事务隔离级别的分类"
|
||||
],
|
||||
"complex": [
|
||||
"比较货币主义和供给学派的经济理论差异",
|
||||
"分析核能发展对能源转型的作用",
|
||||
"解释弦理论与标准模型之间的联系",
|
||||
"讨论云计算技术的发展历程和趋势",
|
||||
"评估深度学习在计算机视觉中的应用效果",
|
||||
"分析数字化转型对传统产业的影响",
|
||||
"比较总统制和议会制的政治体制特点",
|
||||
"探讨海洋酸化对海洋生态系统的影响",
|
||||
"研究人工智能在医疗诊断中的伦理挑战",
|
||||
"分析去中心化金融(DeFi)的发展机遇与风险",
|
||||
"评估Transformer架构在自然语言处理中的优势",
|
||||
"探讨后量子密码学的研究方向和应用",
|
||||
"分析城市化进程对资源配置的影响",
|
||||
"比较实用主义和理想主义对技术伦理的看法",
|
||||
"研究肠道菌群与神经系统疾病的关联",
|
||||
"评估直接空气捕获技术的经济可行性",
|
||||
"分析边缘计算对云计算架构的影响",
|
||||
"探讨神经语言学与自然语言处理的交叉应用",
|
||||
"比较计划经济和市场经济的资源配置效率",
|
||||
"研究量子传感技术在精密测量中的突破"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
1063
docs/tool_memory/tool_bench.md
Normal file
1063
docs/tool_memory/tool_bench.md
Normal file
File diff suppressed because it is too large
Load diff
|
|
@ -1,6 +1,6 @@
|
|||
from reme_ai import react
|
||||
from reme_ai import retrieve
|
||||
from reme_ai import summary
|
||||
from reme_ai import vector_store
|
||||
from . import agent
|
||||
from . import retrieve
|
||||
from . import summary
|
||||
from . import vector_store
|
||||
|
||||
__version__ = "0.1.9"
|
||||
|
|
|
|||
2
reme_ai/agent/__init__.py
Normal file
2
reme_ai/agent/__init__.py
Normal file
|
|
@ -0,0 +1,2 @@
|
|||
from . import react
|
||||
from . import tools
|
||||
3
reme_ai/agent/tools/__init__.py
Normal file
3
reme_ai/agent/tools/__init__.py
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
from .llm_mock_search_op import LLMMockSearchOp
|
||||
from .mock_search_tools import SearchToolA, SearchToolB, SearchToolC
|
||||
from .use_mock_search_op import UseMockSearchOp
|
||||
297
reme_ai/agent/tools/llm_mock_search_op.py
Normal file
297
reme_ai/agent/tools/llm_mock_search_op.py
Normal file
|
|
@ -0,0 +1,297 @@
|
|||
import asyncio
|
||||
import json
|
||||
import random
|
||||
from typing import Dict, Any
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from flowllm.context import FlowContext, C
|
||||
from flowllm.enumeration.role import Role
|
||||
from flowllm.op.base_async_tool_op import BaseAsyncToolOp
|
||||
from flowllm.schema.message import Message
|
||||
from flowllm.schema.tool_call import ToolCall
|
||||
|
||||
|
||||
@C.register_op()
|
||||
class LLMMockSearchOp(BaseAsyncToolOp):
|
||||
"""
|
||||
Mock search operation that uses LLM to classify queries and simulate different scenarios.
|
||||
|
||||
Supports three query complexity levels:
|
||||
- simple: Simple factual queries with short, direct answers
|
||||
- medium: Medium complexity queries requiring balanced performance
|
||||
- complex: Complex research queries requiring comprehensive, in-depth results
|
||||
|
||||
Each scenario can be configured with:
|
||||
- success_rate: Probability of successful response (vs "Service busy" error)
|
||||
- extra_time: Extra sleep time in seconds to simulate latency
|
||||
- relevance_ratio: Probability of returning relevant results (vs random query results)
|
||||
"""
|
||||
file_path: str = __file__
|
||||
|
||||
def __init__(self,
|
||||
llm: str = "qwen3_30b_instruct",
|
||||
simple_config: Dict[str, Any] = None,
|
||||
medium_config: Dict[str, Any] = None,
|
||||
complex_config: Dict[str, Any] = None,
|
||||
**kwargs):
|
||||
"""
|
||||
Initialize the LLM Mock Search Op.
|
||||
|
||||
Args:
|
||||
llm: LLM model name to use for classification and content generation
|
||||
simple_config: Configuration for simple queries
|
||||
- success_rate: float (0-1), default 0.95
|
||||
- extra_time: float (seconds), default 0.5
|
||||
- relevance_ratio: float (0-1), default 0.98
|
||||
medium_config: Configuration for medium complexity queries
|
||||
- success_rate: float (0-1), default 0.85
|
||||
- extra_time: float (seconds), default 1.0
|
||||
- relevance_ratio: float (0-1), default 0.90
|
||||
complex_config: Configuration for complex queries
|
||||
- success_rate: float (0-1), default 0.70
|
||||
- extra_time: float (seconds), default 1.5
|
||||
- relevance_ratio: float (0-1), default 0.80
|
||||
"""
|
||||
super().__init__(llm=llm, **kwargs)
|
||||
|
||||
# Default configurations for each scenario
|
||||
self.simple_config = {
|
||||
"success_rate": 0.95,
|
||||
"extra_time": 0.5,
|
||||
"relevance_ratio": 0.98,
|
||||
"content_length": "short"
|
||||
}
|
||||
if simple_config:
|
||||
self.simple_config.update(simple_config)
|
||||
|
||||
self.medium_config = {
|
||||
"success_rate": 0.85,
|
||||
"extra_time": 1.0,
|
||||
"relevance_ratio": 0.90,
|
||||
"content_length": "medium"
|
||||
}
|
||||
if medium_config:
|
||||
self.medium_config.update(medium_config)
|
||||
|
||||
self.complex_config = {
|
||||
"success_rate": 0.70,
|
||||
"extra_time": 1.5,
|
||||
"relevance_ratio": 0.80,
|
||||
"content_length": "long"
|
||||
}
|
||||
if complex_config:
|
||||
self.complex_config.update(complex_config)
|
||||
|
||||
def build_tool_call(self) -> ToolCall:
|
||||
return ToolCall(**{
|
||||
"description": "Use search keywords to retrieve relevant information from the internet.",
|
||||
"input_schema": {
|
||||
"query": {
|
||||
"type": "string",
|
||||
"description": "search keyword or query",
|
||||
"required": True
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
async def classify_query(self, query: str) -> str:
|
||||
"""
|
||||
Classify the query into simple, medium, or complex using LLM.
|
||||
|
||||
Args:
|
||||
query: The search query to classify
|
||||
|
||||
Returns:
|
||||
Classification result: "simple", "medium", or "complex"
|
||||
"""
|
||||
classification_prompt = self.prompt_format(
|
||||
prompt_name="classification_prompt",
|
||||
query=query
|
||||
)
|
||||
|
||||
messages = [Message(role=Role.USER, content=classification_prompt)]
|
||||
|
||||
response = await self.llm.achat(messages=messages)
|
||||
classification = response.content.strip().lower()
|
||||
|
||||
# Extract classification from response
|
||||
if "simple" in classification:
|
||||
return "simple"
|
||||
elif "complex" in classification:
|
||||
return "complex"
|
||||
else:
|
||||
return "medium"
|
||||
|
||||
async def generate_search_result(self, query: str, complexity: str, config: Dict[str, Any]) -> str:
|
||||
"""
|
||||
Generate mock search results using LLM based on query complexity.
|
||||
|
||||
Args:
|
||||
query: The search query
|
||||
complexity: Query complexity level
|
||||
config: Configuration for this complexity level
|
||||
|
||||
Returns:
|
||||
Generated search result content
|
||||
"""
|
||||
content_length = config["content_length"]
|
||||
|
||||
generation_prompt = self.prompt_format(
|
||||
prompt_name="generation_prompt",
|
||||
query=query,
|
||||
complexity=complexity,
|
||||
content_length=content_length
|
||||
)
|
||||
|
||||
messages = [Message(role=Role.USER, content=generation_prompt)]
|
||||
|
||||
response = await self.llm.achat(messages=messages)
|
||||
return response.content
|
||||
|
||||
async def generate_random_result(self) -> str:
|
||||
"""
|
||||
Generate a random/irrelevant search result to simulate low relevance.
|
||||
|
||||
Returns:
|
||||
Random search result content
|
||||
"""
|
||||
random_topics = [
|
||||
"the history of ancient civilizations",
|
||||
"modern technology trends",
|
||||
"climate change impacts",
|
||||
"space exploration achievements",
|
||||
"culinary traditions around the world",
|
||||
"evolution of music genres",
|
||||
"breakthroughs in medical science",
|
||||
"architectural wonders",
|
||||
"wildlife conservation efforts",
|
||||
"developments in artificial intelligence"
|
||||
]
|
||||
|
||||
random_query = random.choice(random_topics)
|
||||
generation_prompt = self.prompt_format(
|
||||
prompt_name="generation_prompt",
|
||||
query=random_query,
|
||||
complexity="simple",
|
||||
content_length="short"
|
||||
)
|
||||
|
||||
messages = [Message(role=Role.USER, content=generation_prompt)]
|
||||
response = await self.llm.achat(messages=messages)
|
||||
|
||||
return f"[Low Relevance Result]\n{response.content}"
|
||||
|
||||
async def async_execute(self):
|
||||
query: str = self.input_dict["query"]
|
||||
logger.info(f"LLMMockSearchOp processing query: {query}")
|
||||
|
||||
# Step 1: Classify the query
|
||||
complexity = await self.classify_query(query)
|
||||
logger.info(f"Query classified as: {complexity}")
|
||||
|
||||
# Step 2: Get configuration for this complexity
|
||||
if complexity == "simple":
|
||||
config = self.simple_config
|
||||
elif complexity == "medium":
|
||||
config = self.medium_config
|
||||
else: # complex
|
||||
config = self.complex_config
|
||||
|
||||
logger.info(f"Using config: {config}")
|
||||
|
||||
# Step 3: Simulate extra time delay
|
||||
extra_time = config["extra_time"]
|
||||
await asyncio.sleep(extra_time)
|
||||
logger.info(f"Simulated extra delay: {extra_time:.2f}s")
|
||||
|
||||
# Step 4: Check success rate
|
||||
if random.random() > config["success_rate"]:
|
||||
error_message = "Search service is currently busy. Please try again later."
|
||||
logger.warning(f"Simulated failure: {error_message}")
|
||||
result_dict = {
|
||||
"success": False,
|
||||
"content": error_message,
|
||||
"query": query,
|
||||
"complexity": complexity
|
||||
}
|
||||
self.set_result(json.dumps(result_dict, ensure_ascii=False))
|
||||
return
|
||||
|
||||
# Step 5: Check relevance ratio
|
||||
if random.random() > config["relevance_ratio"]:
|
||||
# Generate random/irrelevant result
|
||||
logger.info("Generating low relevance result")
|
||||
content = await self.generate_random_result()
|
||||
result_dict = {
|
||||
"success": False,
|
||||
"content": content,
|
||||
"query": query,
|
||||
"complexity": complexity
|
||||
}
|
||||
else:
|
||||
# Generate relevant result
|
||||
logger.info("Generating relevant result")
|
||||
content = await self.generate_search_result(query, complexity, config)
|
||||
result_dict = {
|
||||
"success": True,
|
||||
"content": content,
|
||||
"query": query,
|
||||
"complexity": complexity
|
||||
}
|
||||
|
||||
self.set_result(json.dumps(result_dict, ensure_ascii=False))
|
||||
|
||||
|
||||
async def async_main():
|
||||
from flowllm.app import FlowLLMApp
|
||||
|
||||
async with FlowLLMApp(load_default_config=True):
|
||||
# Test with different query types
|
||||
test_queries = [
|
||||
"What is the capital of France?", # Simple
|
||||
"How does quantum computing work?", # Medium
|
||||
"Analyze the impact of artificial intelligence on global economy, employment, and society", # Complex
|
||||
]
|
||||
|
||||
# Custom configurations for testing
|
||||
custom_simple = {
|
||||
"success_rate": 1,
|
||||
"extra_time": 0,
|
||||
"relevance_ratio": 1
|
||||
}
|
||||
|
||||
custom_medium = {
|
||||
"success_rate": 1,
|
||||
"extra_time": 0,
|
||||
"relevance_ratio": 1
|
||||
}
|
||||
|
||||
custom_complex = {
|
||||
"success_rate": 1,
|
||||
"extra_time": 0,
|
||||
"relevance_ratio": 1
|
||||
}
|
||||
|
||||
op = LLMMockSearchOp(
|
||||
simple_config=custom_simple,
|
||||
medium_config=custom_medium,
|
||||
complex_config=custom_complex
|
||||
)
|
||||
|
||||
for query in test_queries:
|
||||
print(f"\n{'=' * 80}")
|
||||
print(f"Testing query: {query}")
|
||||
print(f"{'=' * 80}")
|
||||
|
||||
context = FlowContext(query=query)
|
||||
await op.async_call(context=context)
|
||||
result = json.loads(context.llm_mock_search_result)
|
||||
print(f"Success: {result['success']}")
|
||||
print(f"Query: {result['query']}")
|
||||
print(f"Complexity: {result['complexity']}")
|
||||
print(f"Content:\n{result['content']}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(async_main())
|
||||
76
reme_ai/agent/tools/llm_mock_search_prompt.yaml
Normal file
76
reme_ai/agent/tools/llm_mock_search_prompt.yaml
Normal file
|
|
@ -0,0 +1,76 @@
|
|||
classification_prompt: |
|
||||
You are a query complexity classifier. Analyze the following search query and classify it into one of three categories:
|
||||
|
||||
1. **simple** - Simple factual queries that:
|
||||
- Ask for a single, direct fact
|
||||
- Have a clear, unambiguous answer
|
||||
- Require minimal context or explanation
|
||||
- Examples: "What is the capital of France?", "Who invented the telephone?", "When did World War 2 end?"
|
||||
|
||||
2. **medium** - Medium complexity queries that:
|
||||
- Require some explanation or context
|
||||
- May involve multiple related facts
|
||||
- Need balanced depth without being exhaustive
|
||||
- Examples: "How does photosynthesis work?", "What are the main causes of climate change?", "Explain blockchain technology"
|
||||
|
||||
3. **complex** - Complex research queries that:
|
||||
- Require comprehensive, multi-dimensional analysis
|
||||
- Involve multiple subtopics or perspectives
|
||||
- Need in-depth exploration and connections
|
||||
- Examples: "Analyze the impact of AI on the global economy", "Compare different renewable energy solutions", "What are the geopolitical implications of space exploration?"
|
||||
|
||||
Query to classify: {query}
|
||||
|
||||
Respond with ONLY one word: simple, medium, or complex.
|
||||
|
||||
classification_prompt_zh: |
|
||||
你是一个查询复杂度分类器。分析以下搜索查询并将其分类为以下三类之一:
|
||||
|
||||
1. **simple** - 简单事实查询:
|
||||
- 询问单一、直接的事实
|
||||
- 有明确、无歧义的答案
|
||||
- 需要最少的上下文或解释
|
||||
- 示例:"法国的首都是什么?"、"谁发明了电话?"、"第二次世界大战何时结束?"
|
||||
|
||||
2. **medium** - 中等复杂度查询:
|
||||
- 需要一些解释或上下文
|
||||
- 可能涉及多个相关事实
|
||||
- 需要平衡的深度但不需要详尽无遗
|
||||
- 示例:"光合作用如何工作?"、"气候变化的主要原因是什么?"、"解释区块链技术"
|
||||
|
||||
3. **complex** - 复杂研究查询:
|
||||
- 需要全面、多维度的分析
|
||||
- 涉及多个子主题或观点
|
||||
- 需要深入探索和联系
|
||||
- 示例:"分析人工智能对全球经济的影响"、"比较不同的可再生能源解决方案"、"太空探索的地缘政治影响是什么?"
|
||||
|
||||
要分类的查询:{query}
|
||||
|
||||
只用一个词回答:simple、medium 或 complex。
|
||||
|
||||
generation_prompt: |
|
||||
You are a search engine generating mock search results. Generate a {content_length} response for the following query.
|
||||
|
||||
Query: {query}
|
||||
Complexity Level: {complexity}
|
||||
|
||||
Instructions based on content length:
|
||||
- **short**: Provide a concise answer in 1-3 sentences. Be direct and factual.
|
||||
- **medium**: Provide a balanced answer in 2-4 paragraphs. Include key details and some context.
|
||||
- **long**: Provide a comprehensive answer in 4-6 paragraphs. Include multiple perspectives, detailed explanations, and relevant context.
|
||||
|
||||
Generate the search result content now:
|
||||
|
||||
generation_prompt_zh: |
|
||||
你是一个搜索引擎,正在生成模拟搜索结果。为以下查询生成一个 {content_length} 的响应。
|
||||
|
||||
查询:{query}
|
||||
复杂度级别:{complexity}
|
||||
|
||||
根据内容长度的指示:
|
||||
- **short**(短):提供 1-3 句话的简洁答案。要直接和事实性。
|
||||
- **medium**(中):提供 2-4 段的平衡答案。包括关键细节和一些上下文。
|
||||
- **long**(长):提供 4-6 段的全面答案。包括多个角度、详细解释和相关上下文。
|
||||
|
||||
现在生成搜索结果内容:
|
||||
|
||||
115
reme_ai/agent/tools/mock_search_tools.py
Normal file
115
reme_ai/agent/tools/mock_search_tools.py
Normal file
|
|
@ -0,0 +1,115 @@
|
|||
from flowllm.context import C
|
||||
from flowllm.schema.tool_call import ToolCall
|
||||
|
||||
from reme_ai.agent.tools.llm_mock_search_op import LLMMockSearchOp
|
||||
|
||||
|
||||
@C.register_op()
|
||||
class SearchToolA(LLMMockSearchOp):
|
||||
def __init__(self, llm: str = "qwen3_30b_instruct", **kwargs):
|
||||
# Configure for fast but shallow performance
|
||||
simple_config = {
|
||||
"success_rate": 0.95, # High success rate for simple queries
|
||||
"extra_time": 0.2, # Very fast (0.2-0.5s range)
|
||||
"relevance_ratio": 0.92, # High relevance
|
||||
"content_length": "short" # Concise answers
|
||||
}
|
||||
|
||||
medium_config = {
|
||||
"success_rate": 0.75, # Lower success for medium queries
|
||||
"extra_time": 0.3, # Still fast
|
||||
"relevance_ratio": 0.70, # Moderate relevance
|
||||
"content_length": "short" # Limited depth
|
||||
}
|
||||
|
||||
complex_config = {
|
||||
"success_rate": 0.50, # Poor success rate for complex queries
|
||||
"extra_time": 0.4, # Fast but insufficient
|
||||
"relevance_ratio": 0.50, # Low relevance (often misses key aspects)
|
||||
"content_length": "short" # Too shallow for complex topics
|
||||
}
|
||||
|
||||
super().__init__(llm=llm,
|
||||
simple_config=simple_config,
|
||||
medium_config=medium_config,
|
||||
complex_config=complex_config,
|
||||
**kwargs)
|
||||
|
||||
def build_tool_call(self) -> ToolCall:
|
||||
tool_call = super().build_tool_call()
|
||||
tool_call.description += " Best suited for simple queries."
|
||||
return tool_call
|
||||
|
||||
@C.register_op()
|
||||
class SearchToolB(LLMMockSearchOp):
|
||||
def __init__(self, llm: str = "qwen3_30b_instruct", **kwargs):
|
||||
# Configure for balanced performance
|
||||
simple_config = {
|
||||
"success_rate": 0.8, # Very high success rate
|
||||
"extra_time": 0.8, # Moderate speed (1.0-1.5s range)
|
||||
"relevance_ratio": 0.8, # High relevance
|
||||
"content_length": "medium" # More detailed than needed for simple
|
||||
}
|
||||
|
||||
medium_config = {
|
||||
"success_rate": 0.8, # Excellent success rate
|
||||
"extra_time": 1.0, # Balanced speed
|
||||
"relevance_ratio": 0.8, # High relevance
|
||||
"content_length": "medium" # Perfect depth for medium queries
|
||||
}
|
||||
|
||||
complex_config = {
|
||||
"success_rate": 0.8, # Good success rate
|
||||
"extra_time": 1.2, # Still reasonable speed
|
||||
"relevance_ratio": 0.8, # Decent relevance but not exhaustive
|
||||
"content_length": "medium" # Covers main points but lacks depth
|
||||
}
|
||||
|
||||
super().__init__(llm=llm,
|
||||
simple_config=simple_config,
|
||||
medium_config=medium_config,
|
||||
complex_config=complex_config,
|
||||
**kwargs)
|
||||
|
||||
def build_tool_call(self) -> ToolCall:
|
||||
tool_call = super().build_tool_call()
|
||||
tool_call.description += " Best suited for medium complexity queries."
|
||||
return tool_call
|
||||
|
||||
|
||||
@C.register_op()
|
||||
class SearchToolC(LLMMockSearchOp):
|
||||
|
||||
def __init__(self, llm: str = "qwen3_30b_instruct", **kwargs):
|
||||
# Configure for comprehensive but costly performance
|
||||
simple_config = {
|
||||
"success_rate": 0.7, # Good but not optimal (over-processing)
|
||||
"extra_time": 2.5, # Slow (3.0-4.0s range)
|
||||
"relevance_ratio": 0.7, # High relevance but unnecessary depth
|
||||
"content_length": "long" # Too detailed for simple queries
|
||||
}
|
||||
|
||||
medium_config = {
|
||||
"success_rate": 0.7, # High success rate
|
||||
"extra_time": 2.8, # Slow but thorough
|
||||
"relevance_ratio": 0.7, # High relevance with extra context
|
||||
"content_length": "long" # More depth than needed
|
||||
}
|
||||
|
||||
complex_config = {
|
||||
"success_rate": 0.95, # Excellent success rate
|
||||
"extra_time": 3.5, # Slow but comprehensive (3.5-5.0s range)
|
||||
"relevance_ratio": 0.94, # Very high relevance
|
||||
"content_length": "long" # Perfect depth for complex queries
|
||||
}
|
||||
|
||||
super().__init__(llm=llm,
|
||||
simple_config=simple_config,
|
||||
medium_config=medium_config,
|
||||
complex_config=complex_config,
|
||||
**kwargs)
|
||||
|
||||
def build_tool_call(self) -> ToolCall:
|
||||
tool_call = super().build_tool_call()
|
||||
tool_call.description += " Best suited for complex queries."
|
||||
return tool_call
|
||||
132
reme_ai/agent/tools/test_use_mock_search.py
Normal file
132
reme_ai/agent/tools/test_use_mock_search.py
Normal file
|
|
@ -0,0 +1,132 @@
|
|||
"""
|
||||
Test script for UseMockSearchOp
|
||||
|
||||
This script demonstrates how to use the UseMockSearchOp to intelligently
|
||||
select and execute search tools based on query complexity.
|
||||
"""
|
||||
import asyncio
|
||||
import json
|
||||
|
||||
from flowllm.context import FlowContext
|
||||
from flowllm.app import FlowLLMApp
|
||||
|
||||
from reme_ai.agent.tools.use_mock_search_op import UseMockSearchOp
|
||||
|
||||
|
||||
async def test_single_query(op: UseMockSearchOp, query: str):
|
||||
"""Test a single query and display results."""
|
||||
print(f"\n{'=' * 100}")
|
||||
print(f"Testing Query: {query}")
|
||||
print(f"{'=' * 100}")
|
||||
|
||||
context = FlowContext(query=query)
|
||||
await op.async_call(context=context)
|
||||
|
||||
result = json.loads(context.use_mock_search_result)
|
||||
|
||||
print(f"\n📊 Results:")
|
||||
print(f" Selected Tool: {result['selected_tool']}")
|
||||
print(f" Reasoning: {result['reasoning']}")
|
||||
print(f" Query Complexity: {result['complexity']}")
|
||||
print(f" Success: {result['success']}")
|
||||
print(f"\n📝 Content:")
|
||||
print(f" {result['content'][:500]}..." if len(result['content']) > 500 else f" {result['content']}")
|
||||
print(f"\n{'=' * 100}\n")
|
||||
|
||||
return result
|
||||
|
||||
|
||||
async def main():
|
||||
"""Run comprehensive tests of UseMockSearchOp."""
|
||||
|
||||
print("\n" + "=" * 100)
|
||||
print("UseMockSearchOp Test Suite")
|
||||
print("=" * 100)
|
||||
print("\nThis test demonstrates the intelligent tool selection capability.")
|
||||
print("The LLM analyzes each query and selects the most appropriate search tool:\n")
|
||||
print(" • SearchToolA: Fast & shallow (best for simple factual queries)")
|
||||
print(" • SearchToolB: Balanced (best for medium complexity queries)")
|
||||
print(" • SearchToolC: Comprehensive & slow (best for complex research queries)")
|
||||
print("=" * 100)
|
||||
|
||||
async with FlowLLMApp(load_default_config=True):
|
||||
op = UseMockSearchOp()
|
||||
|
||||
# Test cases covering different complexities
|
||||
test_queries = [
|
||||
# Simple queries (should select SearchToolA)
|
||||
{
|
||||
"query": "What is the capital of France?",
|
||||
"expected_tool": "SearchToolA",
|
||||
"description": "Simple factual query"
|
||||
},
|
||||
{
|
||||
"query": "When was Python programming language created?",
|
||||
"expected_tool": "SearchToolA",
|
||||
"description": "Simple historical fact"
|
||||
},
|
||||
|
||||
# Medium complexity queries (should select SearchToolB)
|
||||
{
|
||||
"query": "How does quantum computing work?",
|
||||
"expected_tool": "SearchToolB",
|
||||
"description": "Medium complexity technical explanation"
|
||||
},
|
||||
{
|
||||
"query": "What are the main causes of climate change?",
|
||||
"expected_tool": "SearchToolB",
|
||||
"description": "Medium complexity scientific question"
|
||||
},
|
||||
|
||||
# Complex queries (should select SearchToolC)
|
||||
{
|
||||
"query": "Analyze the impact of artificial intelligence on global economy, employment, and society",
|
||||
"expected_tool": "SearchToolC",
|
||||
"description": "Complex multi-dimensional analysis"
|
||||
},
|
||||
{
|
||||
"query": "Compare and contrast different renewable energy solutions including their environmental impact, cost-effectiveness, and scalability",
|
||||
"expected_tool": "SearchToolC",
|
||||
"description": "Complex comparative analysis"
|
||||
}
|
||||
]
|
||||
|
||||
results = []
|
||||
correct_selections = 0
|
||||
|
||||
for test_case in test_queries:
|
||||
query = test_case["query"]
|
||||
expected = test_case["expected_tool"]
|
||||
description = test_case["description"]
|
||||
|
||||
print(f"\n🔍 Test Case: {description}")
|
||||
print(f" Expected Tool: {expected}")
|
||||
|
||||
result = await test_single_query(op, query)
|
||||
results.append({
|
||||
"test_case": test_case,
|
||||
"result": result
|
||||
})
|
||||
|
||||
# Check if the selection was correct
|
||||
if result["selected_tool"] == expected:
|
||||
correct_selections += 1
|
||||
print(f" ✅ Correct tool selected!")
|
||||
else:
|
||||
print(f" ⚠️ Different tool selected (this may still be valid)")
|
||||
|
||||
# Summary
|
||||
print("\n" + "=" * 100)
|
||||
print("Test Summary")
|
||||
print("=" * 100)
|
||||
print(f"Total Tests: {len(test_queries)}")
|
||||
print(f"Expected Matches: {correct_selections}/{len(test_queries)}")
|
||||
print(f"Success Rate: {correct_selections/len(test_queries)*100:.1f}%")
|
||||
print("\nNote: Tool selection may vary based on LLM reasoning, and different")
|
||||
print(" selections don't necessarily indicate errors.")
|
||||
print("=" * 100)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
|
||||
88
reme_ai/agent/tools/use_mock_search_op.py
Normal file
88
reme_ai/agent/tools/use_mock_search_op.py
Normal file
|
|
@ -0,0 +1,88 @@
|
|||
import asyncio
|
||||
import json
|
||||
from typing import Dict, Any
|
||||
|
||||
from flowllm.op.gallery.task_react_op import tools_schema_to_qwen_prompt
|
||||
from flowllm.schema.tool_call import ToolCall
|
||||
from loguru import logger
|
||||
|
||||
from flowllm.context import FlowContext, C
|
||||
from flowllm.enumeration.role import Role
|
||||
from flowllm.op.base_async_tool_op import BaseAsyncToolOp
|
||||
from flowllm.schema.message import Message
|
||||
|
||||
from reme_ai.agent.tools.mock_search_tools import SearchToolA, SearchToolB, SearchToolC
|
||||
|
||||
|
||||
@C.register_op()
|
||||
class UseMockSearchOp(BaseAsyncToolOp):
|
||||
file_path: str = __file__
|
||||
|
||||
def __init__(self, llm: str = "qwen3_30b_instruct", **kwargs):
|
||||
super().__init__(llm=llm, **kwargs)
|
||||
|
||||
async def select_tool(self, query: str, tool_ops: list[BaseAsyncToolOp]) -> ToolCall | None:
|
||||
assistant_message = await self.llm.achat(messages=[Message(role=Role.USER, content=query)],
|
||||
tools=[x.tool_call for x in tool_ops])
|
||||
|
||||
if assistant_message.tool_calls:
|
||||
return assistant_message.tool_calls[0]
|
||||
|
||||
return None
|
||||
|
||||
|
||||
async def async_execute(self):
|
||||
query: str = self.input_dict["query"]
|
||||
logger.info(f"query={query}")
|
||||
|
||||
tool_ops = [
|
||||
SearchToolA(),
|
||||
SearchToolB(),
|
||||
SearchToolC(),
|
||||
]
|
||||
|
||||
tool_result = await self.select_tool(query, tool_ops)
|
||||
if tool_result is None:
|
||||
...
|
||||
return
|
||||
|
||||
for op in tool_ops:
|
||||
op.tool_call.name == tool_result.name
|
||||
|
||||
|
||||
async def async_main():
|
||||
"""Test the UseMockSearchOp with different query types."""
|
||||
from flowllm.app import FlowLLMApp
|
||||
|
||||
async with FlowLLMApp(load_default_config=True):
|
||||
# Test queries of different complexities
|
||||
test_queries = [
|
||||
"What is the capital of France?", # Simple - should select SearchToolA
|
||||
"How does quantum computing work?", # Medium - should select SearchToolB
|
||||
"Analyze the impact of artificial intelligence on global economy, employment, and society", # Complex - should select SearchToolC
|
||||
"When was Python programming language created?", # Simple
|
||||
"Compare different types of renewable energy sources", # Complex
|
||||
]
|
||||
|
||||
op = UseMockSearchOp()
|
||||
|
||||
for query in test_queries:
|
||||
print(f"\n{'=' * 100}")
|
||||
print(f"Query: {query}")
|
||||
print(f"{'=' * 100}")
|
||||
|
||||
context = FlowContext(query=query)
|
||||
await op.async_call(context=context)
|
||||
|
||||
result = json.loads(context.use_mock_search_result)
|
||||
print(f"\nSelected Tool: {result['selected_tool']}")
|
||||
print(f"Reasoning: {result['reasoning']}")
|
||||
print(f"Complexity: {result['complexity']}")
|
||||
print(f"Success: {result['success']}")
|
||||
print(f"\nContent:\n{result['content']}")
|
||||
print(f"\n{'=' * 100}\n")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(async_main())
|
||||
|
||||
64
reme_ai/agent/tools/use_mock_search_prompt.yaml
Normal file
64
reme_ai/agent/tools/use_mock_search_prompt.yaml
Normal file
|
|
@ -0,0 +1,64 @@
|
|||
tool_selection_prompt: |
|
||||
You are an intelligent tool selection assistant. You have access to three different search tools with different characteristics:
|
||||
|
||||
**SearchToolA** - Fast but shallow search tool
|
||||
- Best for: Simple factual queries
|
||||
- Strengths: Very fast response (0.2-0.5s), high success rate for simple queries (95%)
|
||||
- Weaknesses: Poor performance on complex queries (50% success rate), limited depth
|
||||
|
||||
**SearchToolB** - Balanced search tool
|
||||
- Best for: Medium complexity queries
|
||||
- Strengths: Good balance of speed and quality (1.0-1.5s), consistent 80% success rate across all query types
|
||||
- Weaknesses: May be slower than needed for simple queries, lacks depth for very complex topics
|
||||
|
||||
**SearchToolC** - Comprehensive but slow search tool
|
||||
- Best for: Complex research queries
|
||||
- Strengths: Excellent for complex queries (95% success rate), comprehensive and detailed results
|
||||
- Weaknesses: Very slow (3.0-5.0s), overkill for simple queries (70% success rate)
|
||||
|
||||
Given the user's query, you need to:
|
||||
1. Analyze the complexity and requirements of the query
|
||||
2. Select the most appropriate tool (SearchToolA, SearchToolB, or SearchToolC)
|
||||
3. Provide the query parameter
|
||||
|
||||
User Query: {query}
|
||||
|
||||
Respond with a JSON object in the following format:
|
||||
{{
|
||||
"selected_tool": "SearchToolA" or "SearchToolB" or "SearchToolC",
|
||||
"reasoning": "Brief explanation of why this tool was selected",
|
||||
"query": "The search query to use"
|
||||
}}
|
||||
|
||||
tool_selection_prompt_zh: |
|
||||
你是一个智能工具选择助手。你可以访问三个具有不同特性的搜索工具:
|
||||
|
||||
**SearchToolA** - 快速但浅层的搜索工具
|
||||
- 最适合:简单的事实查询
|
||||
- 优势:响应非常快(0.2-0.5秒),简单查询的成功率高(95%)
|
||||
- 劣势:复杂查询表现不佳(成功率50%),深度有限
|
||||
|
||||
**SearchToolB** - 平衡的搜索工具
|
||||
- 最适合:中等复杂度的查询
|
||||
- 优势:速度和质量平衡良好(1.0-1.5秒),所有查询类型的成功率一致为80%
|
||||
- 劣势:对于简单查询可能比需要的慢,对于非常复杂的主题缺乏深度
|
||||
|
||||
**SearchToolC** - 全面但慢速的搜索工具
|
||||
- 最适合:复杂的研究查询
|
||||
- 优势:对复杂查询效果极佳(成功率95%),结果全面详细
|
||||
- 劣势:非常慢(3.0-5.0秒),对简单查询来说过度(成功率70%)
|
||||
|
||||
给定用户的查询,你需要:
|
||||
1. 分析查询的复杂度和要求
|
||||
2. 选择最合适的工具(SearchToolA、SearchToolB 或 SearchToolC)
|
||||
3. 提供查询参数
|
||||
|
||||
用户查询:{query}
|
||||
|
||||
用 JSON 对象格式回复:
|
||||
{{
|
||||
"selected_tool": "SearchToolA" 或 "SearchToolB" 或 "SearchToolC",
|
||||
"reasoning": "选择此工具的简要说明",
|
||||
"query": "要使用的搜索查询"
|
||||
}}
|
||||
|
||||
|
|
@ -16,7 +16,7 @@ class SummaryToolMemoryOp(BaseAsyncOp):
|
|||
file_path: str = __file__
|
||||
|
||||
def __init__(self,
|
||||
recent_call_count: int = 20,
|
||||
recent_call_count: int = 30,
|
||||
summary_sleep_interval: float = 1.0,
|
||||
**kwargs):
|
||||
super().__init__(**kwargs)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue