diff --git a/.gitignore b/.gitignore index 69361c03..0e2675c4 100644 --- a/.gitignore +++ b/.gitignore @@ -19,3 +19,4 @@ runs logs alfworld_data beyondagent/dataset/appworld/data +beyond* \ No newline at end of file diff --git a/README.md b/README.md index d96575d9..8dde8646 100644 --- a/README.md +++ b/README.md @@ -33,6 +33,7 @@ Or manually download and load the image. Here, we take elasticsearch-wolfi:9.0.0 ```shell docker pull docker.elastic.co/elasticsearch/elasticsearch-wolfi:9.0.0 docker run -p 9200:9200 \ + --memory='4GB' \ -e "discovery.type=single-node" \ -e "xpack.security.enabled=false" \ -e "xpack.license.self_generated.type=trial" \ diff --git a/experiencemaker/module/agent_wrapper/mcp_react_agent.py b/experiencemaker/module/agent_wrapper/mcp_react_agent.py new file mode 100644 index 00000000..349c4525 --- /dev/null +++ b/experiencemaker/module/agent_wrapper/mcp_react_agent.py @@ -0,0 +1,107 @@ + +# import os +# import time +# import json +# import best_logger +# import agentscope +# from experiencemaker.module.base_module import BaseModule +# from experiencemaker.schema.trajectory import Trajectory as OutputTrajectory +# from experiencemaker.schema.trajectory import Message as OutputTrajectoryMessage + +# from datetime import datetime +# from pydantic import BaseModel, Field +# import uuid +# from typing import ( +# Literal, +# Union, +# List, +# Optional, +# Dict, +# Any, +# Sequence, +# ) + + +# from experiencemaker.module.agent_wrapper.base_agent_wrapper import BaseAgentWrapper +# from agentscope.agents import ReActAgent, DialogAgent +# from beyond.trajectory import Trajectory as TrajectoryOperation +# from beyond.solver import TaskExecutor +# from agentscope.message import Msg +# from beyond.planner import * +# from beyond.debug import * +# from best_logger import * +# from loguru import logger + + +# def run_agent_and_extract_memory(msg_question, traj, agent): +# if not isinstance(msg_question, list): +# raise ValueError("msg_question should be a list of Msg objects") +# agent_ret = agent(msg_question) +# latest_agent_memory_buffer = msg_sort(agent.memory.get_memory()) +# traj.add_steps(latest_agent_memory_buffer) +# return latest_agent_memory_buffer, agent_ret + + +# def msg_sort(msg_list: List[Msg]) -> List[Msg]: +# """ +# Sort the message list by timestamp. +# """ +# sorted_msg = sorted( +# msg_list, +# key=lambda msg: datetime.strptime(msg.timestamp, "%Y-%m-%d %H:%M:%S.%f") +# ) + +# return sorted_msg + + +# class MainAgent(BaseAgentWrapper): +# mcp_url: str = Field( +# default=os.getenv('MCP_URL', 'http://localhost:33333/sse'), +# description="The URL of the MCP server.", +# ) + +# def __init__(self, *args, **kwargs): +# return super().__init__(*args, **kwargs) + +# def execute(self, query, **kwargs): +# question = query.strip() +# ref_answer = "not available" + +# print_dict({ +# 'question': question, +# 'ref_answer': ref_answer, +# }, mod='gaia_result') + +# try: +# except Exception as e: +# logger.exception(f"Error in solving task {question}: {e}") +# raise RuntimeError(f"Error in solving task {question}: {e}") + +# print_dict({ +# 'question': question, +# 'ref_answer': ref_answer, +# 'predicted_result': final_answer, +# }, mod='gaia_result') + + +# role_mapping = { +# 'system': 'system', +# 'end-user': 'user', +# 'commander': 'user', +# 'assistant': 'assistant', +# 'tool-agent': 'user', +# 'tool': 'tool', +# } +# output_trajectory = OutputTrajectory( +# steps=[OutputTrajectoryMessage( +# role=role_mapping[step.executor], +# content=step.content, +# timestamp=step.timestamp +# ) for step in traj.raw_steps], +# done=True, +# query=question, +# answer=final_answer, +# current_step=len(traj.raw_steps), +# ) +# return output_trajectory + diff --git a/experiencemaker/module/context_generator/traj_context_generator.py b/experiencemaker/module/context_generator/traj_context_generator.py new file mode 100644 index 00000000..1ce47dfa --- /dev/null +++ b/experiencemaker/module/context_generator/traj_context_generator.py @@ -0,0 +1 @@ +from experiencemaker.module.context_generator.base_context_generator import BaseContextGenerator \ No newline at end of file diff --git a/experiencemaker/module/summarizer/traj_summarizer.py b/experiencemaker/module/summarizer/traj_summarizer.py index 7e6575f5..ba2ee284 100644 --- a/experiencemaker/module/summarizer/traj_summarizer.py +++ b/experiencemaker/module/summarizer/traj_summarizer.py @@ -1,13 +1,16 @@ +import os from typing import List - from pydantic import Field from experiencemaker.schema.trajectory import Trajectory, Sample, SummaryMessage from experiencemaker.storage.base_vector_store import BaseVectorStore from experiencemaker.module.summarizer.base_summarizer import BaseSummarizer +from beyond.trajectory import Trajectory as TrajectoryOperation +from beyond.solver import TaskExecutor class TrajectorySummarizer(BaseSummarizer): vector_store: BaseVectorStore | None = Field(default=None) + samples: List[Sample] = Field(default=[]) def extract_samples(self, trajectories: List[Trajectory], **kwargs) -> List[Sample]: raise NotImplementedError @@ -15,12 +18,17 @@ class TrajectorySummarizer(BaseSummarizer): def insert_into_vector_store(self, samples: List[Sample], **kwargs): raise NotImplementedError + def process_trajectory(self, traj: Trajectory): + traj_operation = TrajectoryOperation() + for step in traj.steps: + step.executor = step.role.value + traj_operation.raw_steps += [step] + traj_description = traj_operation.chain_work_steps() + mcp_url = os.getenv('MCP_URL', 'http://localhost:33333/sse') + world_summary = traj_operation.generate_failure_ask_for_internet_help_raj_abs_post_level_3(mcp_url=mcp_url) + self.samples += [] + def execute(self, trajectories: List[Trajectory], return_samples: bool = True, **kwargs) -> List[Sample]: - samples: List[Sample] = self.extract_samples(trajectories, **kwargs) - self.insert_into_vector_store(samples, **kwargs) - - if return_samples: - return samples - - return [] - + for traj in trajectories: + self.process_trajectory(traj) + return self.samples diff --git a/experiencemaker/schema/vector_store_node.py b/experiencemaker/schema/vector_store_node.py index 55ede7e9..49ce532d 100644 --- a/experiencemaker/schema/vector_store_node.py +++ b/experiencemaker/schema/vector_store_node.py @@ -1,6 +1,5 @@ from typing import List from uuid import uuid4 - from pydantic import BaseModel, Field diff --git a/experiencemaker/tool/__init__.py b/experiencemaker/tool/__init__.py index f342c5ec..f304756b 100644 --- a/experiencemaker/tool/__init__.py +++ b/experiencemaker/tool/__init__.py @@ -1,10 +1,10 @@ -from experiencemaker.tool.python_tools.code_tool import CodeTool -from experiencemaker.tool.python_tools.dashscope_search_tool import DashscopeSearchTool -from experiencemaker.tool.python_tools.terminate_tool import TerminateTool +# from experiencemaker.tool.python_tools.code_tool import CodeTool +# from experiencemaker.tool.python_tools.dashscope_search_tool import DashscopeSearchTool +# from experiencemaker.tool.python_tools.terminate_tool import TerminateTool -from experiencemaker.utils.registry import Registry +# from experiencemaker.utils.registry import Registry -TOOL_REGISTRY = Registry("tools") -TOOL_REGISTRY.register(CodeTool) -TOOL_REGISTRY.register(DashscopeSearchTool) -TOOL_REGISTRY.register(TerminateTool) +# TOOL_REGISTRY = Registry("tools") +# TOOL_REGISTRY.register(CodeTool) +# TOOL_REGISTRY.register(DashscopeSearchTool) +# TOOL_REGISTRY.register(TerminateTool) diff --git a/pyproject.toml b/pyproject.toml index 7f2199a7..17b447f9 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -21,7 +21,7 @@ dependencies = [ "requests_oauthlib", "teamwork-mcp>=0.2.1", "agentscope", - "beyond @ file:///${PROJECT_ROOT}/beyondagent/third_party/beyond", + "beyond @ file:///${PROJECT_ROOT}/experiencemaker/third_party/beyond", "astor", ] diff --git a/test.py b/test.py new file mode 100644 index 00000000..524f2093 --- /dev/null +++ b/test.py @@ -0,0 +1,891 @@ + +msg = [ +""" +end-user +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`, +what existing methods are mentioned for the problem of formation maintenance? +""", + +""" +commander +─────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +You are the central commander, the brain of a coordinated multi-agent system. Your job is to make plans to solve a task given by +an end-user: + +In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`, +what existing methods are mentioned for the problem of formation maintenance? + + + +However, you should not solve this task directly, instead, you should decompose the task into sub-tasks, and present them in a +markdown to-do list format. As an example, here is a to-do list for user question to "Get the GDP of the country/countries with +highest rare earth production in 2024": + + ... analysis of the task ... + + + - [ ] Get the basic concept of what rare earth is, and whether it has direct relationship with GDP + - [ ] Get the rare earth production of each country in 2024, select the country with highest rare earth production + - [ ] Get the GDP of that country in 2024 + + + Get the GDP of that country in 2024 + + +[IMPORTANT] The to-do list should be clear, short and simple. The number of to-be-completed sub-tasks should be no more than 3. + +[IMPORTANT] Always use English to answer the question. + +[IMPORTANT] There are 3 cases: + + +for this case, the output should be: + + + ... 分析问题 ... + + + ... + +""", + +""" +assistant +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +Okay, so the user wants to know what existing methods are mentioned in the paper "Formation control with collision avoidance +through deep reinforcement learning using model-guided demonstration" for formation maintenance. Let me break this down. + +First, I need to understand the question. The user is asking about existing methods for formation maintenance as mentioned in that +specific paper. My task isn't to answer it directly but to decompose it into subtasks. The example given uses a to-do list with up +to three steps, so I should follow that structure. + +The first step would be to identify the key concepts. Formation maintenance is the problem area, and the paper is proposing a +method using deep reinforcement learning with model-guided demonstrations. But the user wants existing methods mentioned in the +paper, not the proposed one. So the paper probably reviews related work or existing approaches before introducing their own method +. + +So subtask one: Get the basic concept of formation maintenance to understand what the problem entails. That makes sense because +knowing the problem helps in identifying the methods. + +Subtask two: Access the paper and locate the sections where existing methods are discussed. Typically, in research papers, there's +a related work or literature review section that discusses existing approaches. The user needs to find those sections to extract +the mentioned methods. + +Third subtask: Extract and list the existing methods from the paper. Once the relevant sections are located, the next step is to +compile the list of methods they reference. + +I need to make sure the steps are clear and simple. The user might not have access to the paper, so maybe accessing it is part of +the steps. But if the assistant can't access external documents, how would it proceed? Wait, the example given in the problem +includes steps like getting GDP data which might require external sources. But in this case, the user is asking about a specific +paper, so the assumption here is that the assistant has access to the paper or can retrieve the information from it. However, in +reality, I don't have access to the paper. But since this is a hypothetical scenario for decomposition, I can outline the steps as +if the assistant can access the paper. + +So the to-do list would be: + +1. Understand formation maintenance basics. +2. Access the paper and find sections discussing existing methods. +3. Extract the listed existing methods. + +That's three steps, which fits the requirement. The next step would be the last one, extracting the methods. But the user just +needs the decomposition, not the execution. So the answer should present this structure. + + + The task requires identifying existing methods for formation maintenance as referenced in the specified paper. This involves +understanding the problem domain, locating the relevant sections in the paper, and extracting the cited methods. + + + - [ ] Understand the concept of formation maintenance to contextualize the problem + - [ ] Access the paper and locate its literature review/related work section discussing existing methods + - [ ] Extract and list the existing methods mentioned in the paper for formation maintenance + +""", + +""" +commander +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +You're a helpful assistant named tool_exe_agent. + +## What You Should Do: +Look back at the previous problem-solving steps and decide which external tool to use. +You should only focus on the very next unfinished step on the todo list. Do not look forward! + +## Note: +1. Do not make any assumptions about the tool call result, keep your thought and analysis short and concise. +2. Forget everything you know, only information given by the tools are reliable. +3. Keep your thought short and concise. +4. Do not make up fake information. +5. If you need to write python code, please use the `CodeExecutionTool` to execute it. And the `code` argument should use loguru +to print the result. For example: +```python +from loguru import logger +from ... (other imports) ... + +def you_function_name(): + # your code + +if __name__ == "__main__": + try: + logger.success(you_function_name()) + except Exception as e: + logger.exception(e) +``` +6. If there are tool-call histories, consider their results, do not repeat failure over and over again. + + +7. Respect to todo list! Respect to todo list! Respect to todo list! if there is a todo list, you must follow it. +For a example todo list: + + - [x] job 1 (`x` means completed) + - [ ] job 2 + - [ ] job 3 + +You must focus on job 2 only. +8. When you use chrome related tools, do not ask for any confirmation, just ask for materials. +9. field should be a JSON object. For example: {"name": "RealChromeBrowserUse", "arguments": {"task": "在 +小红书中搜索杭州一日游攻略。"}} + +## Tool names and arguments: +[{"type": "function", "function": {"name": "CodeExecutionTool", "description": "用于执行Python代码的工具,可以返回执行结果、打印 +结果和错误Traceback。调用时必须使用loguru打印信息。", "parameters": {"properties": {"tool_name": {"default": "CodeExecutionTool", +"description": "用于执行Python代码的工具,可以返回执行结果、打印结果和错误Traceback。调用时必须使用loguru打印信息。", "title": " +Tool Name", "type": "string"}, "code": {"default": "", "description": "要执行的Python代码", "title": "Code", "type": "string"}}, " +title": "CodeExecutionTool", "type": "object"}}}, {"type": "function", "function": {"name": "RealChromeBrowserUse", "description": +"通过真正的Chrome浏览器搜索网页,效率略低,但更稳定,可以访问更广泛的网站。", "parameters": {"properties": {"tool_name": {"default +": "RealChromeBrowserUse", "description": "通过真正的Chrome浏览器搜索网页,效率略低,但更稳定,可以访问更广泛的网站。", "title": " +Tool Name", "type": "string"}, "task": {"default": "", "description": "一句话,描述一个明确的子任务(task)。例如:'获取2024年5月1 +日阿里巴巴集团控股有限公司的收盘价'。", "title": "Task", "type": "string"}, "startup_url": {"anyOf": [{"type": "string"}, {"type": +"null"}], "default": "", "description": "起始url(非必要参数,默认留空)。可以指定一个url作为任务的起点,从而提高效率。", "title": +"Startup Url"}, "additional_tip": {"anyOf": [{"type": "string"}, {"type": "null"}], "default": "", "description": "执行浏览器任务 +时的一些建议(非必要参数,默认留空)。", "title": "Additional Tip"}}, "title": "RealChromeBrowserUse", "type": "object"}}}, {"type +": "function", "function": {"name": "GetDownloadedFiles", "description": "获取下载完的文件。", "parameters": {"properties": {"tool +_name": {"default": "GetDownloadedFiles", "description": "获取下载完的文件。", "title": "Tool Name", "type": "string"}, "file_path +": {"default": "", "description": "文件的路径。", "title": "File Path", "type": "string"}}, "title": "GetDownloadedFiles", "type": +"object"}}}] + +## Answer Format: + + ... analyze which tool should be used ... + + + + ... next objective from the todo list ... + + + + {"name": ..., "arguments": ...} + +""", + +""" +tool-agent +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── + + The next step is to access the paper and locate sections discussing existing methods. Since the paper is not provided directly +, an external tool like a browser can be used to search for and access the paper. + + + + Access the paper and find sections discussing existing methods. + + + + {"name": "RealChromeBrowserUse", "arguments": {"task": "Search for the paper 'Formation control with collision avoidance +through deep reinforcement learning using model-guided demonstration' and locate sections discussing existing methods."}} +[{'name': 'RealChromeBrowserUse', 'arguments': {'task': "Search for the paper 'Formation control with +collision avoidance through deep reinforcement learning using model-guided demonstration' and locate sections discussing existing +methods."}}] +""", + +""" +tool +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +# The selected tools + +{'name': 'RealChromeBrowserUse', 'arguments': {'task': "Search for the paper 'Formation control with collision avoidance through +deep reinforcement learning using model-guided demonstration' and locate sections discussing existing methods.", 'mcp_kwargs': {}} +} + +# The toolcall result + +[field 'model_output'] + [field 'current_state'] + [field 'evaluation_previous_goal'] + Success - Successfully extracted content from the IEEE page which reveals discussion of existing methods + [field 'memory'] + Found key information about existing methods in the abstract and introduction. The paper mentions that 'most of the existing +works are simple combinations of methods dealing with the two problems separately' - referring to formation maintenance and +collision avoidance. The introduction section provides context about multiagent systems and formation control challenges. However, +access to the full paper requires a subscription. Step 13/16. + [field 'next_goal'] + Since we found the key discussion about existing methods but cannot access the full text due to subscription requirement, we +should summarize our findings and complete the task + [field 'action'] + [field 'done'] + [field 'text'] + I found the paper and located discussions about existing methods. From the accessible content, the key points about +existing methods are: + + 1. Most existing works handle formation maintenance and collision avoidance as separate problems, simply combining +different methods for each issue. + + 2. The paper identifies this as a limitation, as the followers need to handle both formation maintenance and collision +avoidance simultaneously. + + 3. The introduction provides context that formation control is a fundamental problem in multiagent systems, where both +formation maintenance and collision avoidance are essential considerations. + + 4. The authors propose their new method specifically to address the limitations of existing approaches by using deep +reinforcement learning to handle both aspects in an integrated way. + + While more detailed discussion of existing methods likely exists in the full paper, access requires a subscription. The +abstract and introduction sections provide the key critique that existing methods typically separate the two problems rather than +addressing them in an integrated way. + [field 'success'] + True +[field 'result'] + [field 'extracted_content'] + I found the paper and located discussions about existing methods. From the accessible content, the key points about existing +methods are: + + 1. Most existing works handle formation maintenance and collision avoidance as separate problems, simply combining different +methods for each issue. + + 2. The paper identifies this as a limitation, as the followers need to handle both formation maintenance and collision +avoidance simultaneously. + + 3. The introduction provides context that formation control is a fundamental problem in multiagent systems, where both +formation maintenance and collision avoidance are essential considerations. + + 4. The authors propose their new method specifically to address the limitations of existing approaches by using deep +reinforcement learning to handle both aspects in an integrated way. + + While more detailed discussion of existing methods likely exists in the full paper, access requires a subscription. The +abstract and introduction sections provide the key critique that existing methods typically separate the two problems rather than +addressing them in an integrated way. +[field 'state'] + [field 'url'] + https://ieeexplore.ieee.org/abstract/document/9142429 + [field 'title'] + Formation Control With Collision Avoidance Through Deep Reinforcement Learning Using Model-Guided Demonstration | IEEE +Journals & Magazine | IEEE Xplore +[field 'page_meta'] + [field 'content'] + Opens in a new window Opens an external website Opens an external website in a new window + + Close this dialog + + This website utilizes technologies such as cookies to enable essential site functionality, as well as for analytics, +personalization, and targeted advertising. To learn more, view the following link: Privacy Policy + + + + Close Cookie Preferences + + + + + + + + + + + + Formation Control With Collision Avoidance Through Deep Reinforcement Learning Using Model-Guided Demonstration | IEEE +Journals & Magazine | IEEE Xplore + + + + + + + + + + + + + + + + + + + + + + + Skip to Main Content + + * IEEE.org + * IEEE *Xplore* + * IEEE SA + * IEEE Spectrum + * More Sites + + Subscribe + + * + Donate + + + Cart + + + Create Account + + Personal Sign In + + * Browse + * My Settings + * Help + + Institutional Sign In + + Institutional Sign In + + AllBooksConferencesCoursesJournals & MagazinesStandardsAuthorsCitations + + ADVANCED SEARCH + + Journals & Magazines >IEEE Transactions on Neural N... >Volume: 32 Issue: 6 + + Formation Control With Collision Avoidance Through Deep Reinforcement Learning Using Model-Guided Demonstration + =============================================================================================================== + + Publisher: IEEE + + Cite This + + PDF + + Zezhi Sui; Zhiqiang Pu; Jianqiang Yi; Shiguang Wu + + All Authors + + Sign In or Purchase + + 78 + + Cites in + + Papers + + 4925 + + Full + + Text Views + + * Alerts + + Alerts + ====== + + Manage Content Alerts + + Add to Citation Alerts + + --- + + Abstract + + Document Sections + ----------------- + + * I. + + Introduction + * II. + + Preliminaries + * III. + + Problem Formulation + * IV. + + Approach + * V. + + Simulation and Experiment Results + + Show Full Outline + + Authors + + Figures + + References + + Citations + + Keywords + + Metrics + + More Like This + + * Download PDF + * Download References + * Request Permissions + * Save to + * Alerts + + Abstract: + --------- + + Generating collision-free, time-efficient paths in an uncertain dynamic environment poses huge challenges for the formation +control with collision avoidance (FCCA) proble...Show More + + Metadata + -------- + + Abstract: + --------- + + Generating collision-free, time-efficient paths in an uncertain dynamic environment poses huge challenges for the formation +control with collision avoidance (FCCA) problem in a leader-follower structure. In particular, the followers have to take both +formation maintenance and collision avoidance into account simultaneously. Unfortunately, most of the existing works are simple +combinations of methods dealing with the two problems separately. In this article, a new method based on deep reinforcement +learning (RL) is proposed to solve the problem of FCCA. Especially, the learning-based policy is extended to the field of +formation control, which involves a two-stage training framework: an imitation learning (IL) and later an RL. In the IL stage, a +model-guided method consisting of a consensus theory-based formation controller and an optimal reciprocal collision avoidance +strategy is designed to speed up training and increase efficiency. In the RL stage, a compound reward function is presented to +guide the training. In addition, we design a formation-oriented network structure to perceive the environment. Long short-term +memory is adopted to enable the network structure to perceive the information of obstacles of an uncertain number, and a transfer +training approach is adopted to improve the generalization of the network in different scenarios. Numerous representative +simulations are conducted, and our method is further deployed to an experimental platform based on a multiomnidirectional-wheeled +car system. The effectiveness and practicability of our proposed method are validated through both the simulation and experiment +results. + + **Published in:** IEEE Transactions on Neural Networks and Learning Systems ( Volume: 32, Issue: 6, June 2021) + + **Page(s):** 2358 - 2372 + + **Date of Publication:** 16 July 2020 + + ISSN Information: + ----------------- + + **PubMed ID:** 32673195 + + **DOI:** 10.1109/TNNLS.2020.3004893 + + Publisher: IEEE + + Funding Agency: + --------------- + + Contents + + --- + + ### I. Introduction + + Multiagent systems have received increasing attention from researchers in recent years because of their great potential in a +variety of fields. Their applications can be found in collaborative explorations of monitoring and rescue, cooperative control of +satellite clusters, and formation control of unmanned aerial vehicles [1]–[3], and so on. The basic concept of multiagent systems +is to use individuals to cooperatively solve complex tasks that cannot be accomplished by a single agent even with expensive +equipment. Formation control is one fundamental problem for multiagent systems, of which the objective is to achieve and maintain +a certain formation shape so that the multiagent system could collectively accomplish a specific task. It is obvious that +formation maintenance is an essential problem in formation control. In addition, collision avoidance should also be taken into +consideration to guarantee the safety of the multiagent system. Due to the interaction among agents and the tradeoff between +collision avoidance and formation maintenance, finding collision-free, time-efficient paths in an uncertain dynamic environment +remains challenging. + + Sign in to Continue Reading + + Authors + ------- + + Figures + ------- + + References + ---------- + + Citations + --------- + + Keywords + -------- + + Metrics + ------- + + More Like This + + Collision Avoidance and Connectivity Preservation for Time-Varying Formation of Second-Order Multi-Agent Systems With a +Dynamic Leader + + IEEE Access + + Published: 2022 + + Multi-Agent System Based on Machine Learning for Early Diagnosis of Diabetes + + 2020 IEEE 6th International Conference on Optimization and Applications (ICOA) + + Published: 2020 + + Show More + + References + ========== + + **References is not available for this document.** + + IEEE Personal Account + --------------------- + + * Change username/password + + Purchase Details + ---------------- + + * Payment Options + * View Purchased Documents + + Profile Information + ------------------- + + * Communications Preferences + * Profession and Education + * Technical interests + + Need Help? + ---------- + + * US & Canada: +1 800 678 4333 + * Worldwide: +1 732 981 0060 + * Contact & Support + + Follow + ------ + + About IEEE *Xplore* | Contact Us | Help | Accessibility | Terms of Use | Nondiscrimination Policy | IEEE Ethics Reporting | +Sitemap | IEEE Privacy Policy + + A public charity, IEEE is the world's largest technical professional organization dedicated to advancing technology for the +benefit of humanity. + + © Copyright 2025 IEEE - All rights reserved, including rights for text and data mining and training of artificial intelligence +and similar technologies. + + ### IEEE Account + + * Change Username/Password + * Update Address + + ### Purchase Details + + * Payment Options + * Order History + * View Purchased Documents + + ### Profile Information + + * Communications Preferences + * Profession and Education + * Technical Interests + + ### Need Help? + + * **US & Canada:** +1 800 678 4333 + * **Worldwide:** +1 732 981 0060 + * Contact & Support + + * About IEEE *Xplore* + * Contact Us + * Help + * Accessibility + * Terms of Use + * Nondiscrimination Policy + * Sitemap + * Privacy & Opting Out of Cookies + + A not-for-profit organization, IEEE is the world's largest technical professional organization dedicated to advancing +technology for the benefit of humanity. + © Copyright 2025 IEEE - All rights reserved. Use of this web site signifies your agreement to the terms and conditions. +""", + +""" +commander +─────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +Follow the plan and try to solve the task step by step according to the plan and tool execution results. +Your primary job is analyze the tool execution, do not call any external tools this time. + +[IMPORTANT] Always use English to answer the question. +[IMPORTANT] Do not call any external tools this time, analyze only. +""", + +""" +assistant +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +Okay, let's tackle this step by step. The user wants to know the existing methods mentioned in the paper for formation maintenance +. From the tool execution results, I accessed the paper's abstract and introduction via the IEEE link. + +In the abstract, it says, "most of the existing works are simple combinations of methods dealing with the two problems separately. +" The two problems here are formation maintenance and collision avoidance. So existing methods typically handle these two aspects +independently and then combine them. + +The introduction further explains that formation control is a fundamental problem in multiagent systems, where both formation +maintenance and collision avoidance are essential. The authors point out that existing approaches don't integrate these two +aspects effectively. They mention that followers need to consider both simultaneously, which current methods don't address +properly. + +Additionally, the paper's approach section refers to a model-guided method using consensus theory-based formation controllers and +optimal reciprocal collision avoidance strategies during the imitation learning stage. These are likely the existing techniques +they build upon. + +So putting this together, the existing methods mentioned are those that separate formation maintenance and collision avoidance, +using consensus theory for formation control and reciprocal collision avoidance strategies, then combining them. The paper +critiques this approach as insufficient and proposes an integrated solution using deep reinforcement learning. + +Based on the information extracted from the paper's abstract and introduction, the existing methods for formation maintenance +mentioned in the paper can be summarized as follows: + +1. **Decoupled Approaches**: Most existing works treat formation maintenance and collision avoidance as separate problems, +combining solutions for each in a non-integrated manner. This means formation control and collision avoidance are addressed +individually and then merged, rather than being optimized together. + +2. **Consensus Theory-Based Formation Controllers**: The paper references traditional methods that use consensus algorithms (a +common approach in multiagent systems) to achieve formation maintenance. These controllers rely on mathematical frameworks to +ensure agents reach and maintain a desired formation shape. + +3. **Optimal Reciprocal Collision Avoidance (ORCA)**: The paper mentions ORCA as a strategy used in existing methods for collision +avoidance. ORCA is a well-known algorithm that ensures collision-free paths by calculating velocity obstacles for each agent to +avoid collisions with others. + +4. **Model-Guided Demonstration Frameworks**: While the paper's proposed method uses a two-stage training framework (imitation +learning followed by reinforcement learning), it builds on prior work that employs model-guided demonstrations. These +demonstrations often rely on predefined controllers (like consensus or ORCA) to generate training data for learning-based systems. + +### Key Critique in the Paper: +The authors highlight that these existing methods suffer from limitations such as: +- **Suboptimal Integration**: Decoupling formation maintenance and collision avoidance leads to inefficiencies, as agents must +balance both tasks simultaneously in dynamic environments. +- **Over-Reliance on Predefined Models**: Traditional approaches (e.g., consensus theory or ORCA) may struggle in uncertain or +complex environments where adaptability is required. + +The paper proposes a novel deep reinforcement learning (DRL) framework to address these limitations by integrating both tasks into +a unified policy, moving beyond the decoupled and model-dependent strategies of prior work. +""", + +""" +commander +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +You are the central commander, the brain of a coordinated multi-agent system. Your job is to make plans to solve a task given by +an end-user: + +In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`, +what existing methods are mentioned for the problem of formation maintenance? + + + +However, you should not solve this task directly, instead, you should decompose the task into sub-tasks, and present them in a +markdown to-do list format. As an example, here is a to-do list for user question to "Get the GDP of the country/countries with +highest rare earth production in 2024": + + ... analysis of the task ... + + + - [ ] Get the basic concept of what rare earth is, and whether it has direct relationship with GDP + - [ ] Get the rare earth production of each country in 2024, select the country with highest rare earth production + - [ ] Get the GDP of that country in 2024 + + + Get the GDP of that country in 2024 + + +[IMPORTANT] The to-do list should be clear, short and simple. The number of to-be-completed sub-tasks should be no more than 3. + +[IMPORTANT] Always use English to answer the question. + +[IMPORTANT] There are 3 cases: + +=== + 1. If there are no previous to-do list in the conversation, write a new one. + for this case, the output should be: + + + ... we are dealing with the first case (as this is the first time that the problem is raised), now analyze the task ... + + + ... + + +=== + 2. if there are already previous to-do list in the conversation, you should re-write the to-do list again, investigate what +tasks have ALREADY been done, remove tasks that is IMPOSSIBLE to be done and found alternative solutions whenever needed. + for this case, the output should be: + + + ... we are dealing with the 2rd case, analyze the sub-tasks we have already completed, and write a new plan based on +current situation ... + + + ... + + + ... + + +=== + 3. If the task is already completed, you should return a "TASK_COMPLETED" signal, and do not write any to-do list. + for this case, the output should be: + + + ... according to previous conversations, we have already obtained a possible answer, next, we verify the result ... + + TASK_COMPLETED +""", + +""" +commander +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +Now, determine whether the task is completed. If the task is already completed, say `TASK_COMPLETED`, otherwise, update the to-do +list. +Note that if you need data in the analysis, you have to point out that the data is just a assumption. +""", + +""" +assistant +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +Okay, let me check if the task is completed. The user asked about existing methods mentioned in the paper for formation +maintenance. + +From the tool execution, I accessed the paper's abstract and introduction. The abstract states that existing works combine methods +for formation maintenance and collision avoidance separately. The introduction mentions consensus theory-based controllers and +ORCA as part of existing approaches. Also, the paper's approach section refers to model-guided demonstrations using those +traditional methods. + +So, the existing methods identified are: +1. Decoupled approaches handling formation and collision avoidance separately. +2. Consensus theory-based formation controllers. +3. Optimal Reciprocal Collision Avoidance (ORCA). +4. Model-guided demonstration frameworks relying on predefined controllers. + +The analysis seems thorough based on the accessible parts of the paper. The full text might have more details, but the key +existing methods are covered in the abstract and introduction. Since the user's question is about what existing methods are +mentioned, and we've extracted those from the available content, the task should be completed. + + + Based on the information extracted from the paper's abstract and introduction sections (accessible without subscription), we +have identified the existing methods mentioned in the paper for formation maintenance: + + 1. **Decoupled Approaches**: Existing works treat formation maintenance and collision avoidance as separate problems, +combining solutions for each independently. + 2. **Consensus Theory-Based Formation Controllers**: Traditional methods using consensus algorithms to achieve formation +maintenance. + 3. **Optimal Reciprocal Collision Avoidance (ORCA)**: A strategy used in existing methods for collision avoidance. + 4. **Model-Guided Demonstration Frameworks**: Prior work employing predefined controllers (e.g., consensus or ORCA) to +generate training data for learning-based systems. + + While the full paper might contain additional details, the accessible content provides sufficient evidence to answer the user' +s question about the *mentioned* existing methods. Further access to the full text would not invalidate these findings but could +only add supplementary context. + + +TASK_COMPLETED +""", + +""" +end-user +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`, +what existing methods are mentioned for the problem of formation maintenance? +""", + +""" +commander +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +As you can see, the former agents have done a lot of work. +Now please make a final answer of the original task based on our conversation. + +Please pay special attention to the format in which the answer is presented. +You should first analyze the answer format required by the question and then output the final answer that meets the format +requirements. +Your response should include the following content: +- `final_answer`: enclosed by , the final answer to the question. +Here are some hint about the final answer: + +Your final answer must be output exactly in the format specified by the question. It should be a number OR as few words as +possible OR a comma separated list of numbers and/or strings: +- If you are asked for a number, don't use comma to write your number neither use units such as $ or percent sign unless specified +otherwise. +- If you are asked for a string, don't use articles, neither abbreviations (e.g. for cities), and write the digits in plain text +unless specified otherwise. +- If you are asked for a comma separated list, apply the above rules depending of whether the element to be put in the list is a +number or a string. + +""", + +""" +assistant +────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +Okay, let me try to figure out how to structure the final answer based on the user's requirements. + +The original question asked what existing methods are mentioned in the paper for formation maintenance. From the previous steps, +we identified four main methods: decoupled approaches, consensus theory-based controllers, ORCA, and model-guided demonstrations. +However, the user wants the answer in a specific format—either a number, a short string, or a comma-separated list. + +Looking back at the hint provided, if it's a list, each element should follow the rules (no articles, no abbreviations unless +specified). The answer should be concise. The paper's abstract and introduction mention that existing works are simple +combinations of methods dealing with the two problems (formation maintenance and collision avoidance) separately. Then, the +approach section refers to using consensus theory and ORCA in the model-guided phase. + +But the key point is that the existing methods are decoupled approaches, consensus-based formation control, and ORCA. The model- +guided demonstrations are part of the proposed method, not the existing ones. Wait, in the analysis, point 4 was model-guided +demonstration frameworks as existing methods. But in the paper's approach, they use a two-stage training framework starting with +imitation learning (which uses model-guided methods). So the existing methods include consensus and ORCA, which are part of the +model-guided demonstration. + +So the answer should list the existing methods as: +1. Decoupled approaches combining separate methods for formation and collision avoidance. +2. Consensus theory-based formation controllers. +3. Optimal Reciprocal Collision Avoidance (ORCA). + +But the user wants the answer in a comma-separated list with as few words as possible. So perhaps abbreviate ORCA but spell out +the others. However, the hint says no abbreviations unless specified. Wait, ORCA is an abbreviation, but in the paper, it's +referred to as Optimal Reciprocal Collision Avoidance (ORCA). Since the paper uses the acronym, maybe it's acceptable. +Alternatively, use the full name. But the example answer in the hint says "don't use abbreviations (e.g. for cities)", so maybe +avoid ORCA and write out Optimal Reciprocal Collision Avoidance. However, in the context of academic papers, ORCA is a standard +term, so using the acronym might be acceptable. Need to check. Since the paper mentions ORCA in the abstract (from the tool result +), it's okay to use the acronym here. + +So the final answer would be a list of three items: +Decoupled approaches,Consensus theory-based formation controllers,Optimal Reciprocal Collision Avoidance (ORCA) + +Wait, but the user said "as few words as possible" or "comma separated list of numbers and/or strings". So maybe shorten them: +Decoupled approaches,Consensus-based controllers,ORCA + +Yes, that's more concise. The paper refers to "consensus theory-based formation controller" which can be shortened to "consensus- +based controllers". ORCA is acceptable as an acronym. So the final answer is a comma-separated list of these three methods. + +Decoupled approaches,Consensus-based controllers,Optimal Reciprocal Collision Avoidance +""" + +] + +from experiencemaker.module.summarizer.traj_summarizer import TrajectorySummarizer +from experiencemaker.schema.trajectory import Trajectory, Sample, SummaryMessage, Message +from experiencemaker.schema.trajectory import Message as BMessage + +class Message(BMessage): + executor: str = None + +ts = TrajectorySummarizer() +role_mapping = { + 'system': 'system', + 'end-user': 'user', + 'commander': 'user', + 'assistant': 'assistant', + 'tool-agent': 'user', + 'tool': 'tool', +} +msg_f = [] +for m in msg: + role = m.split('\n')[1].strip() + content = '\n'.join(m.split('\n')[3:]).strip() + msg_f += [Message(role=role_mapping[role], content=content)] +traj = Trajectory(steps=msg_f) +ts.execute(trajectories=[traj], return_samples=True)