diff --git a/.gitignore b/.gitignore
index 69361c03..0e2675c4 100644
--- a/.gitignore
+++ b/.gitignore
@@ -19,3 +19,4 @@ runs
logs
alfworld_data
beyondagent/dataset/appworld/data
+beyond*
\ No newline at end of file
diff --git a/README.md b/README.md
index d96575d9..8dde8646 100644
--- a/README.md
+++ b/README.md
@@ -33,6 +33,7 @@ Or manually download and load the image. Here, we take elasticsearch-wolfi:9.0.0
```shell
docker pull docker.elastic.co/elasticsearch/elasticsearch-wolfi:9.0.0
docker run -p 9200:9200 \
+ --memory='4GB' \
-e "discovery.type=single-node" \
-e "xpack.security.enabled=false" \
-e "xpack.license.self_generated.type=trial" \
diff --git a/experiencemaker/module/agent_wrapper/mcp_react_agent.py b/experiencemaker/module/agent_wrapper/mcp_react_agent.py
new file mode 100644
index 00000000..349c4525
--- /dev/null
+++ b/experiencemaker/module/agent_wrapper/mcp_react_agent.py
@@ -0,0 +1,107 @@
+
+# import os
+# import time
+# import json
+# import best_logger
+# import agentscope
+# from experiencemaker.module.base_module import BaseModule
+# from experiencemaker.schema.trajectory import Trajectory as OutputTrajectory
+# from experiencemaker.schema.trajectory import Message as OutputTrajectoryMessage
+
+# from datetime import datetime
+# from pydantic import BaseModel, Field
+# import uuid
+# from typing import (
+# Literal,
+# Union,
+# List,
+# Optional,
+# Dict,
+# Any,
+# Sequence,
+# )
+
+
+# from experiencemaker.module.agent_wrapper.base_agent_wrapper import BaseAgentWrapper
+# from agentscope.agents import ReActAgent, DialogAgent
+# from beyond.trajectory import Trajectory as TrajectoryOperation
+# from beyond.solver import TaskExecutor
+# from agentscope.message import Msg
+# from beyond.planner import *
+# from beyond.debug import *
+# from best_logger import *
+# from loguru import logger
+
+
+# def run_agent_and_extract_memory(msg_question, traj, agent):
+# if not isinstance(msg_question, list):
+# raise ValueError("msg_question should be a list of Msg objects")
+# agent_ret = agent(msg_question)
+# latest_agent_memory_buffer = msg_sort(agent.memory.get_memory())
+# traj.add_steps(latest_agent_memory_buffer)
+# return latest_agent_memory_buffer, agent_ret
+
+
+# def msg_sort(msg_list: List[Msg]) -> List[Msg]:
+# """
+# Sort the message list by timestamp.
+# """
+# sorted_msg = sorted(
+# msg_list,
+# key=lambda msg: datetime.strptime(msg.timestamp, "%Y-%m-%d %H:%M:%S.%f")
+# )
+
+# return sorted_msg
+
+
+# class MainAgent(BaseAgentWrapper):
+# mcp_url: str = Field(
+# default=os.getenv('MCP_URL', 'http://localhost:33333/sse'),
+# description="The URL of the MCP server.",
+# )
+
+# def __init__(self, *args, **kwargs):
+# return super().__init__(*args, **kwargs)
+
+# def execute(self, query, **kwargs):
+# question = query.strip()
+# ref_answer = "not available"
+
+# print_dict({
+# 'question': question,
+# 'ref_answer': ref_answer,
+# }, mod='gaia_result')
+
+# try:
+# except Exception as e:
+# logger.exception(f"Error in solving task {question}: {e}")
+# raise RuntimeError(f"Error in solving task {question}: {e}")
+
+# print_dict({
+# 'question': question,
+# 'ref_answer': ref_answer,
+# 'predicted_result': final_answer,
+# }, mod='gaia_result')
+
+
+# role_mapping = {
+# 'system': 'system',
+# 'end-user': 'user',
+# 'commander': 'user',
+# 'assistant': 'assistant',
+# 'tool-agent': 'user',
+# 'tool': 'tool',
+# }
+# output_trajectory = OutputTrajectory(
+# steps=[OutputTrajectoryMessage(
+# role=role_mapping[step.executor],
+# content=step.content,
+# timestamp=step.timestamp
+# ) for step in traj.raw_steps],
+# done=True,
+# query=question,
+# answer=final_answer,
+# current_step=len(traj.raw_steps),
+# )
+# return output_trajectory
+
diff --git a/experiencemaker/module/context_generator/traj_context_generator.py b/experiencemaker/module/context_generator/traj_context_generator.py
new file mode 100644
index 00000000..1ce47dfa
--- /dev/null
+++ b/experiencemaker/module/context_generator/traj_context_generator.py
@@ -0,0 +1 @@
+from experiencemaker.module.context_generator.base_context_generator import BaseContextGenerator
\ No newline at end of file
diff --git a/experiencemaker/module/summarizer/traj_summarizer.py b/experiencemaker/module/summarizer/traj_summarizer.py
index 7e6575f5..ba2ee284 100644
--- a/experiencemaker/module/summarizer/traj_summarizer.py
+++ b/experiencemaker/module/summarizer/traj_summarizer.py
@@ -1,13 +1,16 @@
+import os
from typing import List
-
from pydantic import Field
from experiencemaker.schema.trajectory import Trajectory, Sample, SummaryMessage
from experiencemaker.storage.base_vector_store import BaseVectorStore
from experiencemaker.module.summarizer.base_summarizer import BaseSummarizer
+from beyond.trajectory import Trajectory as TrajectoryOperation
+from beyond.solver import TaskExecutor
class TrajectorySummarizer(BaseSummarizer):
vector_store: BaseVectorStore | None = Field(default=None)
+ samples: List[Sample] = Field(default=[])
def extract_samples(self, trajectories: List[Trajectory], **kwargs) -> List[Sample]:
raise NotImplementedError
@@ -15,12 +18,17 @@ class TrajectorySummarizer(BaseSummarizer):
def insert_into_vector_store(self, samples: List[Sample], **kwargs):
raise NotImplementedError
+ def process_trajectory(self, traj: Trajectory):
+ traj_operation = TrajectoryOperation()
+ for step in traj.steps:
+ step.executor = step.role.value
+ traj_operation.raw_steps += [step]
+ traj_description = traj_operation.chain_work_steps()
+ mcp_url = os.getenv('MCP_URL', 'http://localhost:33333/sse')
+ world_summary = traj_operation.generate_failure_ask_for_internet_help_raj_abs_post_level_3(mcp_url=mcp_url)
+ self.samples += []
+
def execute(self, trajectories: List[Trajectory], return_samples: bool = True, **kwargs) -> List[Sample]:
- samples: List[Sample] = self.extract_samples(trajectories, **kwargs)
- self.insert_into_vector_store(samples, **kwargs)
-
- if return_samples:
- return samples
-
- return []
-
+ for traj in trajectories:
+ self.process_trajectory(traj)
+ return self.samples
diff --git a/experiencemaker/schema/vector_store_node.py b/experiencemaker/schema/vector_store_node.py
index 55ede7e9..49ce532d 100644
--- a/experiencemaker/schema/vector_store_node.py
+++ b/experiencemaker/schema/vector_store_node.py
@@ -1,6 +1,5 @@
from typing import List
from uuid import uuid4
-
from pydantic import BaseModel, Field
diff --git a/experiencemaker/tool/__init__.py b/experiencemaker/tool/__init__.py
index f342c5ec..f304756b 100644
--- a/experiencemaker/tool/__init__.py
+++ b/experiencemaker/tool/__init__.py
@@ -1,10 +1,10 @@
-from experiencemaker.tool.python_tools.code_tool import CodeTool
-from experiencemaker.tool.python_tools.dashscope_search_tool import DashscopeSearchTool
-from experiencemaker.tool.python_tools.terminate_tool import TerminateTool
+# from experiencemaker.tool.python_tools.code_tool import CodeTool
+# from experiencemaker.tool.python_tools.dashscope_search_tool import DashscopeSearchTool
+# from experiencemaker.tool.python_tools.terminate_tool import TerminateTool
-from experiencemaker.utils.registry import Registry
+# from experiencemaker.utils.registry import Registry
-TOOL_REGISTRY = Registry("tools")
-TOOL_REGISTRY.register(CodeTool)
-TOOL_REGISTRY.register(DashscopeSearchTool)
-TOOL_REGISTRY.register(TerminateTool)
+# TOOL_REGISTRY = Registry("tools")
+# TOOL_REGISTRY.register(CodeTool)
+# TOOL_REGISTRY.register(DashscopeSearchTool)
+# TOOL_REGISTRY.register(TerminateTool)
diff --git a/pyproject.toml b/pyproject.toml
index 7f2199a7..17b447f9 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -21,7 +21,7 @@ dependencies = [
"requests_oauthlib",
"teamwork-mcp>=0.2.1",
"agentscope",
- "beyond @ file:///${PROJECT_ROOT}/beyondagent/third_party/beyond",
+ "beyond @ file:///${PROJECT_ROOT}/experiencemaker/third_party/beyond",
"astor",
]
diff --git a/test.py b/test.py
new file mode 100644
index 00000000..524f2093
--- /dev/null
+++ b/test.py
@@ -0,0 +1,891 @@
+
+msg = [
+"""
+end-user
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`,
+what existing methods are mentioned for the problem of formation maintenance?
+""",
+
+"""
+commander
+───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+You are the central commander, the brain of a coordinated multi-agent system. Your job is to make plans to solve a task given by
+an end-user:
+
+In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`,
+what existing methods are mentioned for the problem of formation maintenance?
+
+
+
+However, you should not solve this task directly, instead, you should decompose the task into sub-tasks, and present them in a
+markdown to-do list format. As an example, here is a to-do list for user question to "Get the GDP of the country/countries with
+highest rare earth production in 2024":
+
+ ... analysis of the task ...
+
+
+ - [ ] Get the basic concept of what rare earth is, and whether it has direct relationship with GDP
+ - [ ] Get the rare earth production of each country in 2024, select the country with highest rare earth production
+ - [ ] Get the GDP of that country in 2024
+
+
+ Get the GDP of that country in 2024
+
+
+[IMPORTANT] The to-do list should be clear, short and simple. The number of to-be-completed sub-tasks should be no more than 3.
+
+[IMPORTANT] Always use English to answer the question.
+
+[IMPORTANT] There are 3 cases:
+
+
+for this case, the output should be:
+
+
+ ... 分析问题 ...
+
+
+ ...
+
+""",
+
+"""
+assistant
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+Okay, so the user wants to know what existing methods are mentioned in the paper "Formation control with collision avoidance
+through deep reinforcement learning using model-guided demonstration" for formation maintenance. Let me break this down.
+
+First, I need to understand the question. The user is asking about existing methods for formation maintenance as mentioned in that
+specific paper. My task isn't to answer it directly but to decompose it into subtasks. The example given uses a to-do list with up
+to three steps, so I should follow that structure.
+
+The first step would be to identify the key concepts. Formation maintenance is the problem area, and the paper is proposing a
+method using deep reinforcement learning with model-guided demonstrations. But the user wants existing methods mentioned in the
+paper, not the proposed one. So the paper probably reviews related work or existing approaches before introducing their own method
+.
+
+So subtask one: Get the basic concept of formation maintenance to understand what the problem entails. That makes sense because
+knowing the problem helps in identifying the methods.
+
+Subtask two: Access the paper and locate the sections where existing methods are discussed. Typically, in research papers, there's
+a related work or literature review section that discusses existing approaches. The user needs to find those sections to extract
+the mentioned methods.
+
+Third subtask: Extract and list the existing methods from the paper. Once the relevant sections are located, the next step is to
+compile the list of methods they reference.
+
+I need to make sure the steps are clear and simple. The user might not have access to the paper, so maybe accessing it is part of
+the steps. But if the assistant can't access external documents, how would it proceed? Wait, the example given in the problem
+includes steps like getting GDP data which might require external sources. But in this case, the user is asking about a specific
+paper, so the assumption here is that the assistant has access to the paper or can retrieve the information from it. However, in
+reality, I don't have access to the paper. But since this is a hypothetical scenario for decomposition, I can outline the steps as
+if the assistant can access the paper.
+
+So the to-do list would be:
+
+1. Understand formation maintenance basics.
+2. Access the paper and find sections discussing existing methods.
+3. Extract the listed existing methods.
+
+That's three steps, which fits the requirement. The next step would be the last one, extracting the methods. But the user just
+needs the decomposition, not the execution. So the answer should present this structure.
+
+
+ The task requires identifying existing methods for formation maintenance as referenced in the specified paper. This involves
+understanding the problem domain, locating the relevant sections in the paper, and extracting the cited methods.
+
+
+ - [ ] Understand the concept of formation maintenance to contextualize the problem
+ - [ ] Access the paper and locate its literature review/related work section discussing existing methods
+ - [ ] Extract and list the existing methods mentioned in the paper for formation maintenance
+
+""",
+
+"""
+commander
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+You're a helpful assistant named tool_exe_agent.
+
+## What You Should Do:
+Look back at the previous problem-solving steps and decide which external tool to use.
+You should only focus on the very next unfinished step on the todo list. Do not look forward!
+
+## Note:
+1. Do not make any assumptions about the tool call result, keep your thought and analysis short and concise.
+2. Forget everything you know, only information given by the tools are reliable.
+3. Keep your thought short and concise.
+4. Do not make up fake information.
+5. If you need to write python code, please use the `CodeExecutionTool` to execute it. And the `code` argument should use loguru
+to print the result. For example:
+```python
+from loguru import logger
+from ... (other imports) ...
+
+def you_function_name():
+ # your code
+
+if __name__ == "__main__":
+ try:
+ logger.success(you_function_name())
+ except Exception as e:
+ logger.exception(e)
+```
+6. If there are tool-call histories, consider their results, do not repeat failure over and over again.
+
+
+7. Respect to todo list! Respect to todo list! Respect to todo list! if there is a todo list, you must follow it.
+For a example todo list:
+
+ - [x] job 1 (`x` means completed)
+ - [ ] job 2
+ - [ ] job 3
+
+You must focus on job 2 only.
+8. When you use chrome related tools, do not ask for any confirmation, just ask for materials.
+9. field should be a JSON object. For example: {"name": "RealChromeBrowserUse", "arguments": {"task": "在
+小红书中搜索杭州一日游攻略。"}}
+
+## Tool names and arguments:
+[{"type": "function", "function": {"name": "CodeExecutionTool", "description": "用于执行Python代码的工具,可以返回执行结果、打印
+结果和错误Traceback。调用时必须使用loguru打印信息。", "parameters": {"properties": {"tool_name": {"default": "CodeExecutionTool",
+"description": "用于执行Python代码的工具,可以返回执行结果、打印结果和错误Traceback。调用时必须使用loguru打印信息。", "title": "
+Tool Name", "type": "string"}, "code": {"default": "", "description": "要执行的Python代码", "title": "Code", "type": "string"}}, "
+title": "CodeExecutionTool", "type": "object"}}}, {"type": "function", "function": {"name": "RealChromeBrowserUse", "description":
+"通过真正的Chrome浏览器搜索网页,效率略低,但更稳定,可以访问更广泛的网站。", "parameters": {"properties": {"tool_name": {"default
+": "RealChromeBrowserUse", "description": "通过真正的Chrome浏览器搜索网页,效率略低,但更稳定,可以访问更广泛的网站。", "title": "
+Tool Name", "type": "string"}, "task": {"default": "", "description": "一句话,描述一个明确的子任务(task)。例如:'获取2024年5月1
+日阿里巴巴集团控股有限公司的收盘价'。", "title": "Task", "type": "string"}, "startup_url": {"anyOf": [{"type": "string"}, {"type":
+"null"}], "default": "", "description": "起始url(非必要参数,默认留空)。可以指定一个url作为任务的起点,从而提高效率。", "title":
+"Startup Url"}, "additional_tip": {"anyOf": [{"type": "string"}, {"type": "null"}], "default": "", "description": "执行浏览器任务
+时的一些建议(非必要参数,默认留空)。", "title": "Additional Tip"}}, "title": "RealChromeBrowserUse", "type": "object"}}}, {"type
+": "function", "function": {"name": "GetDownloadedFiles", "description": "获取下载完的文件。", "parameters": {"properties": {"tool
+_name": {"default": "GetDownloadedFiles", "description": "获取下载完的文件。", "title": "Tool Name", "type": "string"}, "file_path
+": {"default": "", "description": "文件的路径。", "title": "File Path", "type": "string"}}, "title": "GetDownloadedFiles", "type":
+"object"}}}]
+
+## Answer Format:
+
+ ... analyze which tool should be used ...
+
+
+
+ ... next objective from the todo list ...
+
+
+
+ {"name": ..., "arguments": ...}
+
+""",
+
+"""
+tool-agent
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+
+ The next step is to access the paper and locate sections discussing existing methods. Since the paper is not provided directly
+, an external tool like a browser can be used to search for and access the paper.
+
+
+
+ Access the paper and find sections discussing existing methods.
+
+
+
+ {"name": "RealChromeBrowserUse", "arguments": {"task": "Search for the paper 'Formation control with collision avoidance
+through deep reinforcement learning using model-guided demonstration' and locate sections discussing existing methods."}}
+[{'name': 'RealChromeBrowserUse', 'arguments': {'task': "Search for the paper 'Formation control with
+collision avoidance through deep reinforcement learning using model-guided demonstration' and locate sections discussing existing
+methods."}}]
+""",
+
+"""
+tool
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+# The selected tools
+
+{'name': 'RealChromeBrowserUse', 'arguments': {'task': "Search for the paper 'Formation control with collision avoidance through
+deep reinforcement learning using model-guided demonstration' and locate sections discussing existing methods.", 'mcp_kwargs': {}}
+}
+
+# The toolcall result
+
+[field 'model_output']
+ [field 'current_state']
+ [field 'evaluation_previous_goal']
+ Success - Successfully extracted content from the IEEE page which reveals discussion of existing methods
+ [field 'memory']
+ Found key information about existing methods in the abstract and introduction. The paper mentions that 'most of the existing
+works are simple combinations of methods dealing with the two problems separately' - referring to formation maintenance and
+collision avoidance. The introduction section provides context about multiagent systems and formation control challenges. However,
+access to the full paper requires a subscription. Step 13/16.
+ [field 'next_goal']
+ Since we found the key discussion about existing methods but cannot access the full text due to subscription requirement, we
+should summarize our findings and complete the task
+ [field 'action']
+ [field 'done']
+ [field 'text']
+ I found the paper and located discussions about existing methods. From the accessible content, the key points about
+existing methods are:
+
+ 1. Most existing works handle formation maintenance and collision avoidance as separate problems, simply combining
+different methods for each issue.
+
+ 2. The paper identifies this as a limitation, as the followers need to handle both formation maintenance and collision
+avoidance simultaneously.
+
+ 3. The introduction provides context that formation control is a fundamental problem in multiagent systems, where both
+formation maintenance and collision avoidance are essential considerations.
+
+ 4. The authors propose their new method specifically to address the limitations of existing approaches by using deep
+reinforcement learning to handle both aspects in an integrated way.
+
+ While more detailed discussion of existing methods likely exists in the full paper, access requires a subscription. The
+abstract and introduction sections provide the key critique that existing methods typically separate the two problems rather than
+addressing them in an integrated way.
+ [field 'success']
+ True
+[field 'result']
+ [field 'extracted_content']
+ I found the paper and located discussions about existing methods. From the accessible content, the key points about existing
+methods are:
+
+ 1. Most existing works handle formation maintenance and collision avoidance as separate problems, simply combining different
+methods for each issue.
+
+ 2. The paper identifies this as a limitation, as the followers need to handle both formation maintenance and collision
+avoidance simultaneously.
+
+ 3. The introduction provides context that formation control is a fundamental problem in multiagent systems, where both
+formation maintenance and collision avoidance are essential considerations.
+
+ 4. The authors propose their new method specifically to address the limitations of existing approaches by using deep
+reinforcement learning to handle both aspects in an integrated way.
+
+ While more detailed discussion of existing methods likely exists in the full paper, access requires a subscription. The
+abstract and introduction sections provide the key critique that existing methods typically separate the two problems rather than
+addressing them in an integrated way.
+[field 'state']
+ [field 'url']
+ https://ieeexplore.ieee.org/abstract/document/9142429
+ [field 'title']
+ Formation Control With Collision Avoidance Through Deep Reinforcement Learning Using Model-Guided Demonstration | IEEE
+Journals & Magazine | IEEE Xplore
+[field 'page_meta']
+ [field 'content']
+ Opens in a new window Opens an external website Opens an external website in a new window
+
+ Close this dialog
+
+ This website utilizes technologies such as cookies to enable essential site functionality, as well as for analytics,
+personalization, and targeted advertising. To learn more, view the following link: Privacy Policy
+
+
+
+ Close Cookie Preferences
+
+
+
+
+
+
+
+
+
+
+
+ Formation Control With Collision Avoidance Through Deep Reinforcement Learning Using Model-Guided Demonstration | IEEE
+Journals & Magazine | IEEE Xplore
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ Skip to Main Content
+
+ * IEEE.org
+ * IEEE *Xplore*
+ * IEEE SA
+ * IEEE Spectrum
+ * More Sites
+
+ Subscribe
+
+ * + Donate
+
+ + Cart
+
+ + Create Account
+ + Personal Sign In
+
+ * Browse
+ * My Settings
+ * Help
+
+ Institutional Sign In
+
+ Institutional Sign In
+
+ AllBooksConferencesCoursesJournals & MagazinesStandardsAuthorsCitations
+
+ ADVANCED SEARCH
+
+ Journals & Magazines >IEEE Transactions on Neural N... >Volume: 32 Issue: 6
+
+ Formation Control With Collision Avoidance Through Deep Reinforcement Learning Using Model-Guided Demonstration
+ ===============================================================================================================
+
+ Publisher: IEEE
+
+ Cite This
+
+ PDF
+
+ Zezhi Sui; Zhiqiang Pu; Jianqiang Yi; Shiguang Wu
+
+ All Authors
+
+ Sign In or Purchase
+
+ 78
+
+ Cites in
+
+ Papers
+
+ 4925
+
+ Full
+
+ Text Views
+
+ * Alerts
+
+ Alerts
+ ======
+
+ Manage Content Alerts
+
+ Add to Citation Alerts
+
+ ---
+
+ Abstract
+
+ Document Sections
+ -----------------
+
+ * I.
+
+ Introduction
+ * II.
+
+ Preliminaries
+ * III.
+
+ Problem Formulation
+ * IV.
+
+ Approach
+ * V.
+
+ Simulation and Experiment Results
+
+ Show Full Outline
+
+ Authors
+
+ Figures
+
+ References
+
+ Citations
+
+ Keywords
+
+ Metrics
+
+ More Like This
+
+ * Download PDF
+ * Download References
+ * Request Permissions
+ * Save to
+ * Alerts
+
+ Abstract:
+ ---------
+
+ Generating collision-free, time-efficient paths in an uncertain dynamic environment poses huge challenges for the formation
+control with collision avoidance (FCCA) proble...Show More
+
+ Metadata
+ --------
+
+ Abstract:
+ ---------
+
+ Generating collision-free, time-efficient paths in an uncertain dynamic environment poses huge challenges for the formation
+control with collision avoidance (FCCA) problem in a leader-follower structure. In particular, the followers have to take both
+formation maintenance and collision avoidance into account simultaneously. Unfortunately, most of the existing works are simple
+combinations of methods dealing with the two problems separately. In this article, a new method based on deep reinforcement
+learning (RL) is proposed to solve the problem of FCCA. Especially, the learning-based policy is extended to the field of
+formation control, which involves a two-stage training framework: an imitation learning (IL) and later an RL. In the IL stage, a
+model-guided method consisting of a consensus theory-based formation controller and an optimal reciprocal collision avoidance
+strategy is designed to speed up training and increase efficiency. In the RL stage, a compound reward function is presented to
+guide the training. In addition, we design a formation-oriented network structure to perceive the environment. Long short-term
+memory is adopted to enable the network structure to perceive the information of obstacles of an uncertain number, and a transfer
+training approach is adopted to improve the generalization of the network in different scenarios. Numerous representative
+simulations are conducted, and our method is further deployed to an experimental platform based on a multiomnidirectional-wheeled
+car system. The effectiveness and practicability of our proposed method are validated through both the simulation and experiment
+results.
+
+ **Published in:** IEEE Transactions on Neural Networks and Learning Systems ( Volume: 32, Issue: 6, June 2021)
+
+ **Page(s):** 2358 - 2372
+
+ **Date of Publication:** 16 July 2020
+
+ ISSN Information:
+ -----------------
+
+ **PubMed ID:** 32673195
+
+ **DOI:** 10.1109/TNNLS.2020.3004893
+
+ Publisher: IEEE
+
+ Funding Agency:
+ ---------------
+
+ Contents
+
+ ---
+
+ ### I. Introduction
+
+ Multiagent systems have received increasing attention from researchers in recent years because of their great potential in a
+variety of fields. Their applications can be found in collaborative explorations of monitoring and rescue, cooperative control of
+satellite clusters, and formation control of unmanned aerial vehicles [1]–[3], and so on. The basic concept of multiagent systems
+is to use individuals to cooperatively solve complex tasks that cannot be accomplished by a single agent even with expensive
+equipment. Formation control is one fundamental problem for multiagent systems, of which the objective is to achieve and maintain
+a certain formation shape so that the multiagent system could collectively accomplish a specific task. It is obvious that
+formation maintenance is an essential problem in formation control. In addition, collision avoidance should also be taken into
+consideration to guarantee the safety of the multiagent system. Due to the interaction among agents and the tradeoff between
+collision avoidance and formation maintenance, finding collision-free, time-efficient paths in an uncertain dynamic environment
+remains challenging.
+
+ Sign in to Continue Reading
+
+ Authors
+ -------
+
+ Figures
+ -------
+
+ References
+ ----------
+
+ Citations
+ ---------
+
+ Keywords
+ --------
+
+ Metrics
+ -------
+
+ More Like This
+
+ Collision Avoidance and Connectivity Preservation for Time-Varying Formation of Second-Order Multi-Agent Systems With a
+Dynamic Leader
+
+ IEEE Access
+
+ Published: 2022
+
+ Multi-Agent System Based on Machine Learning for Early Diagnosis of Diabetes
+
+ 2020 IEEE 6th International Conference on Optimization and Applications (ICOA)
+
+ Published: 2020
+
+ Show More
+
+ References
+ ==========
+
+ **References is not available for this document.**
+
+ IEEE Personal Account
+ ---------------------
+
+ * Change username/password
+
+ Purchase Details
+ ----------------
+
+ * Payment Options
+ * View Purchased Documents
+
+ Profile Information
+ -------------------
+
+ * Communications Preferences
+ * Profession and Education
+ * Technical interests
+
+ Need Help?
+ ----------
+
+ * US & Canada: +1 800 678 4333
+ * Worldwide: +1 732 981 0060
+ * Contact & Support
+
+ Follow
+ ------
+
+ About IEEE *Xplore* | Contact Us | Help | Accessibility | Terms of Use | Nondiscrimination Policy | IEEE Ethics Reporting |
+Sitemap | IEEE Privacy Policy
+
+ A public charity, IEEE is the world's largest technical professional organization dedicated to advancing technology for the
+benefit of humanity.
+
+ © Copyright 2025 IEEE - All rights reserved, including rights for text and data mining and training of artificial intelligence
+and similar technologies.
+
+ ### IEEE Account
+
+ * Change Username/Password
+ * Update Address
+
+ ### Purchase Details
+
+ * Payment Options
+ * Order History
+ * View Purchased Documents
+
+ ### Profile Information
+
+ * Communications Preferences
+ * Profession and Education
+ * Technical Interests
+
+ ### Need Help?
+
+ * **US & Canada:** +1 800 678 4333
+ * **Worldwide:** +1 732 981 0060
+ * Contact & Support
+
+ * About IEEE *Xplore*
+ * Contact Us
+ * Help
+ * Accessibility
+ * Terms of Use
+ * Nondiscrimination Policy
+ * Sitemap
+ * Privacy & Opting Out of Cookies
+
+ A not-for-profit organization, IEEE is the world's largest technical professional organization dedicated to advancing
+technology for the benefit of humanity.
+ © Copyright 2025 IEEE - All rights reserved. Use of this web site signifies your agreement to the terms and conditions.
+""",
+
+"""
+commander
+───────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+Follow the plan and try to solve the task step by step according to the plan and tool execution results.
+Your primary job is analyze the tool execution, do not call any external tools this time.
+
+[IMPORTANT] Always use English to answer the question.
+[IMPORTANT] Do not call any external tools this time, analyze only.
+""",
+
+"""
+assistant
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+Okay, let's tackle this step by step. The user wants to know the existing methods mentioned in the paper for formation maintenance
+. From the tool execution results, I accessed the paper's abstract and introduction via the IEEE link.
+
+In the abstract, it says, "most of the existing works are simple combinations of methods dealing with the two problems separately.
+" The two problems here are formation maintenance and collision avoidance. So existing methods typically handle these two aspects
+independently and then combine them.
+
+The introduction further explains that formation control is a fundamental problem in multiagent systems, where both formation
+maintenance and collision avoidance are essential. The authors point out that existing approaches don't integrate these two
+aspects effectively. They mention that followers need to consider both simultaneously, which current methods don't address
+properly.
+
+Additionally, the paper's approach section refers to a model-guided method using consensus theory-based formation controllers and
+optimal reciprocal collision avoidance strategies during the imitation learning stage. These are likely the existing techniques
+they build upon.
+
+So putting this together, the existing methods mentioned are those that separate formation maintenance and collision avoidance,
+using consensus theory for formation control and reciprocal collision avoidance strategies, then combining them. The paper
+critiques this approach as insufficient and proposes an integrated solution using deep reinforcement learning.
+
+Based on the information extracted from the paper's abstract and introduction, the existing methods for formation maintenance
+mentioned in the paper can be summarized as follows:
+
+1. **Decoupled Approaches**: Most existing works treat formation maintenance and collision avoidance as separate problems,
+combining solutions for each in a non-integrated manner. This means formation control and collision avoidance are addressed
+individually and then merged, rather than being optimized together.
+
+2. **Consensus Theory-Based Formation Controllers**: The paper references traditional methods that use consensus algorithms (a
+common approach in multiagent systems) to achieve formation maintenance. These controllers rely on mathematical frameworks to
+ensure agents reach and maintain a desired formation shape.
+
+3. **Optimal Reciprocal Collision Avoidance (ORCA)**: The paper mentions ORCA as a strategy used in existing methods for collision
+avoidance. ORCA is a well-known algorithm that ensures collision-free paths by calculating velocity obstacles for each agent to
+avoid collisions with others.
+
+4. **Model-Guided Demonstration Frameworks**: While the paper's proposed method uses a two-stage training framework (imitation
+learning followed by reinforcement learning), it builds on prior work that employs model-guided demonstrations. These
+demonstrations often rely on predefined controllers (like consensus or ORCA) to generate training data for learning-based systems.
+
+### Key Critique in the Paper:
+The authors highlight that these existing methods suffer from limitations such as:
+- **Suboptimal Integration**: Decoupling formation maintenance and collision avoidance leads to inefficiencies, as agents must
+balance both tasks simultaneously in dynamic environments.
+- **Over-Reliance on Predefined Models**: Traditional approaches (e.g., consensus theory or ORCA) may struggle in uncertain or
+complex environments where adaptability is required.
+
+The paper proposes a novel deep reinforcement learning (DRL) framework to address these limitations by integrating both tasks into
+a unified policy, moving beyond the decoupled and model-dependent strategies of prior work.
+""",
+
+"""
+commander
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+You are the central commander, the brain of a coordinated multi-agent system. Your job is to make plans to solve a task given by
+an end-user:
+
+In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`,
+what existing methods are mentioned for the problem of formation maintenance?
+
+
+
+However, you should not solve this task directly, instead, you should decompose the task into sub-tasks, and present them in a
+markdown to-do list format. As an example, here is a to-do list for user question to "Get the GDP of the country/countries with
+highest rare earth production in 2024":
+
+ ... analysis of the task ...
+
+
+ - [ ] Get the basic concept of what rare earth is, and whether it has direct relationship with GDP
+ - [ ] Get the rare earth production of each country in 2024, select the country with highest rare earth production
+ - [ ] Get the GDP of that country in 2024
+
+
+ Get the GDP of that country in 2024
+
+
+[IMPORTANT] The to-do list should be clear, short and simple. The number of to-be-completed sub-tasks should be no more than 3.
+
+[IMPORTANT] Always use English to answer the question.
+
+[IMPORTANT] There are 3 cases:
+
+===
+ 1. If there are no previous to-do list in the conversation, write a new one.
+ for this case, the output should be:
+
+
+ ... we are dealing with the first case (as this is the first time that the problem is raised), now analyze the task ...
+
+
+ ...
+
+
+===
+ 2. if there are already previous to-do list in the conversation, you should re-write the to-do list again, investigate what
+tasks have ALREADY been done, remove tasks that is IMPOSSIBLE to be done and found alternative solutions whenever needed.
+ for this case, the output should be:
+
+
+ ... we are dealing with the 2rd case, analyze the sub-tasks we have already completed, and write a new plan based on
+current situation ...
+
+
+ ...
+
+
+ ...
+
+
+===
+ 3. If the task is already completed, you should return a "TASK_COMPLETED" signal, and do not write any to-do list.
+ for this case, the output should be:
+
+
+ ... according to previous conversations, we have already obtained a possible answer, next, we verify the result ...
+
+ TASK_COMPLETED
+""",
+
+"""
+commander
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+Now, determine whether the task is completed. If the task is already completed, say `TASK_COMPLETED`, otherwise, update the to-do
+list.
+Note that if you need data in the analysis, you have to point out that the data is just a assumption.
+""",
+
+"""
+assistant
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+Okay, let me check if the task is completed. The user asked about existing methods mentioned in the paper for formation
+maintenance.
+
+From the tool execution, I accessed the paper's abstract and introduction. The abstract states that existing works combine methods
+for formation maintenance and collision avoidance separately. The introduction mentions consensus theory-based controllers and
+ORCA as part of existing approaches. Also, the paper's approach section refers to model-guided demonstrations using those
+traditional methods.
+
+So, the existing methods identified are:
+1. Decoupled approaches handling formation and collision avoidance separately.
+2. Consensus theory-based formation controllers.
+3. Optimal Reciprocal Collision Avoidance (ORCA).
+4. Model-guided demonstration frameworks relying on predefined controllers.
+
+The analysis seems thorough based on the accessible parts of the paper. The full text might have more details, but the key
+existing methods are covered in the abstract and introduction. Since the user's question is about what existing methods are
+mentioned, and we've extracted those from the available content, the task should be completed.
+
+
+ Based on the information extracted from the paper's abstract and introduction sections (accessible without subscription), we
+have identified the existing methods mentioned in the paper for formation maintenance:
+
+ 1. **Decoupled Approaches**: Existing works treat formation maintenance and collision avoidance as separate problems,
+combining solutions for each independently.
+ 2. **Consensus Theory-Based Formation Controllers**: Traditional methods using consensus algorithms to achieve formation
+maintenance.
+ 3. **Optimal Reciprocal Collision Avoidance (ORCA)**: A strategy used in existing methods for collision avoidance.
+ 4. **Model-Guided Demonstration Frameworks**: Prior work employing predefined controllers (e.g., consensus or ORCA) to
+generate training data for learning-based systems.
+
+ While the full paper might contain additional details, the accessible content provides sufficient evidence to answer the user'
+s question about the *mentioned* existing methods. Further access to the full text would not invalidate these findings but could
+only add supplementary context.
+
+
+TASK_COMPLETED
+""",
+
+"""
+end-user
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`,
+what existing methods are mentioned for the problem of formation maintenance?
+""",
+
+"""
+commander
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+As you can see, the former agents have done a lot of work.
+Now please make a final answer of the original task based on our conversation.
+
+Please pay special attention to the format in which the answer is presented.
+You should first analyze the answer format required by the question and then output the final answer that meets the format
+requirements.
+Your response should include the following content:
+- `final_answer`: enclosed by , the final answer to the question.
+Here are some hint about the final answer:
+
+Your final answer must be output exactly in the format specified by the question. It should be a number OR as few words as
+possible OR a comma separated list of numbers and/or strings:
+- If you are asked for a number, don't use comma to write your number neither use units such as $ or percent sign unless specified
+otherwise.
+- If you are asked for a string, don't use articles, neither abbreviations (e.g. for cities), and write the digits in plain text
+unless specified otherwise.
+- If you are asked for a comma separated list, apply the above rules depending of whether the element to be put in the list is a
+number or a string.
+
+""",
+
+"""
+assistant
+──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────
+Okay, let me try to figure out how to structure the final answer based on the user's requirements.
+
+The original question asked what existing methods are mentioned in the paper for formation maintenance. From the previous steps,
+we identified four main methods: decoupled approaches, consensus theory-based controllers, ORCA, and model-guided demonstrations.
+However, the user wants the answer in a specific format—either a number, a short string, or a comma-separated list.
+
+Looking back at the hint provided, if it's a list, each element should follow the rules (no articles, no abbreviations unless
+specified). The answer should be concise. The paper's abstract and introduction mention that existing works are simple
+combinations of methods dealing with the two problems (formation maintenance and collision avoidance) separately. Then, the
+approach section refers to using consensus theory and ORCA in the model-guided phase.
+
+But the key point is that the existing methods are decoupled approaches, consensus-based formation control, and ORCA. The model-
+guided demonstrations are part of the proposed method, not the existing ones. Wait, in the analysis, point 4 was model-guided
+demonstration frameworks as existing methods. But in the paper's approach, they use a two-stage training framework starting with
+imitation learning (which uses model-guided methods). So the existing methods include consensus and ORCA, which are part of the
+model-guided demonstration.
+
+So the answer should list the existing methods as:
+1. Decoupled approaches combining separate methods for formation and collision avoidance.
+2. Consensus theory-based formation controllers.
+3. Optimal Reciprocal Collision Avoidance (ORCA).
+
+But the user wants the answer in a comma-separated list with as few words as possible. So perhaps abbreviate ORCA but spell out
+the others. However, the hint says no abbreviations unless specified. Wait, ORCA is an abbreviation, but in the paper, it's
+referred to as Optimal Reciprocal Collision Avoidance (ORCA). Since the paper uses the acronym, maybe it's acceptable.
+Alternatively, use the full name. But the example answer in the hint says "don't use abbreviations (e.g. for cities)", so maybe
+avoid ORCA and write out Optimal Reciprocal Collision Avoidance. However, in the context of academic papers, ORCA is a standard
+term, so using the acronym might be acceptable. Need to check. Since the paper mentions ORCA in the abstract (from the tool result
+), it's okay to use the acronym here.
+
+So the final answer would be a list of three items:
+Decoupled approaches,Consensus theory-based formation controllers,Optimal Reciprocal Collision Avoidance (ORCA)
+
+Wait, but the user said "as few words as possible" or "comma separated list of numbers and/or strings". So maybe shorten them:
+Decoupled approaches,Consensus-based controllers,ORCA
+
+Yes, that's more concise. The paper refers to "consensus theory-based formation controller" which can be shortened to "consensus-
+based controllers". ORCA is acceptable as an acronym. So the final answer is a comma-separated list of these three methods.
+
+Decoupled approaches,Consensus-based controllers,Optimal Reciprocal Collision Avoidance
+"""
+
+]
+
+from experiencemaker.module.summarizer.traj_summarizer import TrajectorySummarizer
+from experiencemaker.schema.trajectory import Trajectory, Sample, SummaryMessage, Message
+from experiencemaker.schema.trajectory import Message as BMessage
+
+class Message(BMessage):
+ executor: str = None
+
+ts = TrajectorySummarizer()
+role_mapping = {
+ 'system': 'system',
+ 'end-user': 'user',
+ 'commander': 'user',
+ 'assistant': 'assistant',
+ 'tool-agent': 'user',
+ 'tool': 'tool',
+}
+msg_f = []
+for m in msg:
+ role = m.split('\n')[1].strip()
+ content = '\n'.join(m.split('\n')[3:]).strip()
+ msg_f += [Message(role=role_mapping[role], content=content)]
+traj = Trajectory(steps=msg_f)
+ts.execute(trajectories=[traj], return_samples=True)