From 0077050050609d3234b9ad3b68f878bf45a54436 Mon Sep 17 00:00:00 2001 From: "jinli.yl" Date: Tue, 17 Jun 2025 20:09:06 +0800 Subject: [PATCH] update readme --- README.md | 102 +++++-- test.py | 891 ------------------------------------------------------ 2 files changed, 70 insertions(+), 923 deletions(-) delete mode 100644 test.py diff --git a/README.md b/README.md index ef9fd418..294b8b08 100644 --- a/README.md +++ b/README.md @@ -1,51 +1,89 @@ -# run service +# Quick Start +## Hello Experience Maker +Here is a simple user guide for ExperienceMaker. + +### Step0: Preparation Work + +#### Prepare LLM & EMBEDDING_MODEL +We need to prepare the API services for the LLM and the Embedding model. +Since we are using an OpenAI-compatible service, we only need to write the `OPENAI_API_KEY` and `OPENAI_BASE_URL` into the environment. ```shell -cd BeyondAgent -python beyondagent/core/service/model_service.py +export OPENAI_API_KEY="sk-xxx" +export OPENAI_BASE_URL="xxx" ``` -# test service +#### Prepare Vector Store +If you want to use vector store, you need to set up a vector database. Don't forget to set up the `ES_HOSTS`. +- Elasticsearch [quick start](../vector_store/elasticsearch.md) -```shell -python beyondagent/test/test_service.py +#### Prepare Your Own Agent +Assume you have a runnable agent. +Here, we use a basic LLM combined with a simple react framework including three tools(code, web_search, terminate) as an example. +```python +class YourOwnAgent(...): + ... + def think(self, **kwargs) -> bool: + ... + + def act(self, **kwargs): + ... + + def run(self, query: str, previous_experience: str): + ... ``` -# qingxu +Here is an [example code](./your_own_agent.py) with [prompt](./your_own_agent_prompt.yaml). To use this simple agent, you will need to set up `DASHSCOPE_API_KEY`. ```shell -# 1. edit query, port, vm etc -nano docker-compose.yml -# 2. run -docker compose down && docker compose build && docker compose up -# 3. then watch vm at http://localhost:16901 (default password is headless) +export DASHSCOPE_API_KEY="sk-xxx" ``` - -# vector store -If a vector database is involved, you will need an Elasticsearch environment. You can refer to the following steps: -- If you don’t have Docker installed, download and install [Docker Desktop](https://www.docker.com/products/docker-desktop) for your operating system. -- To set up [Elasticsearch](https://www.elastic.co/docs/solutions/search/run-elasticsearch-locally) and Kibana locally, run the start-local script in the command line: +### Step1: Start ExperienceMaker Http Service +- Install dependencies. ```shell -curl -fsSL https://elastic.co/start-local | sh +pip install . ``` -Or manually download and load the image. Here, we take elasticsearch-wolfi:9.0.0 as an example: +- We start our context and summary services in `simple` mode. Next, you can use standard HTTP interfaces to call the services. ```shell -docker pull docker.elastic.co/elasticsearch/elasticsearch-wolfi:9.0.0 -docker run -p 9200:9200 \ - --memory='4GB' \ - -e "discovery.type=single-node" \ - -e "xpack.security.enabled=false" \ - -e "xpack.license.self_generated.type=trial" \ - docker.elastic.co/elasticsearch/elasticsearch-wolfi:9.0.0 +pip install . +python -m experiencemaker.em_service \ + --port=8001 \ + --llm='{"backend": "openai_compatible", "model_name": "qwen3-32b", "temperature": 0.6}' \ + --embedding_model='{"backend": "openai_compatible", "model_name": "text-embedding-v4", "dimensions": 1024}' \ + --vector_store='{"backend": "elasticsearch"}' \ + --context_generator='{"backend": "simple"}' \ + --summarizer='{"backend": "simple"}' ``` -# run module service +### Step2: Enhance Your Own Agent -[] 注册函数 +Call the capabilities of ContextGenerator and Summarizer through `EMClient`. -[] 更新workspace_id + index_name -[] trajectory 加 reward -[] summary去重,context加LLM eval -[] +```python +from experiencemaker.em_client import EMClient +em_client = EMClient(base_url="http://0.0.0.0:8001") +``` + +Assume you have a list of messages generated by an agent; you can use the summarizer to generate experiences from them. + +```python +request = SummarizerRequest( + trajectories=[Trajectory(query=query, steps=messages, answer=messages[-1].content, done=True)], + workspace_id="w_1234") +response = em_client.call_summarizer(request) +for experience in response.experiences: + print(experience.model_dump_json()) +``` + +Assume you have a query; you can use the context generator to generate a new query from the query and the related +experiences. + +```python +request = ContextGeneratorRequest(trajectory=Trajectory(query=query), retrieve_top_k=1, workspace_id="w_1234") +response = em_client.call_context_generator(request) +new_query = f"{response.context_msg.content}\n\nUser Question\n{query}" +``` + +The complete code can be found in the implementation of [YourOwnAgentEnhanced](./your_own_agent_enhanced.py). diff --git a/test.py b/test.py deleted file mode 100644 index 524f2093..00000000 --- a/test.py +++ /dev/null @@ -1,891 +0,0 @@ - -msg = [ -""" -end-user -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`, -what existing methods are mentioned for the problem of formation maintenance? -""", - -""" -commander -─────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -You are the central commander, the brain of a coordinated multi-agent system. Your job is to make plans to solve a task given by -an end-user: - -In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`, -what existing methods are mentioned for the problem of formation maintenance? - - - -However, you should not solve this task directly, instead, you should decompose the task into sub-tasks, and present them in a -markdown to-do list format. As an example, here is a to-do list for user question to "Get the GDP of the country/countries with -highest rare earth production in 2024": - - ... analysis of the task ... - - - - [ ] Get the basic concept of what rare earth is, and whether it has direct relationship with GDP - - [ ] Get the rare earth production of each country in 2024, select the country with highest rare earth production - - [ ] Get the GDP of that country in 2024 - - - Get the GDP of that country in 2024 - - -[IMPORTANT] The to-do list should be clear, short and simple. The number of to-be-completed sub-tasks should be no more than 3. - -[IMPORTANT] Always use English to answer the question. - -[IMPORTANT] There are 3 cases: - - -for this case, the output should be: - - - ... 分析问题 ... - - - ... - -""", - -""" -assistant -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -Okay, so the user wants to know what existing methods are mentioned in the paper "Formation control with collision avoidance -through deep reinforcement learning using model-guided demonstration" for formation maintenance. Let me break this down. - -First, I need to understand the question. The user is asking about existing methods for formation maintenance as mentioned in that -specific paper. My task isn't to answer it directly but to decompose it into subtasks. The example given uses a to-do list with up -to three steps, so I should follow that structure. - -The first step would be to identify the key concepts. Formation maintenance is the problem area, and the paper is proposing a -method using deep reinforcement learning with model-guided demonstrations. But the user wants existing methods mentioned in the -paper, not the proposed one. So the paper probably reviews related work or existing approaches before introducing their own method -. - -So subtask one: Get the basic concept of formation maintenance to understand what the problem entails. That makes sense because -knowing the problem helps in identifying the methods. - -Subtask two: Access the paper and locate the sections where existing methods are discussed. Typically, in research papers, there's -a related work or literature review section that discusses existing approaches. The user needs to find those sections to extract -the mentioned methods. - -Third subtask: Extract and list the existing methods from the paper. Once the relevant sections are located, the next step is to -compile the list of methods they reference. - -I need to make sure the steps are clear and simple. The user might not have access to the paper, so maybe accessing it is part of -the steps. But if the assistant can't access external documents, how would it proceed? Wait, the example given in the problem -includes steps like getting GDP data which might require external sources. But in this case, the user is asking about a specific -paper, so the assumption here is that the assistant has access to the paper or can retrieve the information from it. However, in -reality, I don't have access to the paper. But since this is a hypothetical scenario for decomposition, I can outline the steps as -if the assistant can access the paper. - -So the to-do list would be: - -1. Understand formation maintenance basics. -2. Access the paper and find sections discussing existing methods. -3. Extract the listed existing methods. - -That's three steps, which fits the requirement. The next step would be the last one, extracting the methods. But the user just -needs the decomposition, not the execution. So the answer should present this structure. - - - The task requires identifying existing methods for formation maintenance as referenced in the specified paper. This involves -understanding the problem domain, locating the relevant sections in the paper, and extracting the cited methods. - - - - [ ] Understand the concept of formation maintenance to contextualize the problem - - [ ] Access the paper and locate its literature review/related work section discussing existing methods - - [ ] Extract and list the existing methods mentioned in the paper for formation maintenance - -""", - -""" -commander -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -You're a helpful assistant named tool_exe_agent. - -## What You Should Do: -Look back at the previous problem-solving steps and decide which external tool to use. -You should only focus on the very next unfinished step on the todo list. Do not look forward! - -## Note: -1. Do not make any assumptions about the tool call result, keep your thought and analysis short and concise. -2. Forget everything you know, only information given by the tools are reliable. -3. Keep your thought short and concise. -4. Do not make up fake information. -5. If you need to write python code, please use the `CodeExecutionTool` to execute it. And the `code` argument should use loguru -to print the result. For example: -```python -from loguru import logger -from ... (other imports) ... - -def you_function_name(): - # your code - -if __name__ == "__main__": - try: - logger.success(you_function_name()) - except Exception as e: - logger.exception(e) -``` -6. If there are tool-call histories, consider their results, do not repeat failure over and over again. - - -7. Respect to todo list! Respect to todo list! Respect to todo list! if there is a todo list, you must follow it. -For a example todo list: - - - [x] job 1 (`x` means completed) - - [ ] job 2 - - [ ] job 3 - -You must focus on job 2 only. -8. When you use chrome related tools, do not ask for any confirmation, just ask for materials. -9. field should be a JSON object. For example: {"name": "RealChromeBrowserUse", "arguments": {"task": "在 -小红书中搜索杭州一日游攻略。"}} - -## Tool names and arguments: -[{"type": "function", "function": {"name": "CodeExecutionTool", "description": "用于执行Python代码的工具,可以返回执行结果、打印 -结果和错误Traceback。调用时必须使用loguru打印信息。", "parameters": {"properties": {"tool_name": {"default": "CodeExecutionTool", -"description": "用于执行Python代码的工具,可以返回执行结果、打印结果和错误Traceback。调用时必须使用loguru打印信息。", "title": " -Tool Name", "type": "string"}, "code": {"default": "", "description": "要执行的Python代码", "title": "Code", "type": "string"}}, " -title": "CodeExecutionTool", "type": "object"}}}, {"type": "function", "function": {"name": "RealChromeBrowserUse", "description": -"通过真正的Chrome浏览器搜索网页,效率略低,但更稳定,可以访问更广泛的网站。", "parameters": {"properties": {"tool_name": {"default -": "RealChromeBrowserUse", "description": "通过真正的Chrome浏览器搜索网页,效率略低,但更稳定,可以访问更广泛的网站。", "title": " -Tool Name", "type": "string"}, "task": {"default": "", "description": "一句话,描述一个明确的子任务(task)。例如:'获取2024年5月1 -日阿里巴巴集团控股有限公司的收盘价'。", "title": "Task", "type": "string"}, "startup_url": {"anyOf": [{"type": "string"}, {"type": -"null"}], "default": "", "description": "起始url(非必要参数,默认留空)。可以指定一个url作为任务的起点,从而提高效率。", "title": -"Startup Url"}, "additional_tip": {"anyOf": [{"type": "string"}, {"type": "null"}], "default": "", "description": "执行浏览器任务 -时的一些建议(非必要参数,默认留空)。", "title": "Additional Tip"}}, "title": "RealChromeBrowserUse", "type": "object"}}}, {"type -": "function", "function": {"name": "GetDownloadedFiles", "description": "获取下载完的文件。", "parameters": {"properties": {"tool -_name": {"default": "GetDownloadedFiles", "description": "获取下载完的文件。", "title": "Tool Name", "type": "string"}, "file_path -": {"default": "", "description": "文件的路径。", "title": "File Path", "type": "string"}}, "title": "GetDownloadedFiles", "type": -"object"}}}] - -## Answer Format: - - ... analyze which tool should be used ... - - - - ... next objective from the todo list ... - - - - {"name": ..., "arguments": ...} - -""", - -""" -tool-agent -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── - - The next step is to access the paper and locate sections discussing existing methods. Since the paper is not provided directly -, an external tool like a browser can be used to search for and access the paper. - - - - Access the paper and find sections discussing existing methods. - - - - {"name": "RealChromeBrowserUse", "arguments": {"task": "Search for the paper 'Formation control with collision avoidance -through deep reinforcement learning using model-guided demonstration' and locate sections discussing existing methods."}} -[{'name': 'RealChromeBrowserUse', 'arguments': {'task': "Search for the paper 'Formation control with -collision avoidance through deep reinforcement learning using model-guided demonstration' and locate sections discussing existing -methods."}}] -""", - -""" -tool -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -# The selected tools - -{'name': 'RealChromeBrowserUse', 'arguments': {'task': "Search for the paper 'Formation control with collision avoidance through -deep reinforcement learning using model-guided demonstration' and locate sections discussing existing methods.", 'mcp_kwargs': {}} -} - -# The toolcall result - -[field 'model_output'] - [field 'current_state'] - [field 'evaluation_previous_goal'] - Success - Successfully extracted content from the IEEE page which reveals discussion of existing methods - [field 'memory'] - Found key information about existing methods in the abstract and introduction. The paper mentions that 'most of the existing -works are simple combinations of methods dealing with the two problems separately' - referring to formation maintenance and -collision avoidance. The introduction section provides context about multiagent systems and formation control challenges. However, -access to the full paper requires a subscription. Step 13/16. - [field 'next_goal'] - Since we found the key discussion about existing methods but cannot access the full text due to subscription requirement, we -should summarize our findings and complete the task - [field 'action'] - [field 'done'] - [field 'text'] - I found the paper and located discussions about existing methods. From the accessible content, the key points about -existing methods are: - - 1. Most existing works handle formation maintenance and collision avoidance as separate problems, simply combining -different methods for each issue. - - 2. The paper identifies this as a limitation, as the followers need to handle both formation maintenance and collision -avoidance simultaneously. - - 3. The introduction provides context that formation control is a fundamental problem in multiagent systems, where both -formation maintenance and collision avoidance are essential considerations. - - 4. The authors propose their new method specifically to address the limitations of existing approaches by using deep -reinforcement learning to handle both aspects in an integrated way. - - While more detailed discussion of existing methods likely exists in the full paper, access requires a subscription. The -abstract and introduction sections provide the key critique that existing methods typically separate the two problems rather than -addressing them in an integrated way. - [field 'success'] - True -[field 'result'] - [field 'extracted_content'] - I found the paper and located discussions about existing methods. From the accessible content, the key points about existing -methods are: - - 1. Most existing works handle formation maintenance and collision avoidance as separate problems, simply combining different -methods for each issue. - - 2. The paper identifies this as a limitation, as the followers need to handle both formation maintenance and collision -avoidance simultaneously. - - 3. The introduction provides context that formation control is a fundamental problem in multiagent systems, where both -formation maintenance and collision avoidance are essential considerations. - - 4. The authors propose their new method specifically to address the limitations of existing approaches by using deep -reinforcement learning to handle both aspects in an integrated way. - - While more detailed discussion of existing methods likely exists in the full paper, access requires a subscription. The -abstract and introduction sections provide the key critique that existing methods typically separate the two problems rather than -addressing them in an integrated way. -[field 'state'] - [field 'url'] - https://ieeexplore.ieee.org/abstract/document/9142429 - [field 'title'] - Formation Control With Collision Avoidance Through Deep Reinforcement Learning Using Model-Guided Demonstration | IEEE -Journals & Magazine | IEEE Xplore -[field 'page_meta'] - [field 'content'] - Opens in a new window Opens an external website Opens an external website in a new window - - Close this dialog - - This website utilizes technologies such as cookies to enable essential site functionality, as well as for analytics, -personalization, and targeted advertising. To learn more, view the following link: Privacy Policy - - - - Close Cookie Preferences - - - - - - - - - - - - Formation Control With Collision Avoidance Through Deep Reinforcement Learning Using Model-Guided Demonstration | IEEE -Journals & Magazine | IEEE Xplore - - - - - - - - - - - - - - - - - - - - - - - Skip to Main Content - - * IEEE.org - * IEEE *Xplore* - * IEEE SA - * IEEE Spectrum - * More Sites - - Subscribe - - * + Donate - - + Cart - - + Create Account - + Personal Sign In - - * Browse - * My Settings - * Help - - Institutional Sign In - - Institutional Sign In - - AllBooksConferencesCoursesJournals & MagazinesStandardsAuthorsCitations - - ADVANCED SEARCH - - Journals & Magazines >IEEE Transactions on Neural N... >Volume: 32 Issue: 6 - - Formation Control With Collision Avoidance Through Deep Reinforcement Learning Using Model-Guided Demonstration - =============================================================================================================== - - Publisher: IEEE - - Cite This - - PDF - - Zezhi Sui; Zhiqiang Pu; Jianqiang Yi; Shiguang Wu - - All Authors - - Sign In or Purchase - - 78 - - Cites in - - Papers - - 4925 - - Full - - Text Views - - * Alerts - - Alerts - ====== - - Manage Content Alerts - - Add to Citation Alerts - - --- - - Abstract - - Document Sections - ----------------- - - * I. - - Introduction - * II. - - Preliminaries - * III. - - Problem Formulation - * IV. - - Approach - * V. - - Simulation and Experiment Results - - Show Full Outline - - Authors - - Figures - - References - - Citations - - Keywords - - Metrics - - More Like This - - * Download PDF - * Download References - * Request Permissions - * Save to - * Alerts - - Abstract: - --------- - - Generating collision-free, time-efficient paths in an uncertain dynamic environment poses huge challenges for the formation -control with collision avoidance (FCCA) proble...Show More - - Metadata - -------- - - Abstract: - --------- - - Generating collision-free, time-efficient paths in an uncertain dynamic environment poses huge challenges for the formation -control with collision avoidance (FCCA) problem in a leader-follower structure. In particular, the followers have to take both -formation maintenance and collision avoidance into account simultaneously. Unfortunately, most of the existing works are simple -combinations of methods dealing with the two problems separately. In this article, a new method based on deep reinforcement -learning (RL) is proposed to solve the problem of FCCA. Especially, the learning-based policy is extended to the field of -formation control, which involves a two-stage training framework: an imitation learning (IL) and later an RL. In the IL stage, a -model-guided method consisting of a consensus theory-based formation controller and an optimal reciprocal collision avoidance -strategy is designed to speed up training and increase efficiency. In the RL stage, a compound reward function is presented to -guide the training. In addition, we design a formation-oriented network structure to perceive the environment. Long short-term -memory is adopted to enable the network structure to perceive the information of obstacles of an uncertain number, and a transfer -training approach is adopted to improve the generalization of the network in different scenarios. Numerous representative -simulations are conducted, and our method is further deployed to an experimental platform based on a multiomnidirectional-wheeled -car system. The effectiveness and practicability of our proposed method are validated through both the simulation and experiment -results. - - **Published in:** IEEE Transactions on Neural Networks and Learning Systems ( Volume: 32, Issue: 6, June 2021) - - **Page(s):** 2358 - 2372 - - **Date of Publication:** 16 July 2020 - - ISSN Information: - ----------------- - - **PubMed ID:** 32673195 - - **DOI:** 10.1109/TNNLS.2020.3004893 - - Publisher: IEEE - - Funding Agency: - --------------- - - Contents - - --- - - ### I. Introduction - - Multiagent systems have received increasing attention from researchers in recent years because of their great potential in a -variety of fields. Their applications can be found in collaborative explorations of monitoring and rescue, cooperative control of -satellite clusters, and formation control of unmanned aerial vehicles [1]–[3], and so on. The basic concept of multiagent systems -is to use individuals to cooperatively solve complex tasks that cannot be accomplished by a single agent even with expensive -equipment. Formation control is one fundamental problem for multiagent systems, of which the objective is to achieve and maintain -a certain formation shape so that the multiagent system could collectively accomplish a specific task. It is obvious that -formation maintenance is an essential problem in formation control. In addition, collision avoidance should also be taken into -consideration to guarantee the safety of the multiagent system. Due to the interaction among agents and the tradeoff between -collision avoidance and formation maintenance, finding collision-free, time-efficient paths in an uncertain dynamic environment -remains challenging. - - Sign in to Continue Reading - - Authors - ------- - - Figures - ------- - - References - ---------- - - Citations - --------- - - Keywords - -------- - - Metrics - ------- - - More Like This - - Collision Avoidance and Connectivity Preservation for Time-Varying Formation of Second-Order Multi-Agent Systems With a -Dynamic Leader - - IEEE Access - - Published: 2022 - - Multi-Agent System Based on Machine Learning for Early Diagnosis of Diabetes - - 2020 IEEE 6th International Conference on Optimization and Applications (ICOA) - - Published: 2020 - - Show More - - References - ========== - - **References is not available for this document.** - - IEEE Personal Account - --------------------- - - * Change username/password - - Purchase Details - ---------------- - - * Payment Options - * View Purchased Documents - - Profile Information - ------------------- - - * Communications Preferences - * Profession and Education - * Technical interests - - Need Help? - ---------- - - * US & Canada: +1 800 678 4333 - * Worldwide: +1 732 981 0060 - * Contact & Support - - Follow - ------ - - About IEEE *Xplore* | Contact Us | Help | Accessibility | Terms of Use | Nondiscrimination Policy | IEEE Ethics Reporting | -Sitemap | IEEE Privacy Policy - - A public charity, IEEE is the world's largest technical professional organization dedicated to advancing technology for the -benefit of humanity. - - © Copyright 2025 IEEE - All rights reserved, including rights for text and data mining and training of artificial intelligence -and similar technologies. - - ### IEEE Account - - * Change Username/Password - * Update Address - - ### Purchase Details - - * Payment Options - * Order History - * View Purchased Documents - - ### Profile Information - - * Communications Preferences - * Profession and Education - * Technical Interests - - ### Need Help? - - * **US & Canada:** +1 800 678 4333 - * **Worldwide:** +1 732 981 0060 - * Contact & Support - - * About IEEE *Xplore* - * Contact Us - * Help - * Accessibility - * Terms of Use - * Nondiscrimination Policy - * Sitemap - * Privacy & Opting Out of Cookies - - A not-for-profit organization, IEEE is the world's largest technical professional organization dedicated to advancing -technology for the benefit of humanity. - © Copyright 2025 IEEE - All rights reserved. Use of this web site signifies your agreement to the terms and conditions. -""", - -""" -commander -─────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -Follow the plan and try to solve the task step by step according to the plan and tool execution results. -Your primary job is analyze the tool execution, do not call any external tools this time. - -[IMPORTANT] Always use English to answer the question. -[IMPORTANT] Do not call any external tools this time, analyze only. -""", - -""" -assistant -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -Okay, let's tackle this step by step. The user wants to know the existing methods mentioned in the paper for formation maintenance -. From the tool execution results, I accessed the paper's abstract and introduction via the IEEE link. - -In the abstract, it says, "most of the existing works are simple combinations of methods dealing with the two problems separately. -" The two problems here are formation maintenance and collision avoidance. So existing methods typically handle these two aspects -independently and then combine them. - -The introduction further explains that formation control is a fundamental problem in multiagent systems, where both formation -maintenance and collision avoidance are essential. The authors point out that existing approaches don't integrate these two -aspects effectively. They mention that followers need to consider both simultaneously, which current methods don't address -properly. - -Additionally, the paper's approach section refers to a model-guided method using consensus theory-based formation controllers and -optimal reciprocal collision avoidance strategies during the imitation learning stage. These are likely the existing techniques -they build upon. - -So putting this together, the existing methods mentioned are those that separate formation maintenance and collision avoidance, -using consensus theory for formation control and reciprocal collision avoidance strategies, then combining them. The paper -critiques this approach as insufficient and proposes an integrated solution using deep reinforcement learning. - -Based on the information extracted from the paper's abstract and introduction, the existing methods for formation maintenance -mentioned in the paper can be summarized as follows: - -1. **Decoupled Approaches**: Most existing works treat formation maintenance and collision avoidance as separate problems, -combining solutions for each in a non-integrated manner. This means formation control and collision avoidance are addressed -individually and then merged, rather than being optimized together. - -2. **Consensus Theory-Based Formation Controllers**: The paper references traditional methods that use consensus algorithms (a -common approach in multiagent systems) to achieve formation maintenance. These controllers rely on mathematical frameworks to -ensure agents reach and maintain a desired formation shape. - -3. **Optimal Reciprocal Collision Avoidance (ORCA)**: The paper mentions ORCA as a strategy used in existing methods for collision -avoidance. ORCA is a well-known algorithm that ensures collision-free paths by calculating velocity obstacles for each agent to -avoid collisions with others. - -4. **Model-Guided Demonstration Frameworks**: While the paper's proposed method uses a two-stage training framework (imitation -learning followed by reinforcement learning), it builds on prior work that employs model-guided demonstrations. These -demonstrations often rely on predefined controllers (like consensus or ORCA) to generate training data for learning-based systems. - -### Key Critique in the Paper: -The authors highlight that these existing methods suffer from limitations such as: -- **Suboptimal Integration**: Decoupling formation maintenance and collision avoidance leads to inefficiencies, as agents must -balance both tasks simultaneously in dynamic environments. -- **Over-Reliance on Predefined Models**: Traditional approaches (e.g., consensus theory or ORCA) may struggle in uncertain or -complex environments where adaptability is required. - -The paper proposes a novel deep reinforcement learning (DRL) framework to address these limitations by integrating both tasks into -a unified policy, moving beyond the decoupled and model-dependent strategies of prior work. -""", - -""" -commander -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -You are the central commander, the brain of a coordinated multi-agent system. Your job is to make plans to solve a task given by -an end-user: - -In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`, -what existing methods are mentioned for the problem of formation maintenance? - - - -However, you should not solve this task directly, instead, you should decompose the task into sub-tasks, and present them in a -markdown to-do list format. As an example, here is a to-do list for user question to "Get the GDP of the country/countries with -highest rare earth production in 2024": - - ... analysis of the task ... - - - - [ ] Get the basic concept of what rare earth is, and whether it has direct relationship with GDP - - [ ] Get the rare earth production of each country in 2024, select the country with highest rare earth production - - [ ] Get the GDP of that country in 2024 - - - Get the GDP of that country in 2024 - - -[IMPORTANT] The to-do list should be clear, short and simple. The number of to-be-completed sub-tasks should be no more than 3. - -[IMPORTANT] Always use English to answer the question. - -[IMPORTANT] There are 3 cases: - -=== - 1. If there are no previous to-do list in the conversation, write a new one. - for this case, the output should be: - - - ... we are dealing with the first case (as this is the first time that the problem is raised), now analyze the task ... - - - ... - - -=== - 2. if there are already previous to-do list in the conversation, you should re-write the to-do list again, investigate what -tasks have ALREADY been done, remove tasks that is IMPOSSIBLE to be done and found alternative solutions whenever needed. - for this case, the output should be: - - - ... we are dealing with the 2rd case, analyze the sub-tasks we have already completed, and write a new plan based on -current situation ... - - - ... - - - ... - - -=== - 3. If the task is already completed, you should return a "TASK_COMPLETED" signal, and do not write any to-do list. - for this case, the output should be: - - - ... according to previous conversations, we have already obtained a possible answer, next, we verify the result ... - - TASK_COMPLETED -""", - -""" -commander -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -Now, determine whether the task is completed. If the task is already completed, say `TASK_COMPLETED`, otherwise, update the to-do -list. -Note that if you need data in the analysis, you have to point out that the data is just a assumption. -""", - -""" -assistant -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -Okay, let me check if the task is completed. The user asked about existing methods mentioned in the paper for formation -maintenance. - -From the tool execution, I accessed the paper's abstract and introduction. The abstract states that existing works combine methods -for formation maintenance and collision avoidance separately. The introduction mentions consensus theory-based controllers and -ORCA as part of existing approaches. Also, the paper's approach section refers to model-guided demonstrations using those -traditional methods. - -So, the existing methods identified are: -1. Decoupled approaches handling formation and collision avoidance separately. -2. Consensus theory-based formation controllers. -3. Optimal Reciprocal Collision Avoidance (ORCA). -4. Model-guided demonstration frameworks relying on predefined controllers. - -The analysis seems thorough based on the accessible parts of the paper. The full text might have more details, but the key -existing methods are covered in the abstract and introduction. Since the user's question is about what existing methods are -mentioned, and we've extracted those from the available content, the task should be completed. - - - Based on the information extracted from the paper's abstract and introduction sections (accessible without subscription), we -have identified the existing methods mentioned in the paper for formation maintenance: - - 1. **Decoupled Approaches**: Existing works treat formation maintenance and collision avoidance as separate problems, -combining solutions for each independently. - 2. **Consensus Theory-Based Formation Controllers**: Traditional methods using consensus algorithms to achieve formation -maintenance. - 3. **Optimal Reciprocal Collision Avoidance (ORCA)**: A strategy used in existing methods for collision avoidance. - 4. **Model-Guided Demonstration Frameworks**: Prior work employing predefined controllers (e.g., consensus or ORCA) to -generate training data for learning-based systems. - - While the full paper might contain additional details, the accessible content provides sufficient evidence to answer the user' -s question about the *mentioned* existing methods. Further access to the full text would not invalidate these findings but could -only add supplementary context. - - -TASK_COMPLETED -""", - -""" -end-user -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -In the paper `Formation control with collision avoidance through deep reinforcement learning using model-guided demonstration`, -what existing methods are mentioned for the problem of formation maintenance? -""", - -""" -commander -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -As you can see, the former agents have done a lot of work. -Now please make a final answer of the original task based on our conversation. - -Please pay special attention to the format in which the answer is presented. -You should first analyze the answer format required by the question and then output the final answer that meets the format -requirements. -Your response should include the following content: -- `final_answer`: enclosed by , the final answer to the question. -Here are some hint about the final answer: - -Your final answer must be output exactly in the format specified by the question. It should be a number OR as few words as -possible OR a comma separated list of numbers and/or strings: -- If you are asked for a number, don't use comma to write your number neither use units such as $ or percent sign unless specified -otherwise. -- If you are asked for a string, don't use articles, neither abbreviations (e.g. for cities), and write the digits in plain text -unless specified otherwise. -- If you are asked for a comma separated list, apply the above rules depending of whether the element to be put in the list is a -number or a string. - -""", - -""" -assistant -────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── -Okay, let me try to figure out how to structure the final answer based on the user's requirements. - -The original question asked what existing methods are mentioned in the paper for formation maintenance. From the previous steps, -we identified four main methods: decoupled approaches, consensus theory-based controllers, ORCA, and model-guided demonstrations. -However, the user wants the answer in a specific format—either a number, a short string, or a comma-separated list. - -Looking back at the hint provided, if it's a list, each element should follow the rules (no articles, no abbreviations unless -specified). The answer should be concise. The paper's abstract and introduction mention that existing works are simple -combinations of methods dealing with the two problems (formation maintenance and collision avoidance) separately. Then, the -approach section refers to using consensus theory and ORCA in the model-guided phase. - -But the key point is that the existing methods are decoupled approaches, consensus-based formation control, and ORCA. The model- -guided demonstrations are part of the proposed method, not the existing ones. Wait, in the analysis, point 4 was model-guided -demonstration frameworks as existing methods. But in the paper's approach, they use a two-stage training framework starting with -imitation learning (which uses model-guided methods). So the existing methods include consensus and ORCA, which are part of the -model-guided demonstration. - -So the answer should list the existing methods as: -1. Decoupled approaches combining separate methods for formation and collision avoidance. -2. Consensus theory-based formation controllers. -3. Optimal Reciprocal Collision Avoidance (ORCA). - -But the user wants the answer in a comma-separated list with as few words as possible. So perhaps abbreviate ORCA but spell out -the others. However, the hint says no abbreviations unless specified. Wait, ORCA is an abbreviation, but in the paper, it's -referred to as Optimal Reciprocal Collision Avoidance (ORCA). Since the paper uses the acronym, maybe it's acceptable. -Alternatively, use the full name. But the example answer in the hint says "don't use abbreviations (e.g. for cities)", so maybe -avoid ORCA and write out Optimal Reciprocal Collision Avoidance. However, in the context of academic papers, ORCA is a standard -term, so using the acronym might be acceptable. Need to check. Since the paper mentions ORCA in the abstract (from the tool result -), it's okay to use the acronym here. - -So the final answer would be a list of three items: -Decoupled approaches,Consensus theory-based formation controllers,Optimal Reciprocal Collision Avoidance (ORCA) - -Wait, but the user said "as few words as possible" or "comma separated list of numbers and/or strings". So maybe shorten them: -Decoupled approaches,Consensus-based controllers,ORCA - -Yes, that's more concise. The paper refers to "consensus theory-based formation controller" which can be shortened to "consensus- -based controllers". ORCA is acceptable as an acronym. So the final answer is a comma-separated list of these three methods. - -Decoupled approaches,Consensus-based controllers,Optimal Reciprocal Collision Avoidance -""" - -] - -from experiencemaker.module.summarizer.traj_summarizer import TrajectorySummarizer -from experiencemaker.schema.trajectory import Trajectory, Sample, SummaryMessage, Message -from experiencemaker.schema.trajectory import Message as BMessage - -class Message(BMessage): - executor: str = None - -ts = TrajectorySummarizer() -role_mapping = { - 'system': 'system', - 'end-user': 'user', - 'commander': 'user', - 'assistant': 'assistant', - 'tool-agent': 'user', - 'tool': 'tool', -} -msg_f = [] -for m in msg: - role = m.split('\n')[1].strip() - content = '\n'.join(m.split('\n')[3:]).strip() - msg_f += [Message(role=role_mapping[role], content=content)] -traj = Trajectory(steps=msg_f) -ts.execute(trajectories=[traj], return_samples=True)