mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-10-09 03:20:54 +00:00
add appworld react
This commit is contained in:
parent
5874c33c87
commit
202f5a7d75
13 changed files with 359 additions and 67 deletions
2
.gitignore
vendored
2
.gitignore
vendored
|
|
@ -24,3 +24,5 @@ beyond*
|
|||
step_experiences/*
|
||||
build/*
|
||||
*.egg-info/*
|
||||
cookbook/simple_demo/test_appworld/data/*
|
||||
cookbook/simple_demo/test_appworld/experiments/*
|
||||
|
|
@ -1,7 +1,7 @@
|
|||
# ExperienceMaker
|
||||
|
||||
<p align="center">
|
||||
<img src="doc/logo_v2.png" alt="ExperienceMaker Logo" width="50%">
|
||||
<img src="doc/figure/logo_v2.png" alt="ExperienceMaker Logo" width="50%">
|
||||
</p>
|
||||
|
||||
<p align="center">
|
||||
|
|
@ -66,7 +66,7 @@ ExperienceMaker changes this paradigm by:
|
|||
|
||||
### 🏗️ Framework Architecture
|
||||
<p align="center">
|
||||
<img src="doc/framework.png" alt="ExperienceMaker Architecture" width="70%">
|
||||
<img src="doc/figure/framework.png" alt="ExperienceMaker Architecture" width="70%">
|
||||
</p>
|
||||
|
||||
ExperienceMaker follows a modular, production-ready architecture designed for scalability:
|
||||
|
|
|
|||
0
cookbook/simple_demo/test_appworld/__init__.py
Normal file
0
cookbook/simple_demo/test_appworld/__init__.py
Normal file
147
cookbook/simple_demo/test_appworld/appworld_react_agent.py
Normal file
147
cookbook/simple_demo/test_appworld/appworld_react_agent.py
Normal file
|
|
@ -0,0 +1,147 @@
|
|||
import json
|
||||
import os
|
||||
|
||||
os.environ["APPWORLD_ROOT"] = "."
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv("../../../.env")
|
||||
|
||||
import re
|
||||
import time
|
||||
|
||||
from appworld import AppWorld, load_task_ids
|
||||
from jinja2 import Template
|
||||
from loguru import logger
|
||||
from openai import OpenAI
|
||||
|
||||
from prompt import PROMPT_TEMPLATE
|
||||
|
||||
@ray.remote
|
||||
class AppworldReactAgent:
|
||||
"""A minimal ReAct Agent for AppWorld tasks."""
|
||||
|
||||
def __init__(self,
|
||||
task_id: str,
|
||||
experiment_name: str,
|
||||
model_name: str = "qwen3-8b",
|
||||
temperature: float = 0.9,
|
||||
max_interactions: int = 50,
|
||||
max_response_size: int = 2000):
|
||||
|
||||
self.task_id: str = task_id
|
||||
self.experiment_name: str = experiment_name
|
||||
self.model_name: str = model_name
|
||||
self.temperature: float = temperature
|
||||
self.max_interactions: int = max_interactions
|
||||
self.max_response_size: int = max_response_size
|
||||
|
||||
self.world: AppWorld = AppWorld(task_id=task_id, experiment_name=experiment_name)
|
||||
self.history: list[dict] = self.prompt_messages()
|
||||
|
||||
def call_llm(self, messages: list) -> str:
|
||||
for i in range(100):
|
||||
try:
|
||||
client = OpenAI()
|
||||
# Change this function to modify the base llm
|
||||
response = client.chat.completions.create(
|
||||
model=self.model_name,
|
||||
messages=messages,
|
||||
temperature=self.temperature,
|
||||
max_tokens=400,
|
||||
extra_body={"enable_thinking": False},
|
||||
seed=0)
|
||||
|
||||
return response.choices[0].message.content
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"encounter error with {e.args}")
|
||||
time.sleep(1 + i * 10)
|
||||
|
||||
return "call llm error"
|
||||
|
||||
def prompt_messages(self) -> list[dict]:
|
||||
dictionary = {"supervisor": self.world.task.supervisor, "instruction": self.world.task.instruction}
|
||||
prompt = Template(PROMPT_TEMPLATE.lstrip()).render(dictionary)
|
||||
# Extract and return the OpenAI JSON formatted messages from the prompt
|
||||
messages: list[dict] = []
|
||||
last_start = 0
|
||||
for match in re.finditer("(USER|ASSISTANT|SYSTEM):\n", prompt):
|
||||
last_end = match.span()[0]
|
||||
if len(messages) == 0:
|
||||
if last_end != 0:
|
||||
raise ValueError(
|
||||
f"Start of the prompt has no assigned role: {prompt[:last_end]}"
|
||||
)
|
||||
else:
|
||||
messages[-1]["content"] = prompt[last_start:last_end]
|
||||
role_type = match.group(1).lower()
|
||||
messages.append({"role": role_type, "content": None})
|
||||
last_start = match.span()[1]
|
||||
messages[-1]["content"] = prompt[last_start:]
|
||||
return messages
|
||||
|
||||
def next_code_block(self) -> str:
|
||||
return self.call_llm(self.history)
|
||||
|
||||
def next_step(self, code: str) -> str:
|
||||
output = self.world.execute(code)
|
||||
if len(output) > self.max_response_size:
|
||||
logger.warning(f"output exceed max size={len(output)}")
|
||||
output = output[:self.max_response_size]
|
||||
return output
|
||||
|
||||
def get_reward(self) -> float:
|
||||
tracker = self.world.evaluate()
|
||||
num_passes = len(tracker.passes)
|
||||
num_failures = len(tracker.failures)
|
||||
return num_passes / (num_passes + num_failures)
|
||||
|
||||
def execute(self):
|
||||
try:
|
||||
with self.world:
|
||||
before_score = self.get_reward()
|
||||
logger.info(f"instruction={self.world.task.instruction} before_score={before_score:.4f}")
|
||||
|
||||
for i in range(self.max_interactions):
|
||||
code = self.next_code_block()
|
||||
self.history.append({"role": "assistant", "content": code})
|
||||
|
||||
output = self.next_step(code)
|
||||
self.history.append({"role": "user", "content": output})
|
||||
|
||||
logger.info(f"task_id={self.task_id} iteration={i} "
|
||||
f"code=\n{code}\n output=\n{output}\n "
|
||||
f"score={self.get_reward():.4f}")
|
||||
|
||||
if self.world.task_completed():
|
||||
break
|
||||
|
||||
after_score = self.get_reward()
|
||||
uplift_score = after_score - before_score
|
||||
result = {
|
||||
"task_id": self.task_id,
|
||||
"experiment_name": self.experiment_name,
|
||||
"task_completed": self.world.task_completed(),
|
||||
"before_score": before_score,
|
||||
"after_score": after_score,
|
||||
"uplift_score": uplift_score,
|
||||
"task_history": self.history,
|
||||
}
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"encounter error with {e.args}")
|
||||
return {}
|
||||
|
||||
|
||||
def main():
|
||||
dataset_name = "train"
|
||||
task_ids = load_task_ids(dataset_name)
|
||||
agent = AppworldReactAgent(task_id=task_ids[0], experiment_name=f"jinli_{dataset_name}")
|
||||
result = agent.execute()
|
||||
logger.info(f"result={json.dumps(result)}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
156
cookbook/simple_demo/test_appworld/prompt.py
Normal file
156
cookbook/simple_demo/test_appworld/prompt.py
Normal file
|
|
@ -0,0 +1,156 @@
|
|||
# This is a basic prompt template containing all the necessary onboarding information to solve AppWorld tasks. It explains the role of the agent and the supervisor, how to explore the API documentation, how to operate the interactive coding environment and call APIs via a simple task, and provides key instructions and disclaimers.
|
||||
|
||||
# You can adapt it as needed by your agent. You can also choose to bypass API docs app and build your own API retrieval, e.g., for FullCodeRefl, IPFunCall, etc, we asked an LLM to predict relevant APIs separately and put its documentation directly in the prompt.
|
||||
PROMPT_TEMPLATE = """
|
||||
USER:
|
||||
I am your supervisor and you are a super intelligent AI Assistant whose job is to achieve my day-to-day tasks completely autonomously.
|
||||
|
||||
To do this, you will need to interact with app/s (e.g., spotify, venmo, etc) using their associated APIs on my behalf. For this you will undertake a *multi-step conversation* using a python REPL environment. That is, you will write the python code and the environment will execute it and show you the result, based on which, you will write python code for the next step and so on, until you've achieved the goal. This environment will let you interact with app/s using their associated APIs on my behalf.
|
||||
|
||||
Here are three key APIs that you need to know to get more information
|
||||
|
||||
# To get a list of apps that are available to you.
|
||||
print(apis.api_docs.show_app_descriptions())
|
||||
|
||||
# To get the list of apis under any app listed above, e.g. supervisor
|
||||
print(apis.api_docs.show_api_descriptions(app_name='supervisor'))
|
||||
|
||||
# To get the specification of a particular api, e.g. supervisor app's show_account_passwords
|
||||
print(apis.api_docs.show_api_doc(app_name='supervisor', api_name='show_account_passwords'))
|
||||
|
||||
Each code execution will produce an output that you can use in subsequent calls. Using these APIs, you can now generate code, that the environment will execute, to solve the task.
|
||||
|
||||
For example, consider the task:
|
||||
|
||||
My name is: {{ supervisor.first_name }} {{ supervisor.last_name }}. My personal email is {{ supervisor.email }} and phone number is {{ supervisor.phone_number }}.
|
||||
|
||||
Task:
|
||||
|
||||
What is the password for my Spotify account?
|
||||
|
||||
ASSISTANT:
|
||||
# Okay. Lets first find which apps are available to get the password by looking at the app descriptions.
|
||||
print(apis.api_docs.show_app_descriptions())
|
||||
|
||||
USER:
|
||||
[
|
||||
{
|
||||
"name": "api_docs",
|
||||
"description": "An app to search and explore API documentation."
|
||||
},
|
||||
{
|
||||
"name": "supervisor",
|
||||
"description": "An app to access supervisor's personal information, account credentials, addresses, payment cards, and manage the assigned task."
|
||||
},
|
||||
...
|
||||
{
|
||||
"name": "spotify",
|
||||
"description": "A music streaming app to stream songs and manage song, album and playlist libraries."
|
||||
},
|
||||
{
|
||||
"name": "venmo",
|
||||
"description": "A social payment app to send, receive and request money to and from others."
|
||||
},
|
||||
...
|
||||
]
|
||||
|
||||
|
||||
ASSISTANT:
|
||||
# Looks like the supervisor app could help me with that. Lets see what apis are available under this app.
|
||||
print(apis.api_docs.show_api_descriptions(app_name='supervisor'))
|
||||
|
||||
|
||||
USER:
|
||||
[
|
||||
...
|
||||
"show_account_passwords : Show your supervisor's account passwords."
|
||||
...
|
||||
]
|
||||
|
||||
|
||||
ASSISTANT:
|
||||
# I can use `show_account_passwords` to get the passwords. Let me see its detailed specification to understand its arguments and output structure.
|
||||
print(apis.api_docs.show_api_doc(app_name='supervisor', api_name='show_account_passwords'))
|
||||
|
||||
USER:
|
||||
{
|
||||
'app_name': 'supervisor',
|
||||
'api_name': 'show_account_passwords',
|
||||
'path': '/account_passwords',
|
||||
'method': 'GET',
|
||||
'description': "Show your supervisor's app account passwords.",
|
||||
'parameters': [],
|
||||
'response_schemas': {
|
||||
'success': [{'account_name': 'string', 'password': 'string'}],
|
||||
'failure': {'message': 'string'}
|
||||
}
|
||||
}
|
||||
|
||||
ASSISTANT:
|
||||
# Okay, it requires no arguments. So I can just call it directly.
|
||||
print(apis.supervisor.show_account_passwords())
|
||||
|
||||
USER:
|
||||
[
|
||||
{
|
||||
"account_name": "spotify",
|
||||
"password": "dummy_spotify_pass"
|
||||
},
|
||||
{
|
||||
"account_name": "file_system",
|
||||
"password": "dummy_fs_pass"
|
||||
},
|
||||
...
|
||||
]
|
||||
|
||||
|
||||
ASSISTANT:
|
||||
# So the Spotify password is an entry in the `passwords` list with the account_name=spotify.
|
||||
spotify_password = [account_password["account_name"] == "spotify" for account_password in passwords][0]["password"]
|
||||
print(spotify_password)
|
||||
|
||||
|
||||
USER:
|
||||
dummy_spotify_pass
|
||||
|
||||
ASSISTANT:
|
||||
# When the task is completed, I need to call apis.supervisor.complete_task(). If there is an answer, I need to pass it as an argument `answer`. I will pass the spotify_password as an answer.
|
||||
apis.supervisor.complete_task(answer=spotify_password)
|
||||
|
||||
|
||||
USER:
|
||||
Marked the active task complete.
|
||||
|
||||
|
||||
----------------------------------------------
|
||||
|
||||
USER:
|
||||
**Key instructions and disclaimers**:
|
||||
|
||||
1. The email addresses, access tokens and variables (e.g. spotify_password) in the example above were only for demonstration. Obtain the correct information by calling relevant APIs yourself.
|
||||
2. Only generate valid code blocks, i.e., do not put them in ```...``` or add any extra formatting. Any thoughts should be put as code comments.
|
||||
3. You can use the variables from the previous code blocks in the subsequent code blocks.
|
||||
4. Write small chunks of code and only one chunk of code in every step. Make sure everything is working correctly before making any irreversible change.
|
||||
5. The provided Python environment has access to its standard library. But modules and functions that have a risk of affecting the underlying OS, file system or process are disabled. You will get an error if do call them.
|
||||
6. Any reference to a file system in the task instructions means the file system *app*, operable via given APIs, and not the actual file system the code is running on. So do not write code making calls to os-level modules and functions.
|
||||
7. To interact with apps, only use the provided APIs, and not the corresponding Python packages. E.g., do NOT use `spotipy` for Spotify. Remember, the environment only has the standard library.
|
||||
8. The provided API documentation has both the input arguments and the output JSON schemas. All calls to APIs and parsing its outputs must be as per this documentation.
|
||||
9. For APIs that return results in "pages", make sure to consider all pages.
|
||||
10. To obtain current date or time, use Python functions like `datetime.now()` or obtain it from the phone app. Do not rely on your existing knowledge of what the current date or time is.
|
||||
11. For all temporal requests, use proper time boundaries, e.g., if I ask for something that happened yesterday, make sure to consider the time between 00:00:00 and 23:59:59. All requests are concerning a single, default (no) time zone.
|
||||
12. Any reference to my friends, family or any other person or relation refers to the people in my phone's contacts list.
|
||||
13. All my personal information, and information about my app account credentials, physical addresses and owned payment cards are stored in the "supervisor" app. You can access them via the APIs provided by the supervisor app.
|
||||
14. Once you have completed the task, call `apis.supervisor.complete_task()`. If the task asks for some information, return it as the answer argument, i.e. call `apis.supervisor.complete_task(answer=<answer>)`. For tasks that do not require an answer, just skip the answer argument or pass it as None.
|
||||
15. The answers, when given, should be just entity or number, not full sentences, e.g., `answer=10` for "How many songs are in the Spotify queue?". When an answer is a number, it should be in numbers, not in words, e.g., "10" and not "ten".
|
||||
16. You can also pass `status="fail"` in the complete_task API if you are sure you cannot solve it and want to exit.
|
||||
17. You must make all decisions completely autonomously and not ask for any clarifications or confirmations from me or anyone else.
|
||||
|
||||
USER:
|
||||
Using these APIs, now generate code to solve the actual task:
|
||||
|
||||
My name is: {{ supervisor.first_name }} {{ supervisor.last_name }}. My personal email is {{ supervisor.email }} and phone number is {{ supervisor.phone_number }}.
|
||||
|
||||
Task:
|
||||
|
||||
{{ instruction }}
|
||||
"""
|
||||
4
cookbook/simple_demo/test_appworld/requirements.txt
Normal file
4
cookbook/simple_demo/test_appworld/requirements.txt
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
jinja2
|
||||
loguru
|
||||
openai
|
||||
ray
|
||||
43
cookbook/simple_demo/test_appworld/run_appworld.py
Normal file
43
cookbook/simple_demo/test_appworld/run_appworld.py
Normal file
|
|
@ -0,0 +1,43 @@
|
|||
import os
|
||||
|
||||
os.environ["APPWORLD_ROOT"] = "."
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv("../../../.env")
|
||||
|
||||
import json
|
||||
import time
|
||||
from concurrent.futures import ProcessPoolExecutor
|
||||
from pathlib import Path
|
||||
|
||||
from appworld import load_task_ids
|
||||
|
||||
from appworld_react_agent import AppworldReactAgent
|
||||
|
||||
|
||||
def run_agent(dataset_name: str, max_workers: int, experiment_suffix: str):
|
||||
experiment_name = dataset_name + "_" + experiment_suffix
|
||||
path: Path = Path(f"./exp_result")
|
||||
path.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
task_ids = load_task_ids(dataset_name)
|
||||
result: list = []
|
||||
with ProcessPoolExecutor(max_workers=max_workers) as executor:
|
||||
task_list: list = []
|
||||
for index, task_id in enumerate(task_ids):
|
||||
agent = AppworldReactAgent(task_id, experiment_name)
|
||||
task = executor.submit(agent.execute)
|
||||
task_list.append(task)
|
||||
time.sleep(1)
|
||||
|
||||
for task in task_list:
|
||||
result.append(task.result(timeout=600))
|
||||
|
||||
with open(path / f"{experiment_name}.jsonl", "w") as f:
|
||||
for result_item in result:
|
||||
f.write(json.dumps(result_item) + "\n")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
run_agent(dataset_name="train", experiment_suffix="v1", max_workers=1)
|
||||
# run_agent(dataset_name="dev", experiment_suffix="v1", max_workers=1)
|
||||
|
Before Width: | Height: | Size: 417 KiB After Width: | Height: | Size: 417 KiB |
|
Before Width: | Height: | Size: 2.2 MiB After Width: | Height: | Size: 2.2 MiB |
|
Before Width: | Height: | Size: 1,009 KiB After Width: | Height: | Size: 1,009 KiB |
0
experience_store/appworld/default.jsonl
Normal file
0
experience_store/appworld/default.jsonl
Normal file
|
|
@ -54,15 +54,14 @@ class SimpleSummaryOp(BaseOp):
|
|||
|
||||
def execute(self):
|
||||
request: SummarizerRequest = self.context.request
|
||||
response: SummarizerResponse = self.context.response
|
||||
|
||||
for trajectory in request.traj_list:
|
||||
self.submit_task(self.summary_trajectory, trajectory=trajectory)
|
||||
|
||||
experience_list: List[BaseExperience] = self.join_task()
|
||||
|
||||
response: SummarizerResponse = self.context.response
|
||||
response.experience_list = experience_list
|
||||
for e in experience_list:
|
||||
response.experience_list = self.join_task()
|
||||
for e in response.experience_list:
|
||||
logger.info(f"add experience when_to_use={e.when_to_use}\ncontent={e.content}")
|
||||
|
||||
from experiencemaker.op.vector_store.update_vector_store_op import UpdateVectorStoreOp
|
||||
self.context.set_context(UpdateVectorStoreOp.INSERT_EXPERIENCE_LIST, [x.to_vector_node() for x in experience_list])
|
||||
self.context.set_context(UpdateVectorStoreOp.INSERT_EXPERIENCE_LIST, response.experience_list)
|
||||
|
|
|
|||
|
|
@ -1,59 +0,0 @@
|
|||
from typing import List
|
||||
from loguru import logger
|
||||
|
||||
from experiencemaker.op import OP_REGISTRY
|
||||
from experiencemaker.op.base_op import BaseOp
|
||||
from experiencemaker.schema.experience import BaseExperience
|
||||
from experiencemaker.schema.vector_node import VectorNode
|
||||
|
||||
|
||||
@OP_REGISTRY.register()
|
||||
class ExperienceStorageOp(BaseOp):
|
||||
current_path: str = __file__
|
||||
|
||||
def execute(self):
|
||||
"""Store experiences to vector database"""
|
||||
# Get experiences to store
|
||||
experiences: List[BaseExperience] = self.context.get_context("deduplicated_experiences", [])
|
||||
if not experiences:
|
||||
experiences = self.context.get_context("validated_experiences", [])
|
||||
if not experiences:
|
||||
experiences = self.context.get_context("extracted_experiences", [])
|
||||
|
||||
if not experiences:
|
||||
logger.info("No experiences found for storage")
|
||||
return
|
||||
|
||||
logger.info(f"Storing {len(experiences)} experiences to vector database")
|
||||
|
||||
try:
|
||||
# Convert to vector storage nodes
|
||||
nodes: List[VectorNode] = [experience.to_vector_node() for experience in experiences]
|
||||
|
||||
# Get workspace_id
|
||||
workspace_id = self.context.request.workspace_id if hasattr(self.context, 'request') else None
|
||||
if not workspace_id:
|
||||
workspace_id = self.op_params.get("default_workspace_id", "default")
|
||||
|
||||
# Store to vector database
|
||||
if hasattr(self, 'vector_store') and self.vector_store:
|
||||
self.vector_store.insert(nodes, workspace_id=workspace_id)
|
||||
logger.info(f"Successfully stored {len(experiences)} experiences to workspace: {workspace_id}")
|
||||
|
||||
# Set storage result to context
|
||||
self.context.set_context("storage_success", True)
|
||||
self.context.set_context("stored_count", len(experiences))
|
||||
|
||||
else:
|
||||
logger.error("Vector store not available for storage")
|
||||
self.context.set_context("storage_success", False)
|
||||
self.context.set_context("storage_error", "Vector store not available")
|
||||
|
||||
# Log stored experiences
|
||||
for experience in experiences:
|
||||
logger.info(f"Stored experience - Description: {str(experience.when_to_use)[:100]}...")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error storing experiences: {e}")
|
||||
self.context.set_context("storage_success", False)
|
||||
self.context.set_context("storage_error", str(e))
|
||||
Loading…
Add table
Reference in a new issue