From 29b9c1a45db1f99fad62d200e0bce7f909ee8ce4 Mon Sep 17 00:00:00 2001 From: "jinli.yl" Date: Thu, 24 Jul 2025 16:09:59 +0800 Subject: [PATCH] move to another folder --- .gitignore | 6 +- README.md | 12 +-- TODO.md | 32 -------- .../appworld_react_agent.py | 81 +++++++++---------- .../test_appworld => appworld}/prompt.py | 0 .../requirements.txt | 0 cookbook/appworld/run_appworld.py | 71 ++++++++++++++++ .../run_exp_statistic.py | 19 +++-- .../simple_demo/test_appworld/__init__.py | 0 .../simple_demo/test_appworld/run_appworld.py | 54 ------------- doc/quick_start.md | 2 +- 11 files changed, 132 insertions(+), 145 deletions(-) delete mode 100644 TODO.md rename cookbook/{simple_demo/test_appworld => appworld}/appworld_react_agent.py (57%) rename cookbook/{simple_demo/test_appworld => appworld}/prompt.py (100%) rename cookbook/{simple_demo/test_appworld => appworld}/requirements.txt (100%) create mode 100644 cookbook/appworld/run_appworld.py rename cookbook/{simple_demo/test_appworld => appworld}/run_exp_statistic.py (56%) delete mode 100644 cookbook/simple_demo/test_appworld/__init__.py delete mode 100644 cookbook/simple_demo/test_appworld/run_appworld.py diff --git a/.gitignore b/.gitignore index 06be6891..fa2cb275 100644 --- a/.gitignore +++ b/.gitignore @@ -24,6 +24,6 @@ beyond* step_experiences/* build/* *.egg-info/* -cookbook/simple_demo/test_appworld/data/* -cookbook/simple_demo/test_appworld/experiments/* -cookbook/simple_demo/test_appworld/exp_result/* \ No newline at end of file +cookbook/appworld/data/* +cookbook/appworld/experiments/* +cookbook/appworld/exp_result/* \ No newline at end of file diff --git a/README.md b/README.md index 8a5a4197..8b3560b0 100644 --- a/README.md +++ b/README.md @@ -224,7 +224,7 @@ Load the {path}/{workspace_id}.jsonl file into the vector store, workspace_id={w ```python response = requests.post(url=base_url + "vector_store", json={ - "workspace_id": "test_workspace1", + "workspace_id": workspace_id, "action": "load", "path": "./", }) @@ -244,12 +244,12 @@ print(response.json()) We test ExperienceMaker on Appworld with qwen3-8b: -| Method | best@1 | best@2 | best@4 | -|------------------------------------------|--------|----------|------------| -| w/o ExperienceMaker (baseline) | 0.3561 | 0.4052 | 0.4536 | -| **w ExperienceMaker** | | | | +| Method | best@1 | best@2 | best@4 | +|------------------------------------------|------------|--------------|------------| +| w/o ExperienceMaker (baseline) | 0.3561 | 0.4052 | 0.4536 | +| **w ExperienceMaker** | | | | | [1] extract + compare + recall | **0.4069** | **0.5066** | 0.618 | -| [2] extract + compare + recall + rewrite | 0.3910 | 0.5038 | **0.6211** | +| [2] extract + compare + recall + rewrite | 0.3910 | 0.5038 | **0.6211** | ### 🔧 Experiment on BFCL-V3 diff --git a/TODO.md b/TODO.md deleted file mode 100644 index fb245dc8..00000000 --- a/TODO.md +++ /dev/null @@ -1,32 +0,0 @@ -## TODO -1. zhaoyang建议 - 1. rerank prompt优化,去掉不相关的 @jiaji 优化 - 2. 多数据场景数据验证 @jiaji & @zouyin -2. README @jinli -3. cookbook: - 1. 基础的react agent @jinli - 2. appworld 测试 + 主实验 + 模块开关 + topk消融 + readme/code @jiaji - 3. bfcl 测试 + 主实验 + 模块开关 + topk消融 + readme/code @zouyin -4. advanced usage - 1. 全局op参数 @jinli - 2. work介绍 @jinli -5. experience store @jinli - 1. git 仓库 / hf @jinli -6. 合并beyondagent client -7. 设计理念 + RoadMAP @jiaji - 1. 金融experience - 2. making tools - - -# 兆洋 -1. bedrock代码扫一下上下文管理 -2. 和亮哥合作 固话sop, 自动抽取、固化小型的SOP @唤海 @贺世奇 @刺葳 @悦鸿 -3. Experience列大纲,future工作 -4. 周四下午和兆洋对一下 - -# 锦鲤 -1. LOGO更加简洁 agent <-> experience -2. default等名字的说明 -3. 查看所有的op代码 -4. ba增加client -5. 在appworld上跑 \ No newline at end of file diff --git a/cookbook/simple_demo/test_appworld/appworld_react_agent.py b/cookbook/appworld/appworld_react_agent.py similarity index 57% rename from cookbook/simple_demo/test_appworld/appworld_react_agent.py rename to cookbook/appworld/appworld_react_agent.py index 57c2c3cf..5b361edd 100644 --- a/cookbook/simple_demo/test_appworld/appworld_react_agent.py +++ b/cookbook/appworld/appworld_react_agent.py @@ -1,4 +1,7 @@ import os +from typing import List + +from tqdm import tqdm os.environ["APPWORLD_ROOT"] = "." from dotenv import load_dotenv @@ -24,7 +27,7 @@ class AppworldReactAgent: def __init__(self, index: int, - task_id: str, + task_ids: List[str], experiment_name: str, model_name: str = "qwen3-32b", temperature: float = 0.9, @@ -32,22 +35,19 @@ class AppworldReactAgent: max_response_size: int = 2000): self.index: int = index - self.task_id: str = task_id + self.task_ids: List[str] = task_ids self.experiment_name: str = experiment_name self.model_name: str = model_name self.temperature: float = temperature self.max_interactions: int = max_interactions self.max_response_size: int = max_response_size - self.world: AppWorld = AppWorld(task_id=task_id, experiment_name=experiment_name) - self.history: list[dict] = self.prompt_messages() + self.llm_client = OpenAI() def call_llm(self, messages: list) -> str: for i in range(100): try: - client = OpenAI() - # Change this function to modify the base llm - response = client.chat.completions.create( + response = self.llm_client.chat.completions.create( model=self.model_name, messages=messages, temperature=self.temperature, @@ -62,10 +62,10 @@ class AppworldReactAgent: return "call llm error" - def prompt_messages(self) -> list[dict]: - dictionary = {"supervisor": self.world.task.supervisor, "instruction": self.world.task.instruction} + @staticmethod + def prompt_messages(world: AppWorld) -> list[dict]: + dictionary = {"supervisor": world.task.supervisor, "instruction": world.task.instruction} prompt = Template(PROMPT_TEMPLATE.lstrip()).render(dictionary) - # Extract and return the OpenAI JSON formatted messages from the prompt messages: list[dict] = [] last_start = 0 for match in re.finditer("(USER|ASSISTANT|SYSTEM):\n", prompt): @@ -83,64 +83,57 @@ class AppworldReactAgent: messages[-1]["content"] = prompt[last_start:] return messages - def next_code_block(self) -> str: - return self.call_llm(self.history) - - def next_step(self, code: str) -> str: - output = self.world.execute(code) - if len(output) > self.max_response_size: - logger.warning(f"output exceed max size={len(output)}") - output = output[:self.max_response_size] - return output - - def get_reward(self) -> float: - tracker = self.world.evaluate() + @staticmethod + def get_reward(world) -> float: + tracker = world.evaluate() num_passes = len(tracker.passes) num_failures = len(tracker.failures) return num_passes / (num_passes + num_failures) def execute(self): - try: - with self.world: - before_score = self.get_reward() - logger.info(f"instruction={self.world.task.instruction} before_score={before_score:.4f}") + result = [] + for task_index, task_id in enumerate(tqdm(self.task_ids, desc=f"ray_index={self.index}")): + with AppWorld(task_id=task_id, experiment_name=self.experiment_name) as world: + history = self.prompt_messages(world=world) + before_score = self.get_reward(world) + logger.info(f"ray_id={self.index} task_index={task_index} instruction={world.task.instruction} " + f"before_score={before_score:.4f}") for i in range(self.max_interactions): - code = self.next_code_block() - self.history.append({"role": "assistant", "content": code}) + code = self.call_llm(history) + history.append({"role": "assistant", "content": code}) - output = self.next_step(code) - self.history.append({"role": "user", "content": output}) + output = world.execute(code) + if len(output) > self.max_response_size: + logger.warning(f"output exceed max size={len(output)}") + output = output[:self.max_response_size] + history.append({"role": "user", "content": output}) - logger.info(f"index={self.index} task_id={self.task_id} iteration={i}") + logger.info(f"ray_id={self.index} task_index={task_index} step={i} complete~") - if self.world.task_completed(): + if world.task_completed(): break - after_score = self.get_reward() + after_score = self.get_reward(world) uplift_score = after_score - before_score - result = { - "task_id": self.task_id, + t_result = { + "task_id": world.task_id, "experiment_name": self.experiment_name, - "task_completed": self.world.task_completed(), + "task_completed": world.task_completed(), "before_score": before_score, "after_score": after_score, "uplift_score": uplift_score, - "task_history": self.history, + "task_history": history, } - # logger.info(f"result={json.dumps(result)}") - # p_bar.close() - return result + result.append(t_result) - except Exception as e: - logger.exception(f"encounter error with {e.args}") - return {} + return result def main(): dataset_name = "train" task_ids = load_task_ids(dataset_name) - agent = AppworldReactAgent(index=0, task_id=task_ids[0], experiment_name=f"jinli_{dataset_name}") + agent = AppworldReactAgent(index=0, task_ids=task_ids[0:1], experiment_name=f"jinli_{dataset_name}") result = agent.execute() logger.info(f"result={json.dumps(result)}") diff --git a/cookbook/simple_demo/test_appworld/prompt.py b/cookbook/appworld/prompt.py similarity index 100% rename from cookbook/simple_demo/test_appworld/prompt.py rename to cookbook/appworld/prompt.py diff --git a/cookbook/simple_demo/test_appworld/requirements.txt b/cookbook/appworld/requirements.txt similarity index 100% rename from cookbook/simple_demo/test_appworld/requirements.txt rename to cookbook/appworld/requirements.txt diff --git a/cookbook/appworld/run_appworld.py b/cookbook/appworld/run_appworld.py new file mode 100644 index 00000000..0ce947fb --- /dev/null +++ b/cookbook/appworld/run_appworld.py @@ -0,0 +1,71 @@ +import os +import time + +import ray +from ray import logger + +os.environ["APPWORLD_ROOT"] = "." +from dotenv import load_dotenv + +load_dotenv("../../../.env") + +import json +from pathlib import Path + +from appworld import load_task_ids + +from appworld_react_agent import AppworldReactAgent + + +def run_agent(dataset_name: str, experiment_suffix: str, max_workers: int): + experiment_name = dataset_name + "_" + experiment_suffix + path: Path = Path(f"./exp_result") + path.mkdir(parents=True, exist_ok=True) + + task_ids = load_task_ids(dataset_name) + result: list = [] + + def dump_file(): + with open(path / f"{experiment_name}.jsonl", "w") as f: + for x in result: + f.write(json.dumps(x) + "\n") + + if max_workers > 1: + future_list: list = [] + for i in range(max_workers): + actor = AppworldReactAgent.remote(index=i, + task_ids=task_ids[i::max_workers], + experiment_name=experiment_name) + future = actor.execute.remote() + future_list.append(future) + time.sleep(1) + logger.info("submit complete") + + for i, future in enumerate(future_list): + t_result = ray.get(future) + if t_result: + if isinstance(t_result, list): + result.extend(t_result) + else: + result.append(t_result) + + logger.info(f"{i + 1}/{len(task_ids)} complete") + dump_file() + + else: + for index, task_id in enumerate(task_ids): + agent = AppworldReactAgent(index=index, task_ids=[task_id], experiment_name=experiment_name) + result.append(agent.execute()) + dump_file() + + +def main(): + max_workers = 8 + if max_workers > 1: + ray.init(num_cpus=8) + run_agent(dataset_name="train", experiment_suffix="v2", max_workers=max_workers) + # run_agent(dataset_name="dev", experiment_suffix="v2") + + +if __name__ == "__main__": + main() diff --git a/cookbook/simple_demo/test_appworld/run_exp_statistic.py b/cookbook/appworld/run_exp_statistic.py similarity index 56% rename from cookbook/simple_demo/test_appworld/run_exp_statistic.py rename to cookbook/appworld/run_exp_statistic.py index 3e12db91..1cafe731 100644 --- a/cookbook/simple_demo/test_appworld/run_exp_statistic.py +++ b/cookbook/appworld/run_exp_statistic.py @@ -17,16 +17,25 @@ def run_exp_statistic(): if not line.strip(): continue data = json.loads(line) - task_completed_list.append(1 if data["task_completed"] is True else 0) - before_score_list.append(data["before_score"]) - after_score_list.append(data["after_score"]) - task_success_list.append(data["after_score"] > 0.9) + if isinstance(data, list): + for part_data in data: + task_completed_list.append(1 if part_data["task_completed"] is True else 0) + before_score_list.append(part_data["before_score"]) + after_score_list.append(part_data["after_score"]) + task_success_list.append(part_data["after_score"] > 0.9) + + else: + task_completed_list.append(1 if data["task_completed"] is True else 0) + before_score_list.append(data["before_score"]) + after_score_list.append(data["after_score"]) + task_success_list.append(data["after_score"] > 0.9) task_completed_ratio = sum(task_completed_list) / len(task_completed_list) before_score_ratio = sum(before_score_list) / len(before_score_list) after_score_ratio = sum(after_score_list) / len(after_score_list) task_success_ratio = sum(task_success_list) / len(task_success_list) - logger.info(f"task_completed_ratio={task_completed_ratio:.2f} " + logger.info(f"file={file} " + f"task_completed_ratio={task_completed_ratio:.2f} " f"before_score_ratio={before_score_ratio:.2f} " f"after_score_ratio={after_score_ratio:.2f} " f"task_success_ratio={task_success_ratio:.2f}") diff --git a/cookbook/simple_demo/test_appworld/__init__.py b/cookbook/simple_demo/test_appworld/__init__.py deleted file mode 100644 index e69de29b..00000000 diff --git a/cookbook/simple_demo/test_appworld/run_appworld.py b/cookbook/simple_demo/test_appworld/run_appworld.py deleted file mode 100644 index 931b19bc..00000000 --- a/cookbook/simple_demo/test_appworld/run_appworld.py +++ /dev/null @@ -1,54 +0,0 @@ -import os - -import ray -from ray import logger - -os.environ["APPWORLD_ROOT"] = "." -from dotenv import load_dotenv - -load_dotenv("../../../.env") - -import json -from pathlib import Path - -from appworld import load_task_ids - -from appworld_react_agent import AppworldReactAgent - - -def run_agent(dataset_name: str, experiment_suffix: str, multi_process: bool = True): - experiment_name = dataset_name + "_" + experiment_suffix - path: Path = Path(f"./exp_result") - path.mkdir(parents=True, exist_ok=True) - - task_ids = load_task_ids(dataset_name) - result: list = [] - - def dump_file(): - with open(path / f"{experiment_name}.jsonl", "w") as f: - for x in result: - f.write(json.dumps(x) + "\n") - - if multi_process: - future_list: list = [] - for index, task_id in enumerate(task_ids): - actor = AppworldReactAgent.remote(index=index, task_id=task_id, experiment_name=experiment_name) - future = actor.execute.remote() - future_list.append(future) - logger.info("submit complete") - - for future in future_list: - result.append(ray.get(future)) - dump_file() - - else: - for index, task_id in enumerate(task_ids): - agent = AppworldReactAgent(index=index, task_id=task_id, experiment_name=experiment_name) - result.append(agent.execute()) - dump_file() - - -if __name__ == "__main__": - ray.init(num_cpus=4) - # run_agent(dataset_name="train", experiment_suffix="v2") - run_agent(dataset_name="dev", experiment_suffix="v2") diff --git a/doc/quick_start.md b/doc/quick_start.md index 49aea15f..02a24b37 100644 --- a/doc/quick_start.md +++ b/doc/quick_start.md @@ -142,7 +142,7 @@ Load the {path}/{workspace_id}.jsonl file into the vector store, workspace_id={w ```python response = requests.post(url=base_url + "vector_store", json={ - "workspace_id": "test_workspace1", + "workspace_id": workspace_id, "action": "load", "path": "./", })