From 0c2fb9a5221c0019a17130002bde2dcaef539617 Mon Sep 17 00:00:00 2001 From: "jinli.yl" Date: Sun, 31 Aug 2025 15:49:16 +0800 Subject: [PATCH] refactor(reme_ai): formatting and minor code improvements - Removed unnecessary comments - Improved code formatting across multiple files - Updated import statements and removed unused imports - Refactored some function definitions for better readability --- cookbook/simple_demo/memory.jsonl | 26 -- cookbook/simple_demo/messages.jsonl | 129 ++++++-- cookbook/simple_demo/personal_memory.jsonl | 123 ++++++++ cookbook/simple_demo/simple_demo.py | 108 ------- cookbook/simple_demo/task_memory.jsonl | 78 +++++ .../simple_demo/use_personal_memory_demo.py | 85 +++++ cookbook/simple_demo/use_task_memory_demo.py | 291 ++++++++++++++++++ example.env | 14 +- pyproject.toml | 2 +- reme_ai/config/default.yaml | 52 ++-- reme_ai/react/__init__.py | 2 +- reme_ai/react/simple_react_op.py | 6 +- reme_ai/retrieve/__init__.py | 2 +- reme_ai/retrieve/task/build_query_prompt.yaml | 10 +- reme_ai/retrieve/task/merge_memory_op.py | 3 +- reme_ai/retrieve/task/rerank_memory_op.py | 2 +- reme_ai/retrieve/task/rewrite_memory_op.py | 5 +- reme_ai/schema/__init__.py | 2 +- reme_ai/schema/memory.py | 4 +- reme_ai/summary/__init__.py | 2 +- reme_ai/summary/personal/__init__.py | 2 +- .../personal/get_reflection_subject_op.py | 2 +- .../summary/personal/info_filter_prompt.yaml | 16 +- reme_ai/summary/personal/update_insight_op.py | 2 +- .../summary/task/comparative_extraction_op.py | 15 +- reme_ai/summary/task/failure_extraction_op.py | 4 +- .../summary/task/memory_deduplication_op.py | 22 +- reme_ai/summary/task/memory_validation_op.py | 17 +- .../summary/task/pdf_preprocess_op_wrapper.py | 14 +- .../task/simple_comparative_summary_op.py | 6 +- reme_ai/summary/task/success_extraction_op.py | 4 +- .../summary/task/trajectory_preprocess_op.py | 5 +- .../task/trajectory_segmentation_op.py | 5 +- reme_ai/utils/miner_u_pdf_processor.py | 60 ++-- reme_ai/utils/op_utils.py | 208 +------------ reme_ai/vector_store/delete_memory_op.py | 14 +- reme_ai/vector_store/update_memory_freq_op.py | 10 +- .../vector_store/update_memory_utility_op.py | 8 +- .../vector_store/vector_store_action_op.py | 7 +- test_workspace.jsonl | 3 + 40 files changed, 854 insertions(+), 516 deletions(-) delete mode 100644 cookbook/simple_demo/memory.jsonl create mode 100644 cookbook/simple_demo/personal_memory.jsonl delete mode 100644 cookbook/simple_demo/simple_demo.py create mode 100644 cookbook/simple_demo/task_memory.jsonl create mode 100644 cookbook/simple_demo/use_personal_memory_demo.py create mode 100644 cookbook/simple_demo/use_task_memory_demo.py create mode 100644 test_workspace.jsonl diff --git a/cookbook/simple_demo/memory.jsonl b/cookbook/simple_demo/memory.jsonl deleted file mode 100644 index 5054c573..00000000 --- a/cookbook/simple_demo/memory.jsonl +++ /dev/null @@ -1,26 +0,0 @@ -[ - { - "workspace_id": "test_workspace4", - "memory_id": "922a556d988f47bcadd8228d9f2235dc", - "memory_type": "task", - "when_to_use": "When conducting a company analysis requiring multi-dimensional insights (e.g., financials, market position, strategic shifts)", - "content": "Split the analysis into 3-4 distinct subtopics (e.g., financial performance, market challenges, strategic initiatives) and perform targeted web searches for each. Use language matching the user's query (Chinese in this case) to ensure relevant results. Avoid broad single searches that may miss critical angles.", - "score": 0.0, - "time_created": "2025-08-29 01:44:37", - "time_modified": "2025-08-29 01:44:37", - "author": "qwen3-30b-a3b-thinking-2507", - "metadata": {} - }, - { - "workspace_id": "test_workspace4", - "memory_id": "5cac948c7e8141dfb7723ab482d3aa8c", - "memory_type": "task", - "when_to_use": "When verifying if retrieved information is sufficient to answer the user's query", - "content": "Always cross-check that the retrieved data covers all core aspects of the query before terminating. In this case, the financial metrics, market challenges, and strategic pivots were all covered in the search results, making further searches unnecessary. Never assume coverage—explicitly validate completeness.", - "score": 0.0, - "time_created": "2025-08-29 01:44:37", - "time_modified": "2025-08-29 01:44:37", - "author": "qwen3-30b-a3b-thinking-2507", - "metadata": {} - } -] \ No newline at end of file diff --git a/cookbook/simple_demo/messages.jsonl b/cookbook/simple_demo/messages.jsonl index 99475cba..f0ea7dd8 100644 --- a/cookbook/simple_demo/messages.jsonl +++ b/cookbook/simple_demo/messages.jsonl @@ -1,56 +1,145 @@ [ { "role": "user", - "content": "You are a helpful assistant.\nThe current time is 2025-08-29 01:42:07.\nPlease proactively choose the most suitable tool or combination of tools based on the user's question, including web_search,python_execute,terminate etc.\nPlease first think about how to break down the problem into subtasks, what tools and parameters should be used for each subtask, and finally provide the tool call name and parameters.\nTry calling the same tool multiple times with different parameters to obtain information from various perspectives.\nPlease determine the response language based on the language of the user's question.\n\nAnalyze Xiaomi Corporation\n", + "content": "You are a helpful assistant.\nThe current time is 2025-08-31 15:41:22.\nPlease proactively choose the most suitable tool or combination of tools based on the user's question, including web_search,python_execute etc.\nPlease first think about how to break down the problem into subtasks, what tools and parameters should be used for each subtask, and finally provide the tool call name and parameters.\nTry calling the same tool multiple times with different parameters to obtain information from various perspectives.\nPlease determine the response language based on the language of the user's question.\n\nAnalyze Xiaomi Corporation\n", "reasoning_content": "", "tool_calls": [], "tool_call_id": "", - "time_created": "2025-08-29 01:42:07", + "time_created": "2025-08-31 15:41:22", "metadata": {} }, { "role": "assistant", "content": "", - "reasoning_content": "Okay, let's tackle this user query about analyzing Xiaomi Corporation. First, I need to figure out what the user is asking for. They just said \"Analyze Xiaomi Corporation,\" so it's pretty broad. I should break this down into subtasks.\n\nHmm, analyzing a company usually involves looking at their financials, market position, product lines, maybe recent news or trends. But since I don't have real-time data, I need to use web search to get the latest info. The user might want current data, so web_search is probably the first tool to use.\n\nWait, the user mentioned the current time is 2025-08-29, but that's probably a placeholder. The actual date isn't relevant here. I should focus on getting the latest info about Xiaomi.\n\nFirst, I'll need to search for key aspects of Xiaomi. Let me think of relevant keywords. Maybe \"Xiaomi Corporation financial performance 2024\", \"Xiaomi market share smartphone 2024\", \"Xiaomi recent products 2024\", and \"Xiaomi global expansion strategy\". Using multiple search terms will give a more comprehensive analysis.\n\nWait, the tools say if there are multiple keywords, use each separately. So I should make four separate web_search calls with each keyword. Let me check the function description again. Yes, \"if there are multiple search keywords, please use each keyword separately to call this tool.\"\n\nSo, I'll generate four tool calls with each query. Let me list them:\n\n1. \"Xiaomi Corporation financial performance 2024\"\n2. \"Xiaomi market share smartphone 2024\"\n3. \"Xiaomi recent products 2024\"\n4. \"Xiaomi global expansion strategy\"\n\nEach of these will get different pieces of information. Once I have the search results, I can compile the analysis. But since I can't do that yet, the next step is to make those web searches.\n\nWait, the user might not need all four, but to be thorough, it's better to cover different angles. Let me confirm if these keywords are appropriate. Financial performance, market share, recent products, and expansion strategy are all key areas for a company analysis.\n\nI should make sure the queries are specific to 2024 since the current time in the system is 2025, but the user's query is for analysis, so recent data is needed. Using 2024 in the query makes sense.\n\nSo, I'll proceed to call web_search four times with each of those keywords. Then, once the results come back, I can synthesize the information. But since I can't do that now, the immediate action is to make those four tool calls.\n", - "tool_calls": [], + "reasoning_content": "", + "tool_calls": [ + { + "index": 0, + "id": "call_2116799d79ed41048e364f", + "type": "function", + "name": "web_search", + "arguments": "{\"query\": \"Xiaomi Corporation company overview\"}", + "description": "", + "input_schema": {}, + "output_schema": {} + }, + { + "index": 1, + "id": "call_1609fb473ad54e8aab3b96", + "type": "function", + "name": "web_search", + "arguments": "{\"query\": \"Xiaomi Corporation financial performance 2024\"}", + "description": "", + "input_schema": {}, + "output_schema": {} + }, + { + "index": 2, + "id": "call_b6ea98ff91484ea39a6efa", + "type": "function", + "name": "web_search", + "arguments": "{\"query\": \"Xiaomi Corporation product portfolio and market share\"}", + "description": "", + "input_schema": {}, + "output_schema": {} + } + ], "tool_call_id": "", - "time_created": "2025-08-29 01:42:15", + "time_created": "2025-08-31 15:41:23", "metadata": {} }, { - "role": "user", - "content": "\n- Xiaomi's smartphone revenue reached CNY 191.8 billion ($26.5 billion), accounting for 52% of overall sales. Xiaomi sold 169 million units in 2024.\n- Total Revenue for 2024 was RMB 365.9 billion.\n- Xiaomi posted record-breaking financial results for 2024, with the CEO Lei Jun calling it the \"strongest annual report in history.\"\n- Revenue for the first quarter of 2025 surged 47.4 percent year-on-year to 111.3 billion yuan ($15.5 billion), indicating continued strong performance following 2024.\n\n\nHere is the relevant information regarding Xiaomi's smartphone market share in 2024:\n\n1. In Q1 2025, Xiaomi's smartphone revenue is mentioned, and it reclaimed the top spot in mainland China after a decade, capturing 18.8% market share. (Note: Though mentioned in Q1 2025, it indicates Xiaomi's strong performance that might have carried over from 2024.)\n\n2. In 2024 Q4, Xiaomi's market share of smartphone shipments in mainland China increased by 3.0 percentage points YoY to 15.8%, marking its 4 consecutive quarters of market share growth.\n\n3. Additionally, in the RMB3,000 - 4,000 segment in 2024 Q3, Xiaomi's market share reached 18.1%, up 9.3 percentage points YoY.\n\nThese data points reflect Xiaomi's growing market presence in the Chinese smartphone market during 2024.\n\n\nIn 2024, Xiaomi has released several new products across different categories:\n\n- **Xiaomi 15 Series and Xiaomi HyperOS 2**: On October 29, 2024, Xiaomi held a new product launch event, introducing the Xiaomi 15 series smartphones and the Xiaomi HyperOS 2. The Xiaomi 15 series includes four color options, starting at a price of 4499 yuan. Two special editions, the Custom Edition and the Diamond Limited Edition, are priced at 4999 yuan and 5999 yuan, respectively. These smartphones are equipped with the latest Snapdragon 8 Gen 3 processor, featuring enhanced performance and energy efficiency. Additionally, the Xiaomi 15 series boasts improved Leica optical high-speed lens systems and the advanced Xiaomi AISP 2.0 computational photography platform powered by large models.\n\n- **Xiaomi SU7 Ultra Production Version**: As part of the October 29 launch event, Xiaomi also unveiled the Xiaomi SU7 Ultra production version, which is described as the fastest four-door production car available on the streets, capable of accelerating from 0 to 100 km/h in just 1.98 seconds. The pre-sale price is set at 814,900 yuan, with an expected official release in March 2025.\n\n- **Xiaomi Watch S4 and Xiaomi Band 9 Pro**: These wearable devices were also introduced during the October 29 event, contributing to Xiaomi's ecosystem of interconnected devices.\n\n- **Xiaomi MIX Fold4 and Xiaomi MIX Flip**: During a summer product launch event on July 19, Xiaomi introduced the Xiaomi MIX Fold4 and Xiaomi MIX Flip. The Xiaomi MIX Fold4 is noted for its lightweight and slim design, weighing 226g and having a single-edge thickness of 4.59mm. It features Xiaomi's latest Dragon Bone Hinge 2.0 design, ensuring stability and safety over 500,000 folding tests. The Xiaomi MIX Flip is a compact foldable phone with a width of only 74.02mm and a weight of 192g, ideal for one-handed operation.\n\n- **Redmi K70 Ultra**: Also launched on July 19, this smartphone is equipped with the Dimensity 9300+ processor and an independent graphics card D1, achieving a benchmark score exceeding 2.38 million in performance tests.\n\n- **Xiaomi 14 Ultra**: This device was introduced as an annual imaging flagship in February 2024, featuring a Leica-tuned quad-camera system that pushes image quality to new extremes.\n\n- **Xiaomi 14T Series**: At an international launch event in September 2024, Xiaomi showcased the professional night photography capabilities of the Xiaomi 14T series, capable of capturing true-to-life colors in low-light conditions.\n\nThese products highlight Xiaomi's commitment to innovation and technological advancement across multiple product lines and markets.\n\n\nXiaomi's global expansion strategy involves several key elements:\n\n1. **Market Positioning and Branding**: Xiaomi's success in the global market is attributed to its precise market positioning, localization strategies, and strong brand building.\n\n2. **International Market Entry**: By 2016, Xiaomi had started to expand internationally, aiming to make \"good but cheap things\" and replicate its domestic success in new markets.\n\n3. **Product Ecosystem**: The company aggressively built a robust ecosystem around its products, which included selling over 70 million mobile phones in 2015.\n\n4. **Leadership**: Hugo Barra, a top Android executive, left Google a few years earlier to lead Xiaomi’s international growth.\n\n5. **Shipping and Logistics**: Enhancing customer satisfaction through timely deliveries and optimizing time-to-market for new product launches are integral to Xiaomi's global strategy.\n\n6. **Localization Strategies**: Xiaomi employs localization strategies to better fit the markets it enters, which helps in achieving a higher global market share.\n\n7. **Global Market Share**: Xiaomi's global market share has hit a new high of 27%, indicating the effectiveness of its expansion strategies.\n\nThink based on the current content and the user's question: Is the current context sufficient to answer the user's question? \n\n- If the current context is not sufficient to answer the user's question, consider what information is missing. \n Re-plan and think about how to break down the missing information into subtasks. \n For each subtask, determine what tools and parameters should be used for the query.\n Please first provide the reasoning process, then give the tool call name and parameters.\n\n- If the current context is sufficient to answer the user's question, use the **terminate** tool.\n", + "role": "tool", + "content": "以下是从知识库中提取的与用户查询相关的内容:\n\n小米公司正式成立于2010年4月,并于2018年7月9日在香港联交所主板上市。小米公司是一家以智能手机、智能硬件和IoT平台为核心的消费电子及智能制造公司。创新与品质并重,小米不断追求极致的用户体验和运营效率。小米的使命是,始终坚持做\"感动人心、价格厚道\"的好产品,让全球每个人都能享受科技带来的美好生活。\n\n小米公司是一家主要从事智能手机、互联网电视以及智能家居生态链建设的创新型科技企业。其愿景是“让每个人都能享受科技的乐趣”。公司通过四个分部开展业务,主要从事智能手机、物联网(IoT)和生活消费产品的研发和销售业务。\n\n公司名称为XiaomiCorporation(小米集团)。公司股本为675,000美元,分为70,000,000,000股每股面值0.0000025美元的A类普通股和200,000,000,000股每股面值0.0000025美元的B类普通股。公司设立的宗旨没有限制,拥有全部权力及权限实施不受开曼群岛法例禁止的目标。各股东所承担的责任以其所持股份不时未缴付的股款为限。\n\n以上信息截至2025年8月19日。", "reasoning_content": "", "tool_calls": [], - "tool_call_id": "", - "time_created": "2025-08-29 01:42:41", + "tool_call_id": "call_2116799d79ed41048e364f", + "time_created": "2025-08-31 15:41:34", + "metadata": {} + }, + { + "role": "tool", + "content": "The following information pertains to Xiaomi Corporation's financial performance in 2024:\n\n- Xiaomi reported record-breaking financial results for 2024, with smartphone revenue reaching CNY 191.8 billion ($26.5 billion), accounting for 52% of overall sales. The company sold 169 million units in 2024. (Posted on March 19, 2025)\n\n- Xiaomi's annual financial report for 2024 was planned to be announced on March 18, 2025 (Tuesday).\n\n- Xiaomi's Q2 2024 performance was scheduled to be released on August 21, 2024 (Wednesday), with an investor conference call or webcast set for the same day at 7:30 PM (Beijing time).\n\n- Xiaomi's Q3 2024 performance was scheduled to be released on November 18, 2024 (Monday), with an investor conference call or webcast planned for the same day at 8:30 PM HKT (Hong Kong time).", + "reasoning_content": "", + "tool_calls": [], + "tool_call_id": "call_1609fb473ad54e8aab3b96", + "time_created": "2025-08-31 15:41:34", + "metadata": {} + }, + { + "role": "tool", + "content": "Xiaomi Corporation's product portfolio includes:\n\n- **Monitors & Media**: For instance, the Xiaomi 2K Gaming Monitor G27Qi, which features a 27-inch fast LCD with a 180Hz high refresh rate.\n- **Wearables**\n- **Home Appliances**\n- **Lifestyle Companions**\n\nRegarding Xiaomi's market share, the information provided highlights that in the RMB3,000 - 4,000 segment, Xiaomi's market share reached 18.1%, showing an increase of 9.3 percentage points year-over-year. However, no specific figures are given for other segments or the overall market share of Xiaomi's entire product portfolio.", + "reasoning_content": "", + "tool_calls": [], + "tool_call_id": "call_b6ea98ff91484ea39a6efa", + "time_created": "2025-08-31 15:41:34", "metadata": {} }, { "role": "assistant", - "content": "\nThe user's query to \"Analyze Xiaomi Corporation\" has been addressed comprehensively using the provided data from web searches. The analysis covers:\n1. **Financial Performance** (2024 revenue, smartphone sales, Q1 2025 growth)\n2. **Market Share** (China smartphone dominance, segment-specific performance)\n3. **Product Portfolio** (2024 flagship launches across smartphones, EVs, wearables)\n4. **Global Strategy** (market expansion, localization, 27% global market share)\n\nAll key aspects of a company analysis are covered with current data (2024-2025). No additional information is required.\n\n\n\nThe analysis of Xiaomi Corporation for 2024-2025 is complete, based on the following key findings:\n\n✅ **Financials**: \n- Total Revenue: RMB 365.9 billion (2024)\n- Smartphone Revenue: RMB 191.8 billion (52% of total sales)\n- 169 million smartphones sold in 2024\n- Q1 2025 Revenue: RMB 111.3 billion (+47.4% YoY)\n\n✅ **Market Position**:\n- China smartphone market share: 15.8% (Q4 2024), 18.8% (Q1 2025)\n- Top segment performance: 18.1% in RMB 3,000-4,000 range (Q3 2024)\n\n✅ **Product Innovation**:\n- Xiaomi 15 Series (Oct 2024) with Snapdragon 8 Gen 3\n- SU7 Ultra EV (pre-sale at 814,900 yuan)\n- MIX Fold4 (226g, 4.59mm thickness) and MIX Flip\n\n✅ **Global Expansion**:\n- 27% global market share (record high)\n- Localization strategies driving international growth\n- 70+ million phones shipped globally in 2015 (historical context)\n\nThis represents Xiaomi's strongest annual performance to date, with sustained growth across products, markets, and revenue streams.\n\n\n", - "reasoning_content": "Okay, let's see. The user asked to analyze Xiaomi Corporation. I've been provided with four tool responses from web searches covering financial performance, market share, recent products, and global expansion strategy. Let me check if all these pieces are sufficient to form a comprehensive analysis.\n\nFirst, the financial data: total revenue 365.9 billion RMB in 2024, smartphone revenue 191.8 billion, 169 million units sold. That's solid for financials. Then market share in China: 15.8% in Q4 2024, 18.8% in Q1 2025, which shows growth. Recent products include the 15 series, SU7 Ultra, and foldables. Global strategy mentions 27% market share and localization efforts.\n\nWait, the user's question is to \"analyze Xiaomi Corporation.\" The provided data covers key areas: financials, market share, product launches, and expansion. Each of the four search results addresses a different aspect. The user probably needs a summary that ties these together. Since all necessary data is here, there's no need for more web searches. The next step is to compile this into a coherent answer.\n\nI should check if there's anything missing. For example, maybe competition details or future projections, but the user didn't ask for that. The given data seems sufficient for a basic analysis. The financials, market share, product updates, and global strategy are all covered. So the context is sufficient. Therefore, the terminate tool with status success should be used.", - "tool_calls": [], + "content": "", + "reasoning_content": "", + "tool_calls": [ + { + "index": 0, + "id": "call_78001de6ca57472f9c66a9", + "type": "function", + "name": "web_search", + "arguments": "{\"query\": \"Xiaomi Corporation market share smartphone global 2024\"}", + "description": "", + "input_schema": {}, + "output_schema": {} + } + ], "tool_call_id": "", - "time_created": "2025-08-29 01:42:51", + "time_created": "2025-08-31 15:41:35", "metadata": {} }, { - "role": "user", - "content": "Please integrate the context and provide a complete answer to the user's question.\n\n# User's Question\n{query}\n", + "role": "tool", + "content": "- In 2024, Xiaomi smartphone sales increased significantly, reaching over 168 million unit shipments, accounting for over 13 percent of the global smartphone market. (Statista, 2025年2月19日)\n\n- Xiaomi ranked No.4 globally in smartphone shipments in 2024, with a market share of 16.1%. (Xiaomi Corporation - Investor Relations, 2025年3月2日)", "reasoning_content": "", "tool_calls": [], - "tool_call_id": "", - "time_created": "2025-08-29 01:42:52", + "tool_call_id": "call_78001de6ca57472f9c66a9", + "time_created": "2025-08-31 15:41:40", "metadata": {} }, { "role": "assistant", - "content": "terminate\n{\n \"answer\": \"Xiaomi Corporation has delivered its strongest annual performance to date in 2024, with comprehensive growth across all key business dimensions:\\n\\n✅ **Financial Performance**\\n- Total Revenue: RMB 365.9 billion (2024)\\n- Smartphone Revenue: RMB 191.8 billion (52% of total sales)\\n- 169 million smartphones sold globally\\n- Q1 2025 Revenue: RMB 111.3 billion (+47.4% YoY)\\n- CEO Lei Jun described 2024 as \\\"the strongest annual report in history\\\"\\n\\n✅ **Market Dominance**\\n- China smartphone market share: 15.8% (Q4 2024), 18.8% (Q1 2025)\\n- 18.1% share in RMB 3,000-4,000 segment (Q3 2024)\\n- 4 consecutive quarters of market share growth in China\\n\\n✅ **Product Innovation**\\n- Xiaomi 15 Series (Oct 2024): Snapdragon 8 Gen 3, Leica optics, HyperOS 2\\n- SU7 Ultra EV: 1.98s 0-100km/h acceleration, 814,900 yuan pre-sale\\n- MIX Fold4: 226g, 4.59mm thickness, 500k+ fold tests\\n- Redmi K70 Ultra: Dimensity 9300+ processor, 2.38M+ benchmark\\n\\n✅ **Global Expansion**\\n- Achieved 27% global market share (record high)\\n- Localization strategies driving international growth\\n- 70+ million phones shipped globally (2015 historical context)\\n\\nXiaomi's 2024-2025 performance demonstrates successful execution of its product ecosystem strategy, market localization, and technological innovation across smartphones, EVs, and IoT devices.\"\n}", - "reasoning_content": "Okay, let's tackle this. The user asked to \"Analyze Xiaomi Corporation,\" and the assistant has already provided a detailed analysis based on the search results. Now, I need to make sure the answer is complete and integrates all the context provided.\n\nFirst, check if all the necessary information is covered. The previous response listed financials, market share, products, and global strategy. The search results included data from 2024 and Q1 2025, which seems up-to-date. The user's current time is 2025-08-29, so the Q1 2025 data is relevant.\n\nWait, the user's question is just \"Analyze Xiaomi Corporation,\" so the answer should summarize the key points without needing more data. The assistant's previous answer seems to have all the necessary info from the provided context. The tools used were web_search for each subtask, and the results were integrated into the analysis.\n\nI need to confirm if there's anything missing. The financials, market share, products, and global strategy are all covered. The mention of the CEO calling it the \"strongest annual report in history\" adds context. The product launches in 2024 (Xiaomi 15 Series, SU7 Ultra, MIX Fold4) are all included. Global expansion mentions 27% market share, which is a key point.\n\nNo additional tools are needed because all required data is present in the context. The assistant should terminate the process and provide the integrated answer. The previous \"terminate\" response was correct. The user's answer is already complete, so the correct action is to use the terminate tool with the answer.\n", + "content": "", + "reasoning_content": "", + "tool_calls": [ + { + "index": 0, + "id": "call_d1cebb4b219d42099d3ee4", + "type": "function", + "name": "web_search", + "arguments": "{\"query\": \"Xiaomi Corporation innovation strategy and future plans 2025\"}", + "description": "", + "input_schema": {}, + "output_schema": {} + } + ], + "tool_call_id": "", + "time_created": "2025-08-31 15:41:40", + "metadata": {} + }, + { + "role": "tool", + "content": "Xiaomi's innovation strategy and future plans for 2025 are highlighted in several pieces of information:\n\n1. **Investment in Core Technologies**: Xiaomi has committed to investing heavily in core technologies, setting a goal to spend 100 billion yuan ($13.8 billion) on research and development (R&D). By 2025, they have exceeded this target, having invested about 105 billion yuan ($14.5 billion) from 2021 to 2025, with 30 billion yuan ($4.2 billion) specifically allocated for R&D in 2025 alone.\n\n2. **Technological Advancement and Industry Transformation**: Xiaomi believes that technological advancement plays a critical role in transforming traditional industries and fostering emerging sectors. The company will continue to innovate to contribute to China's path to modernization.\n\n3. **Expansion into New Sectors**: Under the leadership of Lei Jun, CEO and chairman of Xiaomi Corp, the company is focusing on transforming traditional industries and fostering new sectors through technological innovation. Over the past 15 years, Xiaomi has deeply understood that technological innovation is the cornerstone of progress.\n\n4. **Smartphone Market and Product Launches**: In Q2 2025, Xiaomi launched the Xiaomi YU7 Series in June, achieving significant locked-in orders. This indicates Xiaomi's ongoing commitment to innovation in the smartphone market.\n\n5. **Diversification into EVs and AI**: In Q1 2025, Xiaomi posted record profits and revenue with key gains in electric vehicles (EVs), artificial intelligence (AI), and chip design, marking a major milestone in tech leadership. This suggests that Xiaomi is diversifying its portfolio beyond smartphones into cutting-edge technologies.\n\n6. **Digital Marketing and User Engagement**:小米 is focusing on innovative digital marketing strategies, including a product strategy that emphasizes high cost-performance and an ecosystem chain matrix to build a user experience closed loop. Additionally, Xiaomi uses a penetration pricing and dynamic pricing strategy, along with limited sale mechanisms to strengthen the perception of scarcity. In terms of user engagement, Xiaomi is依托米粉社区培育忠诚度, forming a word-of-mouth communication ecosystem.\n\nThese strategies and plans reflect Xiaomi's holistic approach to maintaining its competitive edge through innovation, diversification, and strong user engagement in 2025.", + "reasoning_content": "", + "tool_calls": [], + "tool_call_id": "call_d1cebb4b219d42099d3ee4", + "time_created": "2025-08-31 15:41:56", + "metadata": {} + }, + { + "role": "assistant", + "content": "Based on the provided information, here is a comprehensive analysis of Xiaomi Corporation:\n\n### 1. **Company Overview**\n- **Founded**: April 2010.\n- **Listing**: Hong Kong Stock Exchange (HKEX) in July 2018.\n- **Core Focus**: Smartphones, smart hardware, and IoT platforms.\n- **Mission**: \"Make products that move people's hearts and are priced fairly.\"\n- **Vision**: \"Let everyone enjoy the fun of technology.\"\n- **Business Model**: A global leader in consumer electronics and intelligent manufacturing, focusing on innovation and quality.\n\n---\n\n### 2. **Financial Performance (2024)**\n- **Revenue**: Record-breaking performance with smartphone revenue reaching CNY 191.8 billion ($26.5 billion), accounting for 52% of total sales.\n- **Unit Sales**: Sold 169 million smartphones in 2024.\n- **Market Position**: Ranked **No. 4 globally** in smartphone shipments with a market share of **16.1%**.\n- **Global Market Share**: Over 13% of the global smartphone market (168+ million units shipped).\n\n---\n\n### 3. **Product Portfolio & Market Share**\n- **Key Product Lines**:\n - **Smartphones**: Core product line with strong performance in mid-to-high-end segments.\n - **IoT & Smart Home Devices**: Includes wearables, home appliances, monitors, and lifestyle companions.\n - **Electric Vehicles (EVs)**: A major new growth area, contributing significantly to Q1 2025 profits.\n - **AI & Chip Design**: Investments in artificial intelligence and semiconductor technology.\n- **Market Share Highlights**:\n - In the RMB 3,000–4,000 price segment, Xiaomi achieved a **18.1% market share**, up 9.3 percentage points YoY.\n - Global smartphone market share reached **16.1%** in 2024, solidifying its position as a top-tier player.\n\n---\n\n### 4. **Innovation Strategy & Future Plans (2025)**\n- **R&D Investment**: Exceeded its target by investing approximately **CNY 105 billion ($14.5 billion)** from 2021 to 2025, with **CNY 30 billion ($4.2 billion)** allocated for 2025 alone.\n- **Focus Areas**:\n - **Core Technologies**: Heavy investment in AI, chip design, and EV development.\n - **Industry Transformation**: Leveraging technology to modernize traditional industries.\n - **New Markets**: Expansion into electric vehicles and AI-driven solutions.\n- **Product Launches**:\n - Launched the **Xiaomi YU7 Series** in June 2025, achieving significant pre-orders.\n- **Marketing & User Engagement**:\n - Emphasis on high cost-performance ratio.\n - Use of dynamic pricing, limited releases, and community-building (e.g., Mi Fans).\n - Creating a closed-loop user experience ecosystem.\n\n---\n\n### 5. **Conclusion**\nXiaomi Corporation has evolved from a smartphone-focused company into a diversified tech giant with leadership in IoT, EVs, AI, and chip design. Its success in 2024 was driven by strong financial performance, global market expansion, and a relentless focus on innovation. With ambitious plans for 2025—including massive R&D investments and strategic diversification—Xiaomi is well-positioned to maintain its competitive edge and drive long-term growth.", + "reasoning_content": "", "tool_calls": [], "tool_call_id": "", - "time_created": "2025-08-29 01:43:02", + "time_created": "2025-08-31 15:42:05", "metadata": {} } ] \ No newline at end of file diff --git a/cookbook/simple_demo/personal_memory.jsonl b/cookbook/simple_demo/personal_memory.jsonl new file mode 100644 index 00000000..1d255c3d --- /dev/null +++ b/cookbook/simple_demo/personal_memory.jsonl @@ -0,0 +1,123 @@ +{ + "answer": "", + "messages": [], + "success": true, + "metadata": { + "memory_list": [ + { + "workspace_id": "personal_memory_demo", + "memory_id": "42f9d6e7e8c34310b284d36cfb73a68d", + "memory_type": "personal", + "when_to_use": "John Smith, 28, San Francisco, tech company", + "content": "user's name is John Smith, he is 28 years old, and works at a tech company in San Francisco", + "score": 0.0, + "time_created": "2025-08-31 15:45:33", + "time_modified": "2025-08-31 15:45:33", + "author": "qwen3-30b-a3b-instruct-2507", + "metadata": { + "keywords": "John Smith, 28, San Francisco, tech company", + "source_message": "My name is John Smith, I'm 28 years old, and I work at a tech company in San Francisco", + "observation_type": "personal_info" + }, + "target": "user", + "reflection_subject": "" + }, + { + "workspace_id": "personal_memory_demo", + "memory_id": "9be49b1479e540c7abf0257595e49972", + "memory_type": "personal", + "when_to_use": "software engineer, backend, Python, Go", + "content": "user is a software engineer specializing in backend development using Python and Go", + "score": 0.0, + "time_created": "2025-08-31 15:45:33", + "time_modified": "2025-08-31 15:45:33", + "author": "qwen3-30b-a3b-instruct-2507", + "metadata": { + "keywords": "software engineer, backend, Python, Go", + "source_message": "I'm a software engineer, mainly doing backend development using Python and Go", + "observation_type": "personal_info" + }, + "target": "user", + "reflection_subject": "" + }, + { + "workspace_id": "personal_memory_demo", + "memory_id": "5287138b20ff432a920cd07d0e034579", + "memory_type": "personal", + "when_to_use": "basketball, sci-fi movies, Dune Part 2", + "content": "user enjoys playing basketball and watching sci-fi movies, recently watched Dune Part 2", + "score": 0.0, + "time_created": "2025-08-31 15:45:33", + "time_modified": "2025-08-31 15:45:33", + "author": "qwen3-30b-a3b-instruct-2507", + "metadata": { + "keywords": "basketball, sci-fi movies, Dune Part 2", + "source_message": "I enjoy playing basketball and watching sci-fi movies. I recently watched Dune Part 2", + "observation_type": "personal_info" + }, + "target": "user", + "reflection_subject": "" + }, + { + "workspace_id": "personal_memory_demo", + "memory_id": "c8f329903b82404c9d36e8a61bb30d67", + "memory_type": "personal", + "when_to_use": "cat, Shadow, pet", + "content": "user has a 3-year-old cat named Shadow", + "score": 0.0, + "time_created": "2025-08-31 15:45:33", + "time_modified": "2025-08-31 15:45:33", + "author": "qwen3-30b-a3b-instruct-2507", + "metadata": { + "keywords": "cat, Shadow, pet", + "source_message": "I have a cat named Shadow who is 3 years old", + "observation_type": "personal_info" + }, + "target": "user", + "reflection_subject": "" + }, + { + "workspace_id": "personal_memory_demo", + "memory_id": "7166549431bf4f019acb8331377f7def", + "memory_type": "personal", + "when_to_use": "Japanese cuisine, sushi, ramen", + "content": "user is interested in Japanese cuisine, especially sushi and ramen", + "score": 0.0, + "time_created": "2025-08-31 15:45:33", + "time_modified": "2025-08-31 15:45:33", + "author": "qwen3-30b-a3b-instruct-2507", + "metadata": { + "keywords": "Japanese cuisine, sushi, ramen", + "source_message": "I'm really interested in Japanese cuisine, especially sushi and ramen", + "observation_type": "personal_info" + }, + "target": "user", + "reflection_subject": "" + }, + { + "workspace_id": "personal_memory_demo", + "memory_id": "4ab4fe610ce845b6b174d03449df2881", + "memory_type": "personal", + "when_to_use": "Japan, Tokyo, Kyoto, trip", + "content": "user is planning a trip to Japan in September 2025, mainly to Tokyo and Kyoto", + "score": 0.0, + "time_created": "2025-08-31 15:45:32", + "time_modified": "2025-08-31 15:45:32", + "author": "qwen3-30b-a3b-instruct-2507", + "metadata": { + "keywords": "Japan, Tokyo, Kyoto, trip", + "time_info": "September 2025", + "source_message": "I'm planning a trip to Japan next month, mainly to Tokyo and Kyoto", + "observation_type": "personal_info_with_time" + }, + "target": "user", + "reflection_subject": "" + } + ], + "deleted_memory_ids": [], + "update_result": { + "deleted_count": 0, + "inserted_count": 6 + } + } +} \ No newline at end of file diff --git a/cookbook/simple_demo/simple_demo.py b/cookbook/simple_demo/simple_demo.py deleted file mode 100644 index 1fa18d83..00000000 --- a/cookbook/simple_demo/simple_demo.py +++ /dev/null @@ -1,108 +0,0 @@ -import json - -import requests -from dotenv import load_dotenv - -load_dotenv() -base_url = "http://0.0.0.0:8002/" -workspace_id = "test_workspace4" - - -def run_agent(query: str, dump_messages: bool = False): - response = requests.post(url=base_url + "react", json={"query": query}) - if response.status_code != 200: - print(response.text) - return [] - - response = response.json() - - answer = response["answer"] - print(answer) - - messages = response["messages"] - if dump_messages: - with open("messages.jsonl", "w") as f: - f.write(json.dumps(messages, indent=2, ensure_ascii=False)) - - return messages - - -def run_summary(messages: list, enable_dump_memory: bool = True): - response = requests.post(url=base_url + "summary_task_memory_simple", json={ - "workspace_id": workspace_id, - "trajectories": [ - {"messages": messages, "score": 1.0} - ] - }) - - if response.status_code != 200: - print(response.text) - return - - response = response.json() - memory_list = response["metadata"]["memory_list"] - if enable_dump_memory: - with open("memory.jsonl", "w") as f: - f.write(json.dumps(memory_list, indent=2, ensure_ascii=False)) - - -def run_retrieve(query: str): - response = requests.post(url=base_url + "retrieve_task_memory_simple", json={ - "workspace_id": workspace_id, - "query": query, - }) - - if response.status_code != 200: - print(response.text) - return "" - - response = response.json() - answer: str = response["answer"] - print(f"answer={answer}") - return answer - - -def run_agent_with_memory(query_first: str, query_second: str, enable_dump_memory: bool = True): - messages = run_agent(query=query_second) - run_summary(messages, enable_dump_memory) - retrieved_memory = run_retrieve(query_first) - messages = run_agent(query=f"{retrieved_memory}\n\nUser Question:\n{query_first}") - return messages - - -def dump_memory(): - response = requests.post(url=base_url + "vector_store", json={ - "workspace_id": workspace_id, - "action": "dump", - "path": "./", - }) - - if response.status_code != 200: - print(response.text) - return - - print(response.json()) - - -def load_memory(): - response = requests.post(url=base_url + "vector_store", json={ - "workspace_id": "test_workspace2", - "action": "load", - "path": "./", - }) - - if response.status_code != 200: - print(response.text) - return - - print(response.json()) - - -if __name__ == "__main__": - query1 = "Analyze Xiaomi Corporation" - query2 = "Analyze the company Tesla." - - run_agent(query=query1, dump_messages=True) - run_agent_with_memory(query_first=query1, query_second=query2) - dump_memory() - load_memory() diff --git a/cookbook/simple_demo/task_memory.jsonl b/cookbook/simple_demo/task_memory.jsonl new file mode 100644 index 00000000..106759b5 --- /dev/null +++ b/cookbook/simple_demo/task_memory.jsonl @@ -0,0 +1,78 @@ +[ + { + "workspace_id": "test_workspace", + "memory_id": "4ff6a7a783a84940a046e557de7542f0", + "memory_type": "task", + "when_to_use": "When analyzing a complex organization with multiple dimensions (history, products, finances, leadership), especially when the query is open-ended or broad like 'analyze [company]'", + "content": "The agent successfully decomposed the task into four targeted web searches: company overview/history, products/services, financial performance, and leadership impact. This multi-angle approach ensured comprehensive coverage of key domains. Each search was designed to extract distinct, high-value information layers, enabling synthesis into a structured, well-rounded analysis. The use of varied queries prevented redundancy and maximized information diversity.", + "score": 0.92, + "time_created": "2025-08-31 15:42:44", + "time_modified": "2025-08-31 15:42:44", + "author": "qwen3-30b-a3b-instruct-2507", + "metadata": { + "when_to_use": "When analyzing a complex organization with multiple dimensions (history, products, finances, leadership), especially when the query is open-ended or broad like 'analyze [company]'", + "experience": "The agent successfully decomposed the task into four targeted web searches: company overview/history, products/services, financial performance, and leadership impact. This multi-angle approach ensured comprehensive coverage of key domains. Each search was designed to extract distinct, high-value information layers, enabling synthesis into a structured, well-rounded analysis. The use of varied queries prevented redundancy and maximized information diversity.", + "tags": [ + "multi-faceted analysis", + "information decomposition", + "web_search strategy", + "comprehensive coverage" + ], + "confidence": 0.9, + "step_type": "reasoning", + "tools_used": [ + "web_search" + ] + } + }, + { + "workspace_id": "test_workspace", + "memory_id": "2ad5b6aae8724e87944e69ea4476a97c", + "memory_type": "task", + "when_to_use": "When initial results from a general search are insufficient or too broad, and deeper insights are needed on specific subtopics such as financial trends, product portfolios, or leadership dynamics", + "content": "The agent proactively called the same tool (web_search) multiple times with different, focused queries to gather information from various perspectives. This iterative technique allowed for layered understanding—e.g., separating historical milestones from current financial challenges and leadership influence. It demonstrated strategic use of repetition not as redundancy but as a method to triangulate data across domains, significantly improving depth and accuracy.", + "score": 0.85, + "time_created": "2025-08-31 15:42:44", + "time_modified": "2025-08-31 15:42:44", + "author": "qwen3-30b-a3b-instruct-2507", + "metadata": { + "when_to_use": "When initial results from a general search are insufficient or too broad, and deeper insights are needed on specific subtopics such as financial trends, product portfolios, or leadership dynamics", + "experience": "The agent proactively called the same tool (web_search) multiple times with different, focused queries to gather information from various perspectives. This iterative technique allowed for layered understanding—e.g., separating historical milestones from current financial challenges and leadership influence. It demonstrated strategic use of repetition not as redundancy but as a method to triangulate data across domains, significantly improving depth and accuracy.", + "tags": [ + "iterative querying", + "perspective diversification", + "information triangulation", + "search optimization" + ], + "confidence": 0.85, + "step_type": "action", + "tools_used": [ + "web_search" + ] + } + }, + { + "workspace_id": "test_workspace", + "memory_id": "87afc1367c6c4b739bbe191e1f47c875", + "memory_type": "task", + "when_to_use": "After gathering raw data from external sources, when synthesizing findings into a coherent, structured response that balances factual reporting with critical insight", + "content": "The agent effectively integrated results from multiple searches into a unified summary with clear sectioning: history, products, finance, and leadership. It highlighted both strengths (innovation, mission-driven vision) and risks (financial decline, leadership dependency). This synthesis transformed fragmented data into actionable intelligence, demonstrating how structured summarization enhances usability and decision-making value in agent outputs.", + "score": 0.85, + "time_created": "2025-08-31 15:42:44", + "time_modified": "2025-08-31 15:42:44", + "author": "qwen3-30b-a3b-instruct-2507", + "metadata": { + "when_to_use": "After gathering raw data from external sources, when synthesizing findings into a coherent, structured response that balances factual reporting with critical insight", + "experience": "The agent effectively integrated results from multiple searches into a unified summary with clear sectioning: history, products, finance, and leadership. It highlighted both strengths (innovation, mission-driven vision) and risks (financial decline, leadership dependency). This synthesis transformed fragmented data into actionable intelligence, demonstrating how structured summarization enhances usability and decision-making value in agent outputs.", + "tags": [ + "synthesis", + "structured summarization", + "critical insight", + "data integration" + ], + "confidence": 0.9, + "step_type": "observation", + "tools_used": [] + } + } +] \ No newline at end of file diff --git a/cookbook/simple_demo/use_personal_memory_demo.py b/cookbook/simple_demo/use_personal_memory_demo.py new file mode 100644 index 00000000..80e3a650 --- /dev/null +++ b/cookbook/simple_demo/use_personal_memory_demo.py @@ -0,0 +1,85 @@ +import asyncio +import json +import aiohttp + +# API base URL +base_url = "http://0.0.0.0:8002" + +async def main(): + # Create a unique workspace ID + workspace_id = "personal_memory_demo" + + async with aiohttp.ClientSession() as session: + # Step 1: Clear existing memories in the workspace + print("Clearing existing memories...") + async with session.post( + f"{base_url}/vector_store", + json={ + "action": "delete", + "workspace_id": workspace_id, + }, + headers={"Content-Type": "application/json"} + ) as response: + result = await response.json() + print(json.dumps(result, ensure_ascii=False)) + + # Step 2: Create a conversation with rich personal information + print("\nCreating conversation with personal information...") + messages = [ + {"role": "user", "content": "My name is John Smith, I'm 28 years old, and I work at a tech company in San Francisco"}, + {"role": "assistant", "content": "Nice to meet you, John!"}, + {"role": "user", "content": "I'm a software engineer, mainly doing backend development using Python and Go"}, + {"role": "assistant", "content": "I see, you're a backend engineer working with Python and Go."}, + {"role": "user", "content": "I enjoy playing basketball and watching sci-fi movies. I recently watched Dune Part 2"}, + {"role": "assistant", "content": "Basketball and sci-fi movies are great hobbies! Dune Part 2 was indeed amazing."}, + {"role": "user", "content": "I have a cat named Shadow who is 3 years old"}, + {"role": "assistant", "content": "Shadow sounds adorable! 3-year-old cats are quite playful."}, + {"role": "user", "content": "I'm planning a trip to Japan next month, mainly to Tokyo and Kyoto"}, + {"role": "assistant", "content": "Your Japan trip sounds exciting! Tokyo and Kyoto are both wonderful destinations with their own unique charm."}, + {"role": "user", "content": "I'm really interested in Japanese cuisine, especially sushi and ramen"}, + {"role": "assistant", "content": "Japanese cuisine is delicious! Sushi and ramen are very popular choices."}, + ] + + # Step 3: Summarize personal memories from the conversation + print("\nSummarizing personal memories...") + async with session.post( + f"{base_url}/summary_personal_memory", + json={ + "messages": messages, + "workspace_id": workspace_id, + }, + headers={"Content-Type": "application/json"} + ) as response: + result = await response.json() + print(json.dumps(result, ensure_ascii=False, indent=2)) + + # Wait for the memories to be processed and stored + print("\nWaiting for memories to be processed...") + await asyncio.sleep(2) + + # Step 4: Retrieve personal memories with different queries + queries = [ + "What's my name and age?", + "What do I do for work?", + "What are my hobbies?", + "Do I have any pets?", + "What are my travel plans?", + "What foods do I like?" + ] + + print("\nRetrieving personal memories...") + for query in queries: + print(f"\nQuery: {query}") + async with session.post( + f"{base_url}/retrieve_personal_memory", + json={ + "query": query, + "workspace_id": workspace_id, + }, + headers={"Content-Type": "application/json"} + ) as response: + result = await response.json() + print(json.dumps(result, ensure_ascii=False, indent=2)) + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/cookbook/simple_demo/use_task_memory_demo.py b/cookbook/simple_demo/use_task_memory_demo.py new file mode 100644 index 00000000..4e934487 --- /dev/null +++ b/cookbook/simple_demo/use_task_memory_demo.py @@ -0,0 +1,291 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +""" +Task Memory Demo for MemoryScope + +This script demonstrates how to use the task memory capabilities of MemoryScope. +It shows how to run an agent, summarize conversations, retrieve memories, and +manage the memory workspace. +""" + +import json +import time +from typing import List, Dict, Any, Optional + +import requests +from dotenv import load_dotenv + +# Load environment variables from .env file +load_dotenv() + +# API configuration +BASE_URL = "http://0.0.0.0:8002/" +WORKSPACE_ID = "test_workspace" + + +def handle_api_response(response: requests.Response) -> Optional[Dict[str, Any]]: + """ + Handle API response with proper error checking + + Args: + response: Response object from requests + + Returns: + Response JSON if successful, None otherwise + """ + if response.status_code != 200: + print(f"Error: {response.status_code}") + print(response.text) + return None + + return response.json() + + +def delete_workspace() -> None: + """ + Delete the current workspace from the vector store + + Returns: + None + """ + response = requests.post( + url=f"{BASE_URL}vector_store", + json={ + "workspace_id": WORKSPACE_ID, + "action": "delete", + } + ) + + result = handle_api_response(response) + if result: + print(f"Workspace '{WORKSPACE_ID}' deleted successfully") + + +def run_agent(query: str, dump_messages: bool = False) -> List[Dict[str, Any]]: + """ + Run the agent with a specific query + + Args: + query: The query to send to the agent + dump_messages: Whether to save messages to a file + + Returns: + List of message objects from the conversation + """ + response = requests.post( + url=f"{BASE_URL}react", + json={"query": query} + ) + + result = handle_api_response(response) + if not result: + return [] + + # Extract and display the answer + answer = result.get("answer", "") + print(f"Agent response: {answer}") + + # Get the conversation messages + messages = result.get("messages", []) + + # Optionally save messages to file + if dump_messages and messages: + with open("messages.jsonl", "w") as f: + f.write(json.dumps(messages, indent=2, ensure_ascii=False)) + print(f"Messages saved to messages.jsonl") + + return messages + + +def run_summary(messages: List[Dict[str, Any]], enable_dump_memory: bool = True) -> None: + """ + Generate a summary of conversation messages and create task memories + + Args: + messages: List of message objects from a conversation + enable_dump_memory: Whether to save memory list to a file + + Returns: + None + """ + if not messages: + print("No messages to summarize") + return + + response = requests.post( + # url=f"{BASE_URL}summary_task_memory_simple", + url=f"{BASE_URL}summary_task_memory", + json={ + "workspace_id": WORKSPACE_ID, + "trajectories": [ + {"messages": messages, "score": 1.0} + ] + } + ) + + result = handle_api_response(response) + if not result: + return + + # Extract memory list from response + memory_list = result.get("metadata", {}).get("memory_list", []) + print(f"Memory list: {memory_list}") + + # Optionally save memory list to file + if enable_dump_memory and memory_list: + with open("memory.jsonl", "w") as f: + f.write(json.dumps(memory_list, indent=2, ensure_ascii=False)) + print(f"Memory saved to memory.jsonl") + + +def run_retrieve(query: str) -> str: + """ + Retrieve relevant task memories based on a query + + Args: + query: The query to retrieve relevant memories + + Returns: + String containing the retrieved memory answer + """ + response = requests.post( + # url=f"{BASE_URL}retrieve_task_memory_simple", + url=f"{BASE_URL}retrieve_task_memory", + json={ + "workspace_id": WORKSPACE_ID, + "query": query, + } + ) + + result = handle_api_response(response) + if not result: + return "" + + # Extract and return the answer + answer = result.get("answer", "") + print(f"Retrieved memory: {answer}") + return answer + + +def run_agent_with_memory(query_first: str, query_second: str, enable_dump_memory: bool = True) -> List[Dict[str, Any]]: + """ + Run the agent with memory augmentation + + This function demonstrates how to use task memory to enhance agent responses: + 1. First run the agent with the second query to build memory + 2. Then summarize the conversation to create memories + 3. Retrieve relevant memories for the first query + 4. Run the agent with the first query augmented with retrieved memories + + Args: + query_first: The query to run with memory augmentation + query_second: The query to build initial memories + enable_dump_memory: Whether to save memory list to a file + + Returns: + List of message objects from the final conversation + """ + # Run agent with second query to build initial memories + print(f"\n--- Building memories with query: '{query_second}' ---") + messages = run_agent(query=query_second) + + # Summarize conversation to create memories + print("\n--- Summarizing conversation to create memories ---") + run_summary(messages, enable_dump_memory) + time.sleep(1) + + # Retrieve relevant memories for the first query + print(f"\n--- Retrieving memories for query: '{query_first}' ---") + retrieved_memory = run_retrieve(query_first) + + # Run agent with first query augmented with retrieved memories + print(f"\n--- Running agent with memory-augmented query ---") + augmented_query = f"{retrieved_memory}\n\nUser Question:\n{query_first}" + print(f"Augmented query: {augmented_query}") + messages = run_agent(query=augmented_query) + + return messages + + +def dump_memory(path: str = "./") -> None: + """ + Dump the vector store memories to disk + + Args: + path: Directory path to save the memories + + Returns: + None + """ + response = requests.post( + url=f"{BASE_URL}vector_store", + json={ + "workspace_id": WORKSPACE_ID, + "action": "dump", + "path": path, + } + ) + + result = handle_api_response(response) + if result: + print(f"Memory dumped to {path}") + + +def load_memory(path: str = "./") -> None: + """ + Load memories from disk into the vector store + + Args: + path: Directory path to load the memories from + + Returns: + None + """ + response = requests.post( + url=f"{BASE_URL}vector_store", + json={ + "workspace_id": WORKSPACE_ID, + "action": "load", + "path": path, + } + ) + + result = handle_api_response(response) + if result: + print(f"Memory loaded from {path}") + + +def main() -> None: + """ + Main function to demonstrate task memory workflow + """ + # Define example queries + query1 = "Analyze Xiaomi Corporation" + query2 = "Analyze the company Tesla." + + print("=== Task Memory Demo ===") + + # Step 1: Clean up workspace + print("\n1. Deleting workspace...") + delete_workspace() + + # Step 2: Run agent with first query and save messages + print("\n2. Running agent with first query...") + run_agent(query=query1, dump_messages=True) + + # Step 3: Demonstrate memory-augmented agent + print("\n3. Running memory-augmented agent workflow...") + run_agent_with_memory(query_first=query1, query_second=query2) + + # Step 4: Demonstrate memory persistence + print("\n4. Dumping memory to disk...") + dump_memory() + + print("\n5. Loading memory from disk...") + load_memory() + + print("\n=== Demo Complete ===") + + +if __name__ == "__main__": + main() diff --git a/example.env b/example.env index ce62c9ce..d655b58b 100644 --- a/example.env +++ b/example.env @@ -1,11 +1,9 @@ -OPENAI_API_KEY=sk-xxxx -OPENAI_BASE_URL=https://xxxx/v1 +FLOW_EMBEDDING_API_KEY=sk-xxxx +FLOW_EMBEDDING_BASE_URL=https://xxxx/v1 -EMBEDDING_API_KEY=sk-xxxx -EMBEDDING_BASE_URL=https://xxxx/v1 +FLOW_LLM_API_KEY=sk-xxxx +FLOW_LLM_BASE_URL=https://xxxx/v1 -LLM_API_KEY=sk-xxxx -LLM_BASE_URL=https://xxxx/v1 +FLOW_ES_HOSTS=http://0.0.0.0:9200 -ES_HOSTS=http://0.0.0.0:9200 -DASHSCOPE_API_KEY=sk-xxxx +FLOW_USE_FRAMEWORK=true diff --git a/pyproject.toml b/pyproject.toml index 54ca33d8..66fef6bc 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -6,7 +6,7 @@ build-backend = "setuptools.build_meta" name = "reme_ai" version = "0.1.0" description = "Remember me of memory and experience" -authors = [{ name = "reme_ai team", email = "reme_ai@alibaba-inc.com" }] +authors = [{ name = "reme_ai_team", email = "reme_ai_team@alibaba-inc.com" }] license = { file = "LICENSE" } readme = "README.md" requires-python = ">=3.12" diff --git a/reme_ai/config/default.yaml b/reme_ai/config/default.yaml index 43840bed..4e410ab3 100644 --- a/reme_ai/config/default.yaml +++ b/reme_ai/config/default.yaml @@ -97,32 +97,36 @@ flow: description: "Whether to update the utility attribute of the retrieved task memory." required: true - delete_task_memory: - flow_content: delete_memory_op >> update_vector_store_op - description: "Delete task memories when utility/freq < utility_threshold and freq >= freq_threshold" - input_schema: - workspace_id: - type: "str" - description: "workspace id" - required: true - freq_threshold: - type: "int" - description: "The retrieved frequency threshold for deleting task memory." - required: true - utility_threshold: - type: "float" - description: "The utility/freq threshold for deleting task memory." - required: true - -# reconsolidate_personal_memory: -# flow_content: load_not_reflected_memory_op >> get_reflection_subject_op >> update_insight_op >> long_contra_repeat_op >> update_vector_store_op -# description: "Consolidate personal memories by generating topic insights, updating values, resolving conflicts, and updating vector store" + delete_task_memory: + flow_content: delete_memory_op >> update_vector_store_op + description: "Delete task memories when utility/freq < utility_threshold and freq >= freq_threshold" + input_schema: + workspace_id: + type: "str" + description: "workspace id" + required: true + freq_threshold: + type: "int" + description: "The retrieved frequency threshold for deleting task memory." + required: true + utility_threshold: + type: "float" + description: "The utility/freq threshold for deleting task memory." + required: true + react: + flow_content: simple_react_op + description: "React to the current task with an agent" + input_schema: + query: + type: "str" + description: "user query" + required: true llm: default: backend: openai_compatible -# model_name: qwen3-30b-a3b-thinking-2507 + # model_name: qwen3-30b-a3b-thinking-2507 model_name: qwen3-30b-a3b-instruct-2507 params: temperature: 0.6 @@ -138,6 +142,6 @@ vector_store: default: backend: elasticsearch embedding_model: default - params: - # hosts: "http://localhost:9200" - hosts: "http://11.160.132.46:8200" +# params: +# hosts: "http://localhost:9200" +# hosts: "http://11.160.132.46:8200" diff --git a/reme_ai/react/__init__.py b/reme_ai/react/__init__.py index 023c0a6d..84721cb6 100644 --- a/reme_ai/react/__init__.py +++ b/reme_ai/react/__init__.py @@ -1 +1 @@ -from .simple_react_op import SimpleReactOp \ No newline at end of file +from .simple_react_op import SimpleReactOp diff --git a/reme_ai/react/simple_react_op.py b/reme_ai/react/simple_react_op.py index ac518c79..12ab680d 100644 --- a/reme_ai/react/simple_react_op.py +++ b/reme_ai/react/simple_react_op.py @@ -1,10 +1,10 @@ from flowllm import C from flowllm.context.flow_context import FlowContext -from flowllm.op.agent.react_v1_op import ReactV1Op +from flowllm.op.agent.react_v2_op import ReactV2Op @C.register_op() -class SimpleReactOp(ReactV1Op): +class SimpleReactOp(ReactV2Op): ... @@ -18,4 +18,4 @@ if __name__ == "__main__": op(context=context) # from reme_ai.schema import Message # result = op.llm.chat(messages=[Message(**{"role": "user", "content": "你叫什么名字?"})]) - # print("!!!", result) \ No newline at end of file + # print("!!!", result) diff --git a/reme_ai/retrieve/__init__.py b/reme_ai/retrieve/__init__.py index 0d054f82..8922c1ab 100644 --- a/reme_ai/retrieve/__init__.py +++ b/reme_ai/retrieve/__init__.py @@ -1,2 +1,2 @@ -from . import task from . import personal +from . import task diff --git a/reme_ai/retrieve/task/build_query_prompt.yaml b/reme_ai/retrieve/task/build_query_prompt.yaml index 8b45b63d..73d025d3 100644 --- a/reme_ai/retrieve/task/build_query_prompt.yaml +++ b/reme_ai/retrieve/task/build_query_prompt.yaml @@ -1,6 +1,6 @@ query_build: | - # Execution Process - {execution_process} - - Read through the entire execution process to understand which part is currently being executed. - Generate a `query` that reflects the current state, which will later be used to search for similar problems in the database and help resolve the issue at hand. + # Execution Process + {execution_process} + + Read through the entire execution process to understand which part is currently being executed. + Generate a `query` that reflects the current state, which will later be used to search for similar problems in the database and help resolve the issue at hand. diff --git a/reme_ai/retrieve/task/merge_memory_op.py b/reme_ai/retrieve/task/merge_memory_op.py index 5ae1a184..75b776fb 100644 --- a/reme_ai/retrieve/task/merge_memory_op.py +++ b/reme_ai/retrieve/task/merge_memory_op.py @@ -20,8 +20,7 @@ class MergeMemoryOp(BaseOp): if not memory.content: continue - content_collector.append(f"- when_to_use: {memory.when_to_use}\n" - f"content: {memory.content}\n") + content_collector.append(f"- {memory.when_to_use} {memory.content}\n") content_collector.append("Please consider the helpful parts from these in answering the question, " "to make the response more comprehensive and substantial.") self.context.response.answer = "\n".join(content_collector) diff --git a/reme_ai/retrieve/task/rerank_memory_op.py b/reme_ai/retrieve/task/rerank_memory_op.py index 029d2ed4..ded5b62a 100644 --- a/reme_ai/retrieve/task/rerank_memory_op.py +++ b/reme_ai/retrieve/task/rerank_memory_op.py @@ -146,4 +146,4 @@ class RerankMemoryOp(BaseLLMOp): except Exception as e: logger.error(f"Error parsing rerank response: {e}") - return [] \ No newline at end of file + return [] diff --git a/reme_ai/retrieve/task/rewrite_memory_op.py b/reme_ai/retrieve/task/rewrite_memory_op.py index f7308dcc..aa36f304 100644 --- a/reme_ai/retrieve/task/rewrite_memory_op.py +++ b/reme_ai/retrieve/task/rewrite_memory_op.py @@ -33,7 +33,7 @@ class RewriteMemoryOp(BaseLLMOp): # Generate initial context message rewritten_memory = self._generate_context_message(query, messages, memory_list) - + # Store results in context self.context.response.answer = rewritten_memory self.context.response.metadata["memory_list"] = [memory.model_dump() for memory in memory_list] @@ -41,7 +41,6 @@ class RewriteMemoryOp(BaseLLMOp): def _generate_context_message(self, query: str, messages: List[Message], memories: List[BaseMemory]) -> str: """Generate context message from retrieved memories""" if not memories: - return "" try: @@ -147,4 +146,4 @@ class RewriteMemoryOp(BaseLLMOp): # If JSON parsing fails, return the response as-is for fallback return response.strip() - return "" \ No newline at end of file + return "" diff --git a/reme_ai/schema/__init__.py b/reme_ai/schema/__init__.py index adb4a0af..271b1234 100644 --- a/reme_ai/schema/__init__.py +++ b/reme_ai/schema/__init__.py @@ -1 +1 @@ -from flowllm.schema.message import Message, Trajectory, ToolCall, Role \ No newline at end of file +from flowllm.schema.message import Message, Role, Trajectory # noqa diff --git a/reme_ai/schema/memory.py b/reme_ai/schema/memory.py index 4fa7a62a..8ed090cd 100644 --- a/reme_ai/schema/memory.py +++ b/reme_ai/schema/memory.py @@ -22,8 +22,8 @@ class BaseMemory(BaseModel, ABC): metadata: dict = Field(default_factory=dict) def update_modified_time(self): - self.modified_time = datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S") - + self.time_modified = datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S") + def update_metadata(self, new_metadata): self.metadata = new_metadata diff --git a/reme_ai/summary/__init__.py b/reme_ai/summary/__init__.py index dec8dbf7..8922c1ab 100644 --- a/reme_ai/summary/__init__.py +++ b/reme_ai/summary/__init__.py @@ -1,2 +1,2 @@ +from . import personal from . import task -from . import personal \ No newline at end of file diff --git a/reme_ai/summary/personal/__init__.py b/reme_ai/summary/personal/__init__.py index 235ee091..1e2a444f 100644 --- a/reme_ai/summary/personal/__init__.py +++ b/reme_ai/summary/personal/__init__.py @@ -5,4 +5,4 @@ from .get_reflection_subject_op import GetReflectionSubjectOp from .info_filter_op import InfoFilterOp from .load_today_memory_op import LoadTodayMemoryOp from .long_contra_repeat_op import LongContraRepeatOp -from .update_insight_op import UpdateInsightOp \ No newline at end of file +from .update_insight_op import UpdateInsightOp diff --git a/reme_ai/summary/personal/get_reflection_subject_op.py b/reme_ai/summary/personal/get_reflection_subject_op.py index f79c9c25..232c8c1d 100644 --- a/reme_ai/summary/personal/get_reflection_subject_op.py +++ b/reme_ai/summary/personal/get_reflection_subject_op.py @@ -101,7 +101,7 @@ class GetReflectionSubjectOp(BaseLLMOp): Args: memory_contents: List of memory content strings existing_subjects: List of already existing subject strings - user_name: Target user name + user_name: Target username num_questions: Maximum number of new subjects to generate Returns: diff --git a/reme_ai/summary/personal/info_filter_prompt.yaml b/reme_ai/summary/personal/info_filter_prompt.yaml index 21b1146d..f553d9e0 100644 --- a/reme_ai/summary/personal/info_filter_prompt.yaml +++ b/reme_ai/summary/personal/info_filter_prompt.yaml @@ -1,11 +1,11 @@ info_filter_system_zh: | - 任务:对所给{batch_size}个句子中所含有的关于{user_name}的信息打分,分数为0,1,2或3。 - 注意:其中0表示不包含用户信息,1表示句子中只包含用户假设的信息或者用户虚构的内容比如用户创作的小说或剧本,2表示包含用户的一般信息,时效性信息或者需要猜测才能得到的用户信息,3表示明确含有或者可以确定推断出关于用户的重要信息,或者用户要求记录。 - {user_name}的重要信息可以包含用户基本信息,用户画像信息,用户兴趣偏好信息,用户性格,用户价值观,用户人际关系,用户重大事件转折点等等重要信息。 - 对每个句子都做一次信息打分,一共输出{batch_size}个分数,不需要写最终结果。 - 请一定要按如下格式依次输出,最后的结果一定要加<>: - 思考:思考的依据和过程,30字以内。 - 结果:<句子序号> <分数:0或1或2或3> + 任务:对所给{batch_size}个句子中所含有的关于{user_name}的信息打分,分数为0,1,2或3。 + 注意:其中0表示不包含用户信息,1表示句子中只包含用户假设的信息或者用户虚构的内容比如用户创作的小说或剧本,2表示包含用户的一般信息,时效性信息或者需要猜测才能得到的用户信息,3表示明确含有或者可以确定推断出关于用户的重要信息,或者用户要求记录。 + {user_name}的重要信息可以包含用户基本信息,用户画像信息,用户兴趣偏好信息,用户性格,用户价值观,用户人际关系,用户重大事件转折点等等重要信息。 + 对每个句子都做一次信息打分,一共输出{batch_size}个分数,不需要写最终结果。 + 请一定要按如下格式依次输出,最后的结果一定要加<>: + 思考:思考的依据和过程,30字以内。 + 结果:<句子序号> <分数:0或1或2或3> info_filter_system: | Task: Score the information about {user_name} contained in the given batch of {batch_size} sentences, with scores of 0, 1, 2, or 3. @@ -118,7 +118,7 @@ info_filter_few_shot: | Sentences: 1 {user_name}: I've just joined Google. 2 {user_name}: There are too many mosquitoes when sleeping outdoors. What should I do? - 3 {user_name}: Is creativity related to extraversion? + 3 {user_name}: Is creativity related to extroversion? 4 {user_name}: Should the review, modification, and standardization of all personnel files in a district or county be handled by the cadre archive center or by the county human resources and social security bureau? 5 {user_name}: If I am planning to have a child with a woman, as a man, how can I protect the woman and the baby and how can I prepare to ensure high sperm quality for the benefit of the offspring? 6 {user_name}: If we go out to play together, would you feel happy? diff --git a/reme_ai/summary/personal/update_insight_op.py b/reme_ai/summary/personal/update_insight_op.py index c1cd0028..57ccfafa 100644 --- a/reme_ai/summary/personal/update_insight_op.py +++ b/reme_ai/summary/personal/update_insight_op.py @@ -76,7 +76,7 @@ class UpdateInsightOp(BaseLLMOp): logger.info(f"Successfully updated {len(updated_insights)} insight memories") def _score_insights_by_relevance(self, insight_memories: List[PersonalMemory], - observation_memories: List[PersonalMemory], + observation_memories: List[PersonalMemory], threshold: float) -> List[tuple]: """ Score insight memories based on relevance to observation memories. diff --git a/reme_ai/summary/task/comparative_extraction_op.py b/reme_ai/summary/task/comparative_extraction_op.py index 13052329..c611abab 100644 --- a/reme_ai/summary/task/comparative_extraction_op.py +++ b/reme_ai/summary/task/comparative_extraction_op.py @@ -31,7 +31,7 @@ class ComparativeExtractionOp(BaseLLMOp): # Hard comparison: success vs failure (if similarity search is enabled) if (success_trajectories and failure_trajectories and - self.op_params.get("enable_similarity_comparison", False)): + self.op_params.get("enable_similarity_comparison", False)): similar_pairs = self._find_similar_step_sequences(success_trajectories, failure_trajectories) logger.info(f"Found {len(similar_pairs)} similar pairs for hard comparison") @@ -47,7 +47,8 @@ class ComparativeExtractionOp(BaseLLMOp): self.context.comparative_task_memories = comparative_task_memories @staticmethod - def _find_highest_lowest_scoring_trajectories(trajectories: List[Trajectory]) -> Tuple[Optional[Trajectory], Optional[Trajectory]]: + def _find_highest_lowest_scoring_trajectories(trajectories: List[Trajectory]) -> Tuple[ + Optional[Trajectory], Optional[Trajectory]]: """Find the highest and lowest scoring trajectories""" if len(trajectories) < 2: return None, None @@ -107,7 +108,8 @@ class ComparativeExtractionOp(BaseLLMOp): return self.llm.chat(messages=[Message(content=prompt)], callback_fn=parse_task_memories) def _extract_hard_comparative_task_memory(self, success_steps: List[Message], - failure_steps: List[Message], similarity_score: float) -> List[BaseMemory]: + failure_steps: List[Message], similarity_score: float) -> List[ + BaseMemory]: """Extract hard comparative task memory (success vs failure)""" prompt = self.prompt_format( prompt_name="hard_comparative_step_task_memory_prompt", @@ -134,7 +136,6 @@ class ComparativeExtractionOp(BaseLLMOp): return self.llm.chat(messages=[Message(content=prompt)], callback_fn=parse_task_memories) - @staticmethod def _get_trajectory_steps(trajectory: Trajectory) -> List[Message]: """Get trajectory steps, prioritizing segmented steps""" @@ -148,7 +149,8 @@ class ComparativeExtractionOp(BaseLLMOp): return trajectory.messages def _find_similar_step_sequences(self, success_trajectories: List[Trajectory], - failure_trajectories: List[Trajectory]) -> List[Tuple[List[Message], List[Message], float]]: + failure_trajectories: List[Trajectory]) -> List[ + Tuple[List[Message], List[Message], float]]: """Find similar step sequences for comparison""" if not self.op_params.get("enable_similarity_comparison", False): return [] @@ -184,7 +186,8 @@ class ComparativeExtractionOp(BaseLLMOp): failure_texts = [merge_messages_content(seq) for seq in failure_step_sequences] # Get embedding vectors - if hasattr(self.context, 'vector_store') and self.context.vector_store and hasattr(self.context.vector_store, 'embedding_model'): + if hasattr(self.context, 'vector_store') and self.context.vector_store and hasattr( + self.context.vector_store, 'embedding_model'): success_embeddings = self.context.vector_store.embedding_model.get_embeddings(success_texts) failure_embeddings = self.context.vector_store.embedding_model.get_embeddings(failure_texts) diff --git a/reme_ai/summary/task/failure_extraction_op.py b/reme_ai/summary/task/failure_extraction_op.py index e64f0baf..21258d9a 100644 --- a/reme_ai/summary/task/failure_extraction_op.py +++ b/reme_ai/summary/task/failure_extraction_op.py @@ -15,7 +15,7 @@ class FailureExtractionOp(BaseLLMOp): def execute(self): """Extract task memories from failed trajectories""" failure_trajectories: List[Trajectory] = self.context.get("failure_trajectories", []) - + if not failure_trajectories: logger.info("No failure trajectories found for extraction") return @@ -23,7 +23,7 @@ class FailureExtractionOp(BaseLLMOp): logger.info(f"Extracting task memories from {len(failure_trajectories)} failed trajectories") failure_task_memories = [] - + # Process trajectories for trajectory in failure_trajectories: if hasattr(trajectory, 'segments') and trajectory.segments: diff --git a/reme_ai/summary/task/memory_deduplication_op.py b/reme_ai/summary/task/memory_deduplication_op.py index a8d33266..bbbdd5e2 100644 --- a/reme_ai/summary/task/memory_deduplication_op.py +++ b/reme_ai/summary/task/memory_deduplication_op.py @@ -1,7 +1,8 @@ from typing import List -from loguru import logger from flowllm import C, BaseOp +from loguru import logger + from reme_ai.schema.memory import BaseMemory @@ -23,8 +24,9 @@ class MemoryDeduplicationOp(BaseOp): # Perform deduplication deduplicated_task_memories = self._deduplicate_task_memories(task_memories) - logger.info(f"Deduplication complete: {len(deduplicated_task_memories)} deduplicated task memories out of {len(task_memories)}") - + logger.info( + f"Deduplication complete: {len(deduplicated_task_memories)} deduplicated task memories out of {len(task_memories)}") + # Update context self.context.memory_list = deduplicated_task_memories @@ -84,7 +86,8 @@ class MemoryDeduplicationOp(BaseOp): if hasattr(node, 'embedding') and node.embedding: existing_embeddings.append(node.embedding) - logger.debug(f"Retrieved {len(existing_embeddings)} existing task memory embeddings from workspace {workspace_id}") + logger.debug( + f"Retrieved {len(existing_embeddings)} existing task memory embeddings from workspace {workspace_id}") return existing_embeddings except Exception as e: @@ -111,10 +114,9 @@ class MemoryDeduplicationOp(BaseOp): logger.error(f"Error generating embedding for task memory: {e}") return None - def _is_similar_to_existing_task_memories(self, current_embedding: List[float], - existing_embeddings: List[List[float]], - threshold: float) -> bool: + existing_embeddings: List[List[float]], + threshold: float) -> bool: """Check if current embedding is similar to existing embeddings""" for existing_embedding in existing_embeddings: similarity = self._calculate_cosine_similarity(current_embedding, existing_embedding) @@ -124,8 +126,8 @@ class MemoryDeduplicationOp(BaseOp): return False def _is_similar_to_current_task_memories(self, current_embedding: List[float], - current_task_memories: List[BaseMemory], - threshold: float) -> bool: + current_task_memories: List[BaseMemory], + threshold: float) -> bool: for existing_task_memory in current_task_memories: existing_embedding = self._get_task_memory_embedding(existing_task_memory) if existing_embedding is None: @@ -158,4 +160,4 @@ class MemoryDeduplicationOp(BaseOp): except Exception as e: logger.error(f"Error calculating cosine similarity: {e}") - return 0.0 \ No newline at end of file + return 0.0 diff --git a/reme_ai/summary/task/memory_validation_op.py b/reme_ai/summary/task/memory_validation_op.py index 4c702362..878948f2 100644 --- a/reme_ai/summary/task/memory_validation_op.py +++ b/reme_ai/summary/task/memory_validation_op.py @@ -40,7 +40,7 @@ class MemoryValidationOp(BaseLLMOp): logger.warning(f"Task memory validation failed: {reason}") logger.info(f"Validated {len(validated_task_memories)} out of {len(task_memories)} task memories") - + # Update context self.context.response.answer = json.dumps([x.model_dump() for x in validated_task_memories]) self.context.response.metadata["memory_list"] = validated_task_memories @@ -62,7 +62,7 @@ class MemoryValidationOp(BaseLLMOp): def parse_validation(message: Message) -> Dict[str, Any]: try: response_content = message.content - + # Parse validation result # Extract JSON blocks json_pattern = r'```json\s*([\s\S]*?)\s*```' @@ -73,19 +73,20 @@ class MemoryValidationOp(BaseLLMOp): else: parsed = {} - is_valid = parsed.get("is_valid",True) - score = parsed.get("score",0.5) + is_valid = parsed.get("is_valid", True) + score = parsed.get("score", 0.5) # Set validation threshold validation_threshold = self.op_params.get("validation_threshold", 0.5) - + return { "is_valid": is_valid and score >= validation_threshold, "score": score, "feedback": response_content, - "reason": "" if (is_valid and score >= validation_threshold) else f"Low validation score ({score:.2f}) or marked as invalid" + "reason": "" if ( + is_valid and score >= validation_threshold) else f"Low validation score ({score:.2f}) or marked as invalid" } - + except Exception as e_inner: logger.exception(f"Error parsing validation response: {e_inner}") return { @@ -104,4 +105,4 @@ class MemoryValidationOp(BaseLLMOp): "score": 0.0, "feedback": "", "reason": f"LLM validation error: {str(e)}" - } \ No newline at end of file + } diff --git a/reme_ai/summary/task/pdf_preprocess_op_wrapper.py b/reme_ai/summary/task/pdf_preprocess_op_wrapper.py index 38556433..2c2f5f83 100644 --- a/reme_ai/summary/task/pdf_preprocess_op_wrapper.py +++ b/reme_ai/summary/task/pdf_preprocess_op_wrapper.py @@ -12,14 +12,14 @@ class PDFPreprocessOp(BaseOp): """Process PDF files using MinerU and chunk content""" pdf_path = self.context.get("pdf_path") output_dir = self.context.get("output_dir") - + if not pdf_path: logger.error("No PDF path provided in context") return # Process PDF processor = MinerUPDFProcessor(log_level="INFO") - + try: content_list, markdown_content = processor.process_pdf( pdf_path=pdf_path, @@ -28,21 +28,21 @@ class PDFPreprocessOp(BaseOp): lang=self.op_params.get("lang"), backend=self.op_params.get("backend", "pipeline") ) - + # Create chunks if requested chunks = [] if self.op_params.get("create_chunks", True): max_length = self.op_params.get("max_chunk_length", 4000) chunks = chunk_pdf_content(content_list, max_length=max_length) - + # Store results in context self.context.pdf_content_list = content_list self.context.pdf_markdown_content = markdown_content self.context.pdf_chunks = chunks - + logger.info(f"PDF processing completed: {len(content_list)} content blocks, " - f"{len(chunks)} chunks, {len(markdown_content)} characters of markdown") - + f"{len(chunks)} chunks, {len(markdown_content)} characters of markdown") + except Exception as e: logger.error(f"PDF processing failed: {e}") self.context.pdf_content_list = [] diff --git a/reme_ai/summary/task/simple_comparative_summary_op.py b/reme_ai/summary/task/simple_comparative_summary_op.py index e92b429a..9ef04617 100644 --- a/reme_ai/summary/task/simple_comparative_summary_op.py +++ b/reme_ai/summary/task/simple_comparative_summary_op.py @@ -32,9 +32,9 @@ class SimpleComparativeSummaryOp(BaseLLMOp): task_memory_content = tm_dict.get("experience", "").strip() if when_to_use and task_memory_content: task_memory_list.append(TaskMemory(workspace_id=self.context.get("workspace_id", ""), - when_to_use=when_to_use, + when_to_use=when_to_use, content=task_memory_content, - author=getattr(self.llm, 'model_name', 'system'))) + author=getattr(self.llm, 'model_name', 'system'))) return task_memory_list @@ -62,7 +62,7 @@ class SimpleComparativeSummaryOp(BaseLLMOp): if task_trajectories[0].score > task_trajectories[-1].score: task_memories = self.compare_summary_trajectory(trajectory_a=task_trajectories[0], - trajectory_b=task_trajectories[-1]) + trajectory_b=task_trajectories[-1]) memory_list.extend(task_memories) self.context.response.answer = json.dumps([x.model_dump() for x in memory_list]) diff --git a/reme_ai/summary/task/success_extraction_op.py b/reme_ai/summary/task/success_extraction_op.py index 7b72a335..d0585f28 100644 --- a/reme_ai/summary/task/success_extraction_op.py +++ b/reme_ai/summary/task/success_extraction_op.py @@ -15,7 +15,7 @@ class SuccessExtractionOp(BaseLLMOp): def execute(self): """Extract task memories from successful trajectories""" success_trajectories: List[Trajectory] = self.context.success_trajectories - + if not success_trajectories: logger.info("No success trajectories found for extraction") return @@ -23,7 +23,7 @@ class SuccessExtractionOp(BaseLLMOp): logger.info(f"Extracting task memories from {len(success_trajectories)} successful trajectories") success_task_memories = [] - + # Process trajectories for trajectory in success_trajectories: if "segments" in trajectory.metadata: diff --git a/reme_ai/summary/task/trajectory_preprocess_op.py b/reme_ai/summary/task/trajectory_preprocess_op.py index 5245aad1..9ada5865 100644 --- a/reme_ai/summary/task/trajectory_preprocess_op.py +++ b/reme_ai/summary/task/trajectory_preprocess_op.py @@ -1,4 +1,3 @@ -import json from typing import List, Dict from flowllm import C, BaseOp @@ -57,7 +56,7 @@ class TrajectoryPreprocessOp(BaseOp): def _modify_tool_calls(self, messages: List[Dict]) -> List[Dict]: new_messages = [] - + for msg in messages: if 'tool_calls' in msg: processed_tool_calls = [] @@ -72,5 +71,5 @@ class TrajectoryPreprocessOp(BaseOp): processed_tool_calls.append(tool_call) msg['tool_calls'] = processed_tool_calls new_messages.append(msg) - + return new_messages \ No newline at end of file diff --git a/reme_ai/summary/task/trajectory_segmentation_op.py b/reme_ai/summary/task/trajectory_segmentation_op.py index 8b2452e6..93a0a6c5 100644 --- a/reme_ai/summary/task/trajectory_segmentation_op.py +++ b/reme_ai/summary/task/trajectory_segmentation_op.py @@ -35,9 +35,8 @@ class TrajectorySegmentationOp(BaseLLMOp): segmented_count += 1 logger.info(f"Segmented {segmented_count} trajectories") - - # Update context with segmented trajectories + # Update context with segmented trajectories def _get_target_trajectories(self, all_trajectories: List[Trajectory], success_trajectories: List[Trajectory], @@ -116,4 +115,4 @@ class TrajectorySegmentationOp(BaseLLMOp): numbers = re.findall(r'\b\d+\b', response) segment_points = [int(num) for num in numbers if int(num) > 0] - return sorted(list(set(segment_points))) \ No newline at end of file + return sorted(list(set(segment_points))) diff --git a/reme_ai/utils/miner_u_pdf_processor.py b/reme_ai/utils/miner_u_pdf_processor.py index 2dd92b61..c2fbae43 100644 --- a/reme_ai/utils/miner_u_pdf_processor.py +++ b/reme_ai/utils/miner_u_pdf_processor.py @@ -231,7 +231,7 @@ class MinerUPDFProcessor: Returns: Tuple[List[Dict[str, Any]], str]: A tuple containing: - content_list: Structured content list with metadata - - markdown_content: Raw markdown text content + - markdown_content: Raw Markdown text content """ # Locate generated output files - handle both flat and nested directory structures md_file = output_dir / f"{file_stem}.md" @@ -277,7 +277,8 @@ class MinerUPDFProcessor: item[field_name] = str(absolute_img_path) self.logger.debug(f"Updated {field_name}: {img_path} -> {item[field_name]}") - self.logger.info(f"Successfully read JSON file: {json_file}, containing {len(content_list)} content blocks") + self.logger.info( + f"Successfully read JSON file: {json_file}, containing {len(content_list)} content blocks") except Exception as e: self.logger.warning(f"Failed to read JSON file {json_file}: {e}") @@ -466,32 +467,32 @@ class MinerUPDFProcessor: "table_count": 0, "has_formulas": False } - + for item in content_list: if not isinstance(item, dict): continue - + content_type = item.get("type", "unknown") stats["content_types"][content_type] = stats["content_types"].get(content_type, 0) + 1 - + if content_type == "text": text = item.get("text", "") stats["text_stats"]["total_characters"] += len(text) stats["text_stats"]["total_words"] += len(text.split()) - + level = item.get("text_level", 0) if level > 0: stats["text_stats"]["title_levels"][level] = stats["text_stats"]["title_levels"].get(level, 0) + 1 - + elif content_type == "image": stats["image_count"] += 1 - + elif content_type == "table": stats["table_count"] += 1 - + elif content_type == "formula": stats["has_formulas"] = True - + return stats def validate_output_quality(self, content_list: List[Dict[str, Any]], markdown_content: str) -> Dict[str, Any]: @@ -518,20 +519,20 @@ class MinerUPDFProcessor: "suggestions": [], "quality_score": 100 } - + # Check if content was extracted if not content_list and not markdown_content.strip(): validation["is_valid"] = False validation["warnings"].append("No content was extracted from the PDF") validation["quality_score"] = 0 return validation - + # Check content diversity stats = self.get_content_statistics(content_list) if stats["total_blocks"] < 5: validation["warnings"].append("Very few content blocks extracted - document may be complex or image-heavy") validation["quality_score"] -= 20 - + # Check text content ratio text_blocks = stats["content_types"].get("text", 0) if text_blocks == 0: @@ -540,16 +541,17 @@ class MinerUPDFProcessor: elif text_blocks / stats["total_blocks"] < 0.3: validation["suggestions"].append("Low text content ratio - document may benefit from OCR processing") validation["quality_score"] -= 10 - + # Check for images without processing if stats["image_count"] > 0 and stats["content_types"].get("text", 0) == 0: - validation["suggestions"].append("Images detected but no text extracted - consider using VLM backend for image analysis") - + validation["suggestions"].append( + "Images detected but no text extracted - consider using VLM backend for image analysis") + # Check markdown length vs content blocks if len(markdown_content.strip()) < 100 and stats["total_blocks"] > 10: validation["warnings"].append("Markdown content seems unusually short for the number of content blocks") validation["quality_score"] -= 15 - + return validation @@ -656,22 +658,23 @@ if __name__ == "__main__": and working with the extracted content. """ import sys - + + # Example usage def example_usage(): """Demonstrate basic PDF processing workflow.""" try: # Initialize processor processor = MinerUPDFProcessor.create_with_defaults(log_level="INFO") - + # Example PDF path (replace with actual PDF file) pdf_path = "example_document.pdf" - + if not Path(pdf_path).exists(): print(f"Example PDF file not found: {pdf_path}") print("Please provide a valid PDF file path to test the processor.") return - + # Process PDF with different methods print("Processing PDF with auto method...") content_list, markdown_content = processor.process_pdf( @@ -679,7 +682,7 @@ if __name__ == "__main__": method="auto", lang="en" # Specify language for better OCR results ) - + # Generate statistics stats = processor.get_content_statistics(content_list) print(f"Processing Statistics:") @@ -687,7 +690,7 @@ if __name__ == "__main__": print(f" Content types: {stats['content_types']}") print(f" Text characters: {stats['text_stats']['total_characters']}") print(f" Text words: {stats['text_stats']['total_words']}") - + # Validate output quality validation = processor.validate_output_quality(content_list, markdown_content) print(f"Quality Score: {validation['quality_score']}/100") @@ -695,7 +698,7 @@ if __name__ == "__main__": print("Warnings:", validation['warnings']) if validation['suggestions']: print("Suggestions:", validation['suggestions']) - + # Save results output_path = Path(pdf_path).stem + "_processed" saved_files = processor.save_results( @@ -704,19 +707,20 @@ if __name__ == "__main__": output_path=output_path ) print(f"Results saved to: {saved_files}") - + # Create text chunks for downstream processing chunks = chunk_pdf_content(content_list, max_length=2000) print(f"Created {len(chunks)} text chunks") - + # Display first chunk as example if chunks: print("First chunk preview:") print(chunks[0][:200] + "..." if len(chunks[0]) > 200 else chunks[0]) - + except Exception as e: print(f"Error during processing: {e}") sys.exit(1) - + + # Run example if script is executed directly example_usage() diff --git a/reme_ai/utils/op_utils.py b/reme_ai/utils/op_utils.py index 32a7caf6..6ecb0ead 100644 --- a/reme_ai/utils/op_utils.py +++ b/reme_ai/utils/op_utils.py @@ -10,6 +10,7 @@ from loguru import logger def merge_messages_content(messages: List[Message | dict]) -> str: return merge_messages_content_flowllm(messages) + def parse_json_experience_response(response: str) -> List[dict]: """Parse JSON formatted experience response""" try: @@ -82,155 +83,6 @@ def get_trajectory_context(trajectory: Trajectory, step_sequence: List[Message]) logger.error(f"Error getting trajectory context: {e}") return f"Query: {trajectory.metadata.get('query', 'N/A')}" - -def parse_observation_response(response_text: str) -> List[dict]: - """Parse observation response to extract structured data""" - # Pattern to match both Chinese and English observation formats - pattern = r"信息:<(\d+)>\s*<>\s*<([^<>]+)>\s*<([^<>]*)>|Information:\s*<(\d+)>\s*<>\s*<([^<>]+)>\s*<([^<>]*)>" - matches = re.findall(pattern, response_text, re.IGNORECASE | re.MULTILINE) - - observations = [] - for match in matches: - # Handle both Chinese and English patterns - if match[0]: # Chinese pattern - idx_str, content, keywords = match[0], match[1], match[2] - else: # English pattern - idx_str, content, keywords = match[3], match[4], match[5] - - try: - idx = int(idx_str) - # Skip if content indicates no meaningful observation - content_lower = content.lower().strip() - if content_lower not in ['无', 'none', '', 'repeat']: - observations.append({ - "index": idx, - "content": content.strip(), - "keywords": keywords.strip() if keywords else "" - }) - except ValueError: - logger.warning(f"Invalid index format: {idx_str}") - continue - - return observations - - -def parse_observation_with_time_response(response_text: str) -> List[dict]: - """Parse observation with time response to extract structured data""" - # Pattern to match both Chinese and English observation formats with time information - # Chinese: 信息:<1> <时间信息或不输出> <明确的重要信息或"无"> <关键词> - # English: Information: <1>