mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-10-11 03:40:03 +00:00
* feat(service): add CLI service for local job execution - Introduce CliService to execute single jobs locally without serving ports - Add prepare_start_config and should_precheck_start functions for CLI job setup - Update reme start command to use CLI service when job argument is provided - Change default service backend from http to cli in jinli_lme config - Modify SearchStep to use constants and rename configuration parameters - Add unit tests for CLI service functionality and configuration handling - Update file extension support to include json format in addition to md and jsonl * feat(search): add BM25 and vector search steps with configuration updates - Add Bm25SearchStep and VectorSearchStep classes with tool context deduplication - Register new search step components in index module - Update configuration to use separate vector_search and bm25_search endpoints - Modify LLM models from qwen3.7-plus/glm-5.1 to glm-5.2 variants - Adjust search parameters and remove hybrid search implementation - Configure embedding store as default in storage settings - Remove auto-memory and file catalog configurations - Update watch directories from multiple paths to session_dir only * feat(agent): add tool result offloading and workspace management - Add tool_results_dir configuration option for offloaded tool results storage - Implement ToolResultOffloadMiddleware to persist large tool results to files - Create WorkspaceBackend to standardize file operations across tools - Add configurable builtin tools selection with sequential execution option - Integrate middleware support for agent wrapper with offloading capability - Update application initialization to create tool results directory - Add safety mechanisms for filesystem operations with sanitized filenames - Enhance agent wrapper with configurable working directory handling - Upgrade agentscope dependency to version 2.0.4 for improved features # Conflicts: # reme/application.py * feat(benchmark): add LongMemEval agentic search and result management - Introduce AgenticAnswerStep for agent-based history search - Add LmePrepareJudgeStep and LmeSaveResultStep for evaluation pipeline - Implement AddDraftStep and ReadAllDraftStep for evidence accumulation - Update configuration with new agent wrapper and search parameters - Add comparison script for analyzing agent run differences - Include documentation for LongMemEval failure analysis - Enhance tool result offloading with skip options - Modify search defaults and indexing behavior * feat(agent): implement tool result offloading with system reminders - Added tool_result_offload_message parameter to agent wrapper reply method - Implemented configurable reminder template for offloaded tool results - Created system reminder messages when tool results are offloaded to files - Added Chinese user message template for agentic answer step - Updated tool result offloading middleware to use custom reminder templates - Enhanced agentic answer instructions to handle long tool results via draft storage * feat(scripts): add LongMemEval results summarization tool - Create summarize_lme_results.py script to analyze result JSON files - Implement command line interface with answer id and dataset root options - Add support for specifying index range with start and end parameters - Include option to show failure details and non-successful completions - Calculate completion statistics and accuracy metrics - Display detailed breakdown of yes/no/other judgements - Handle missing and unreadable result files gracefully - Format output with percentages and comprehensive summary statistics * feat(summarize_lme_results): add question type breakdown to result summary - Import defaultdict from collections module - Add by_type dictionary to track statistics by question type - Count completed, yes, no, and other responses for each question type - Display detailed breakdown table showing accuracy by question type - Include question type column when processing judgements - Print comprehensive summary with question type distribution - Calculate and display accuracy percentage for each question type category * feat(lme): switch to qwen3.7-max model and add shuffle functionality - Changed default LLM model from glm-5.1 to qwen3.7-max in jinli_lme.yaml - Added random module import for shuffle functionality - Implemented --shuffle argument with BooleanOptionalAction for dataset shuffling - Added --seed argument to control random seed for reproducible shuffling - Applied random shuffle to dataset indices when shuffle is enabled - Added console output showing shuffle operation and seed information * fix(cli): set default random seed for shuffle functionality - Changed default seed value from None to 42 for consistent shuffling behavior - Ensures reproducible results when using shuffle option without explicit seed - Maintains backward compatibility while providing deterministic defaults * refactor(benchmark): update agentic answer guidelines for grounding - Updated English instruction to emphasize strict grounding in retrieved context - Modified Chinese instruction to stress evidence-based responses without inference - Removed redundant conciseness requirement in both language versions - Enhanced clarity on proper use of draft saving and retrieval mechanisms - Strengthened emphasis against hallucination of unsupported facts * refactor(benchmark): update agentic search instructions and configuration - Replace separate vector_search and bm25_search with unified search tool - Update agent instructions to use single search tool with multiple strategies - Simplify Chinese instructions for search methodology - Add comprehensive search tool configuration with hybrid vector/BM25 capabilities - Increase model retry attempts from 1 to 3 for better reliability - Remove redundant tool references from job_tools list * feat(search): add configurable search limit with environment variable support - Remove hardcoded limit and min_score parameters from config schema - Increase LLM context size from 200000 to 1000000 - Add REME_SEARCH_LIMIT environment variable support for search configuration - Implement command line argument --search-limit to override default search limit - Add input validation to ensure search limit is positive - Modify subprocess execution to pass environment variables - Update search step to use dynamic default limit from environment or fallback to 5 * refactor(benchmark): remove agentic answer step and related configurations - Removed AgenticAnswerStep class and its registration - Deleted agentic_answer.yaml prompt configuration file - Removed agentic answer related job definitions from jinli_lme.yaml - Cleaned up tool result offloading middleware implementation - Removed tool_results_dir configuration field from application config - Deleted comparison and analysis scripts for agent runs - Removed agentic answer step from LME init module exports - Updated agent wrapper to remove tool result offloading functionality - Removed unused imports and dependencies in agent wrapper module * refactor(benchmark): remove unused LME result processing components - Removed LmePrepareJudgeStep and LmeSaveResultStep classes from benchmark module - Cleaned up imports and exports in lme module initialization - Removed unused middleware configuration from agent wrapper - Deleted obsolete result.py file containing deprecated result processing logic - Simplified agent instantiation by removing middleware parameter - Updated import statements to reflect removed dependencies * refactor(index): remove unused search steps and update imports - Remove Bm25SearchStep and VectorSearchStep from index steps module - Remove unused prepare_start_config and should_precheck_start exports - Move import statements to proper location in reme.py - Update test module to use direct import path for CliService - Remove vector_search and bm25_search configurations from jinli_lme.yaml - Add workspace directory environment variable configuration - Add docstring to getcwd method in agent wrapper - Remove empty middleware list from agent wrapper initialization * feat(index): add BM25 and vector search steps with tool context deduplication - Add Bm25SearchStep for plain BM25 keyword search with tool_context deduplication - Add VectorSearchStep for plain vector search with tool_context deduplication - Implement tool context state management with TTL-based deduplication - Add support for chunk deduplication across tool contexts within TTL window - Update index steps module to include new search step classes - Add test coverage for CLI metadata output functionality - Refactor CLI service to remove unused show_status parameter - Update documentation comments to reflect internal service configuration * feat(steps): add Python code execution capability - Introduce PythonExecuteStep to run Python code in subprocess - Add configuration for python_execute step in jinli_lme.yaml - Register python_execute in available tools list - Implement timeout handling with default 60 second limit - Capture stdout/stderr output and return code metadata - Add comprehensive unit tests for execution scenarios - Support workspace directory context for code execution - Handle timeout errors and runtime exceptions gracefully * refactor(python_execute): replace subprocess with asyncio for Python code execution - Replace subprocess.run with asyncio.create_subprocess_exec for non-blocking execution - Add _PythonResult dataclass to encapsulate execution results and timeout status - Implement proper timeout handling with asyncio.wait_for and process.kill() - Update metadata to include returncode and stderr when timeout occurs - Convert synchronous _run_python method to asynchronous implementation - Maintain backward compatibility while improving execution reliability * refactor(python_execute): replace subprocess with asyncio for Python code execution - Replace subprocess.run with asyncio.create_subprocess_exec for non-blocking execution - Add _PythonResult dataclass to encapsulate execution results and timeout status - Implement proper timeout handling with asyncio.wait_for and process.kill() - Update metadata to include returncode and stderr when timeout occurs - Convert synchronous _run_python method to asynchronous implementation - Maintain backward compatibility while improving execution reliability
337 lines
11 KiB
Python
337 lines
11 KiB
Python
"""Tests for reme common steps with only local dependencies."""
|
|
|
|
# pylint: disable=protected-access
|
|
|
|
import asyncio
|
|
import os
|
|
import tempfile
|
|
import warnings
|
|
|
|
from reme.components.agent_wrapper import BaseAgentWrapper
|
|
from reme.components.application_context import ApplicationContext
|
|
from reme.components.file_store import LocalFileStore
|
|
from reme.schema import FileLink, FileNode
|
|
from reme.steps.common.add import AddStep
|
|
from reme.steps.common.health_check import _file_graph_status
|
|
from reme.steps.common.llm_demo import LLMDemoStep
|
|
from reme.steps.common.python_execute import PythonExecuteStep
|
|
from reme.steps.index import traverse as traverse_mod
|
|
|
|
warnings.filterwarnings("ignore", category=DeprecationWarning, module="jieba")
|
|
warnings.filterwarnings("ignore", category=DeprecationWarning, module="pkg_resources")
|
|
|
|
|
|
class _temp_chdir:
|
|
"""chdir to path for the duration of the block; restore on exit."""
|
|
|
|
def __init__(self, path):
|
|
self.path = path
|
|
self._old = None
|
|
|
|
def __enter__(self):
|
|
self._old = os.getcwd()
|
|
os.chdir(self.path)
|
|
return self
|
|
|
|
def __exit__(self, *exc):
|
|
os.chdir(self._old)
|
|
|
|
|
|
def _run(coro):
|
|
"""Run an async coroutine on a fresh isolated event loop."""
|
|
asyncio.run(coro)
|
|
|
|
|
|
def _node(path: str, links: list[tuple[str, str | None, str | None]] | None = None) -> FileNode:
|
|
"""Build a FileNode with (target_path, target_anchor, predicate) outgoing edges."""
|
|
return FileNode(
|
|
path=path,
|
|
st_mtime=1.0,
|
|
links=[FileLink(source_path=path, target_path=t, target_anchor=a, predicate=p) for t, a, p in (links or [])],
|
|
)
|
|
|
|
|
|
async def _make_store(nodes: list[FileNode]) -> LocalFileStore:
|
|
"""LocalFileStore seeded with the given graph nodes (no files on disk)."""
|
|
store = LocalFileStore(name="t", embedding_store="")
|
|
await store.start()
|
|
if nodes:
|
|
await store.file_graph.upsert_nodes(nodes)
|
|
return store
|
|
|
|
|
|
def _edges(step) -> list[dict]:
|
|
return step.context.response.metadata.get("edges", [])
|
|
|
|
|
|
def test_add_step_coerces_numeric_inputs():
|
|
"""add accepts numeric strings as numbers, not string concatenation."""
|
|
|
|
async def run():
|
|
step = AddStep()
|
|
resp = await step(a="1", b="2.5")
|
|
assert resp.success is True
|
|
assert resp.answer == "3.5"
|
|
assert resp.metadata["result"] == 3.5
|
|
print("✓ test_add_step_coerces_numeric_inputs passed")
|
|
|
|
_run(run())
|
|
|
|
|
|
def test_add_step_rejects_invalid_inputs():
|
|
"""invalid add arguments should return a failed response instead of throwing or concatenating."""
|
|
|
|
async def run():
|
|
step = AddStep()
|
|
resp = await step(a="one", b=2)
|
|
assert resp.success is False
|
|
assert "Invalid add arguments" in resp.answer
|
|
print("✓ test_add_step_rejects_invalid_inputs passed")
|
|
|
|
_run(run())
|
|
|
|
|
|
def test_python_execute_step_returns_printed_stdout(tmp_path):
|
|
"""python_execute returns stdout as answer and runs under workspace_dir."""
|
|
|
|
async def run():
|
|
app_context = ApplicationContext(workspace_dir=str(tmp_path))
|
|
step = PythonExecuteStep(app_context=app_context)
|
|
resp = await step(code="from pathlib import Path\nprint(Path.cwd().name)")
|
|
assert resp.success is True
|
|
assert resp.answer == f"{tmp_path.name}\n"
|
|
assert resp.metadata["returncode"] == 0
|
|
assert resp.metadata["stderr"] == ""
|
|
print("✓ test_python_execute_step_returns_printed_stdout passed")
|
|
|
|
_run(run())
|
|
|
|
|
|
def test_python_execute_step_reports_stderr_on_failure():
|
|
"""python_execute captures traceback stderr instead of throwing."""
|
|
|
|
async def run():
|
|
step = PythonExecuteStep()
|
|
resp = await step(code='raise RuntimeError("boom")')
|
|
assert resp.success is False
|
|
assert resp.metadata["returncode"] != 0
|
|
assert "RuntimeError: boom" in resp.answer
|
|
assert "RuntimeError: boom" in resp.metadata["stderr"]
|
|
print("✓ test_python_execute_step_reports_stderr_on_failure passed")
|
|
|
|
_run(run())
|
|
|
|
|
|
def test_python_execute_step_times_out():
|
|
"""python_execute converts subprocess timeout into a failed response."""
|
|
|
|
async def run():
|
|
step = PythonExecuteStep()
|
|
resp = await step(code="import time\ntime.sleep(1)", timeout=0.01)
|
|
assert resp.success is False
|
|
assert resp.answer == "Python execution timed out after 0.01s"
|
|
assert resp.metadata["timeout"] == 0.01
|
|
print("✓ test_python_execute_step_times_out passed")
|
|
|
|
_run(run())
|
|
|
|
|
|
class _FakeAgentWrapper(BaseAgentWrapper):
|
|
"""Capture reply kwargs without calling a real model."""
|
|
|
|
def __init__(self):
|
|
super().__init__()
|
|
self.last_kwargs = None
|
|
|
|
async def reply(self, inputs, **kwargs) -> dict:
|
|
self.last_kwargs = kwargs
|
|
return {"result": "ok"}
|
|
|
|
|
|
def test_llm_demo_always_registers_add_tool():
|
|
"""LLM demo always passes the add job as a tool."""
|
|
|
|
async def run():
|
|
wrapper = _FakeAgentWrapper()
|
|
step = LLMDemoStep()
|
|
resp = await step(query="hello", agent_wrapper=wrapper)
|
|
assert resp.success is True
|
|
assert wrapper.last_kwargs["job_tools"] == ["add"]
|
|
assert "job_tools" not in resp.metadata
|
|
print("✓ test_llm_demo_always_registers_add_tool passed")
|
|
|
|
_run(run())
|
|
|
|
|
|
def test_file_graph_health_reports_neo4j_cached_counts():
|
|
"""Neo4j file graph health should not be reported as an empty local graph."""
|
|
|
|
class FakeNeo4jGraph:
|
|
"""Minimal Neo4j graph stub with cached health counters."""
|
|
|
|
is_started = True
|
|
_driver = object()
|
|
_uri = "bolt://example"
|
|
_database = "neo4j"
|
|
_n_nodes = 3
|
|
_n_edges = 4
|
|
_n_virtual = 1
|
|
|
|
status = _file_graph_status(FakeNeo4jGraph())
|
|
assert status["n_nodes"] == 3
|
|
assert status["n_edges"] == 4
|
|
assert status["n_virtual"] == 1
|
|
print("✓ test_file_graph_health_reports_neo4j_cached_counts passed")
|
|
|
|
|
|
# ===========================================================================
|
|
# Direct unit tests: TraverseStep
|
|
# (LocalFileStore, no HTTP server — BFS over wikilink edges)
|
|
# ===========================================================================
|
|
|
|
|
|
def test_traverse_forward_depth_1():
|
|
"""depth=1 forward returns direct outbound neighbors."""
|
|
|
|
async def run():
|
|
with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp):
|
|
store = await _make_store(
|
|
[
|
|
_node("a.md", [("b.md", None, None), ("c.md", "intro", "ref")]),
|
|
_node("b.md"),
|
|
_node("c.md"),
|
|
],
|
|
)
|
|
step = traverse_mod.TraverseStep(file_store=store)
|
|
await step(path="a.md", direction="forward", depth=1)
|
|
results = _edges(step)
|
|
paths = {r["path"] for r in results}
|
|
assert paths == {"b.md", "c.md"}
|
|
# The 'ref' edge should report its predicate/anchor.
|
|
c_edge = next(r for r in results if r["path"] == "c.md")
|
|
assert c_edge["predicate"] == "ref"
|
|
assert c_edge["anchor"] == "intro"
|
|
assert c_edge["via"] == "a.md"
|
|
assert c_edge["depth"] == 1
|
|
await store.close()
|
|
print("✓ test_traverse_forward_depth_1 passed")
|
|
|
|
asyncio.run(run())
|
|
|
|
|
|
def test_traverse_backward_returns_inlinks():
|
|
"""direction=backward walks inbound edges."""
|
|
|
|
async def run():
|
|
with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp):
|
|
store = await _make_store(
|
|
[
|
|
_node("a.md", [("b.md", None, None)]),
|
|
_node("c.md", [("b.md", None, None)]),
|
|
_node("b.md"),
|
|
],
|
|
)
|
|
step = traverse_mod.TraverseStep(file_store=store)
|
|
await step(path="b.md", direction="backward", depth=1)
|
|
results = _edges(step)
|
|
assert {r["path"] for r in results} == {"a.md", "c.md"}
|
|
await store.close()
|
|
print("✓ test_traverse_backward_returns_inlinks passed")
|
|
|
|
asyncio.run(run())
|
|
|
|
|
|
def test_traverse_depth_2_expands():
|
|
"""depth=2 traverses one hop beyond direct neighbors."""
|
|
|
|
async def run():
|
|
with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp):
|
|
store = await _make_store(
|
|
[
|
|
_node("a.md", [("b.md", None, None)]),
|
|
_node("b.md", [("c.md", None, None)]),
|
|
_node("c.md"),
|
|
],
|
|
)
|
|
step = traverse_mod.TraverseStep(file_store=store)
|
|
await step(path="a.md", direction="forward", depth=2)
|
|
results = _edges(step)
|
|
depth_map = {r["path"]: r["depth"] for r in results}
|
|
assert depth_map.get("b.md") == 1
|
|
assert depth_map.get("c.md") == 2
|
|
await store.close()
|
|
print("✓ test_traverse_depth_2_expands passed")
|
|
|
|
asyncio.run(run())
|
|
|
|
|
|
def test_traverse_short_seed_yields_empty():
|
|
"""A short (not relative to the workspace) seed isn't resolved anymore — BFS simply
|
|
finds no edges from a path that doesn't match any graph node."""
|
|
|
|
async def run():
|
|
with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp):
|
|
store = await _make_store(
|
|
[
|
|
_node("topics/Bob.md"),
|
|
_node("people/Bob.md"),
|
|
],
|
|
)
|
|
step = traverse_mod.TraverseStep(file_store=store)
|
|
await step(path="Bob", direction="forward", depth=1)
|
|
payload = _edges(step)
|
|
# No error, just empty results because "Bob" isn't a graph key.
|
|
assert payload == []
|
|
await store.close()
|
|
print("✓ test_traverse_short_seed_yields_empty passed")
|
|
|
|
asyncio.run(run())
|
|
|
|
|
|
def test_traverse_not_found_seed():
|
|
"""A seed not in the graph returns an empty list (no error)."""
|
|
|
|
async def run():
|
|
with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp):
|
|
store = await _make_store([_node("a.md")])
|
|
step = traverse_mod.TraverseStep(file_store=store)
|
|
await step(path="topics/ghost.md", direction="forward", depth=1)
|
|
payload = _edges(step)
|
|
assert payload == []
|
|
await store.close()
|
|
print("✓ test_traverse_not_found_seed passed")
|
|
|
|
asyncio.run(run())
|
|
|
|
|
|
def test_traverse_both_directions():
|
|
"""direction=both walks out- and in-bound; depth=1 returns one hop in each direction."""
|
|
|
|
async def run():
|
|
with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp):
|
|
store = await _make_store(
|
|
[
|
|
_node("upstream.md", [("center.md", None, None)]),
|
|
_node("center.md", [("downstream.md", None, None)]),
|
|
_node("downstream.md"),
|
|
],
|
|
)
|
|
step = traverse_mod.TraverseStep(file_store=store)
|
|
await step(path="center.md", direction="both", depth=1)
|
|
results = _edges(step)
|
|
assert {r["path"] for r in results} == {"upstream.md", "downstream.md"}
|
|
await store.close()
|
|
print("✓ test_traverse_both_directions passed")
|
|
|
|
asyncio.run(run())
|
|
|
|
|
|
if __name__ == "__main__":
|
|
print("\n=== traverse step tests ===")
|
|
test_traverse_forward_depth_1()
|
|
test_traverse_backward_returns_inlinks()
|
|
test_traverse_depth_2_expands()
|
|
test_traverse_short_seed_yields_empty()
|
|
test_traverse_not_found_seed()
|
|
test_traverse_both_directions()
|
|
print("\n所有测试通过!")
|