mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-30 01:52:29 +00:00
* fix(bm25_index): 修正BM25索引计算中的文档长度归一化问题 修复了在计算BM25相似度时对文档长度进行不正确归一化的bug,确保所有查询都能得到准确的相关性评分。 * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * refactor(steps): Rename and adjust indexing step logic - Rename `scan_changes.py` and `reindex.py` to `clear_and_scan.py` - Update implementation details of `ScanChangesStep` and `ClearAndScanStep` - Modify the scheduling mechanism in `WatchChangesStep` - Adjust step registration and parameter configuration in config files - Update related tests to align with the new interface changes * up * feat(daily): replace daily CRUD operations with slug provisioning approach * refactor(tests): migrate CRUD step tests from HTTP server to direct LocalFileStore * up * up * up * up --------- Co-authored-by: huangsen <huangsen.huang@alibaba-inc.com>
67 lines
2.6 KiB
Python
67 lines
2.6 KiB
Python
"""One-shot scan: diff watch_paths vs file_store and write changes into context.
|
|
|
|
Designed to be chained before ``update_index_step`` so that the second step
|
|
performs the actual writes and persistence.
|
|
"""
|
|
|
|
from pathlib import Path
|
|
|
|
from ..base_step import BaseStep
|
|
from ...components import R
|
|
|
|
|
|
@R.register("scan_changes_step")
|
|
class ScanChangesStep(BaseStep):
|
|
"""One-shot scan: compute added/modified/deleted vs file_store and write to context."""
|
|
|
|
def __init__(self, recursive: bool = True, **kwargs):
|
|
super().__init__(**kwargs)
|
|
self.recursive: bool = recursive
|
|
|
|
async def execute(self):
|
|
assert self.context is not None
|
|
if self.file_store is None:
|
|
raise RuntimeError("file_store is not initialized!")
|
|
|
|
raw: list[str] = self.context.get("watch_paths", [])
|
|
suffixes: list[str] = self.context.get("suffix_filters", ["md"])
|
|
vault_path = self.vault_path
|
|
|
|
paths = [raw] if isinstance(raw, str) else raw
|
|
watch_paths = [vault_path / x for x in paths if (vault_path / x).exists()]
|
|
|
|
existing: dict[str, float] = {}
|
|
for path in watch_paths:
|
|
candidates = [path] if path.is_file() else (path.rglob("*") if self.recursive else path.iterdir())
|
|
for p in candidates:
|
|
if not p.is_file():
|
|
continue
|
|
if suffixes and not any(str(p).endswith("." + s.strip(".")) for s in suffixes):
|
|
continue
|
|
abs_p = p.absolute()
|
|
existing[str(abs_p)] = abs_p.stat().st_mtime
|
|
|
|
indexed: dict[str, float] = {
|
|
str(Path(n.path) if Path(n.path).is_absolute() else vault_path / n.path): n.st_mtime
|
|
for n in await self.file_store.get_nodes()
|
|
}
|
|
|
|
to_delete = list(indexed.keys() - existing.keys())
|
|
to_add = list(existing.keys() - indexed.keys())
|
|
to_modify = [p for p in existing.keys() & indexed.keys() if existing[p] != indexed[p]]
|
|
|
|
changes: list[dict] = (
|
|
[{"change": "added", "path": p} for p in to_add]
|
|
+ [{"change": "modified", "path": p} for p in to_modify]
|
|
+ [{"change": "deleted", "path": p} for p in to_delete]
|
|
)
|
|
counts = {"added": len(to_add), "modified": len(to_modify), "deleted": len(to_delete)}
|
|
|
|
self.context["changes"] = changes
|
|
if changes:
|
|
self.logger.info(f"[{self.name}] scan: {counts}")
|
|
else:
|
|
self.logger.info(f"[{self.name}] store is up to date")
|
|
|
|
self.context.response.metadata["counts"] = counts
|
|
return self.context.response
|