ReMe/reme4/steps/index/update_index.py
jinliyl a4efc0f776
refactor(reme4): restructure steps packages (#258)
* fix(bm25_index): 修正BM25索引计算中的文档长度归一化问题

修复了在计算BM25相似度时对文档长度进行不正确归一化的bug,确保所有查询都能得到准确的相关性评分。

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* refactor(steps): Rename and adjust indexing step logic

- Rename `scan_changes.py` and `reindex.py` to `clear_and_scan.py`
- Update implementation details of `ScanChangesStep` and `ClearAndScanStep`
- Modify the scheduling mechanism in `WatchChangesStep`
- Adjust step registration and parameter configuration in config files
- Update related tests to align with the new interface changes

* up

* feat(daily): replace daily CRUD operations with slug provisioning approach

* refactor(tests): migrate CRUD step tests from HTTP server to direct LocalFileStore

* up

* up

* up

* up

---------

Co-authored-by: huangsen <huangsen.huang@alibaba-inc.com>
2026-05-28 14:30:30 +08:00

88 lines
3.8 KiB
Python

"""Update index with a batch of file changes."""
from pathlib import Path
from watchfiles import Change
from ..base_step import BaseStep
from ...components import R
from ...schema import FileChunk, FileNode
@R.register("update_index_step")
class UpdateIndexStep(BaseStep):
"""Classify raw watcher changes and update the file_store index."""
def __init__(self, persist: bool = False, **kwargs):
super().__init__(**kwargs)
self.persist: bool = persist
async def execute(self):
assert self.context is not None
# Each item: {"change": Change | "added"|"modified"|"deleted", "path": absolute path}
changes: list[dict] = self.context.get("changes") or []
buckets: dict[Change, list[str]] = {Change.added: [], Change.modified: [], Change.deleted: []}
for item in changes:
c = item["change"]
if isinstance(c, str):
c = Change.__members__.get(c)
if isinstance(c, Change) and c in buckets:
buckets[c].append(item["path"])
results: list[dict] = []
for change, action in ((Change.added, "Adding"), (Change.modified, "Updating")):
paths = buckets[change]
if not paths:
continue
self.logger.info(f"Detected {len(paths)} {change.name} file(s)")
parsed: list[tuple[FileNode, list[FileChunk]]] = []
ok_paths: list[str] = []
for path in paths:
abs_path = Path(path)
if not abs_path.is_file():
results.append({"change": change.name, "path": path, "success": False, "error": "not a file"})
continue
self.logger.info(f"{action} file: {path}")
try:
parsed.append(await self.parse_file(abs_path))
ok_paths.append(path)
except Exception as e:
self.logger.exception(f"Failed to parse {path}")
results.append({"change": change.name, "path": path, "success": False, "error": str(e)})
if parsed:
try:
await self.file_store.delete([n.path for n, _ in parsed])
await self.file_store.upsert(parsed)
results.extend({"change": change.name, "path": p, "success": True} for p in ok_paths)
except Exception as e:
self.logger.exception(f"Failed to persist {len(parsed)} {change.name} file(s)")
results.extend(
{"change": change.name, "path": p, "success": False, "error": str(e)} for p in ok_paths
)
if deleted := buckets[Change.deleted]:
if self.file_store is None:
raise RuntimeError("file_store is not initialized!")
self.logger.info(f"Detected {len(deleted)} deleted file(s)")
rel_deleted: list[str] = []
for path in deleted:
p = Path(path).absolute()
try:
rel_deleted.append(str(p.relative_to(self.vault_path)))
except ValueError:
rel_deleted.append(str(p))
try:
await self.file_store.delete(rel_deleted)
results.extend({"change": "deleted", "path": p, "success": True} for p in deleted)
except Exception as e:
self.logger.exception(f"Failed to delete {len(deleted)} file(s)")
results.extend({"change": "deleted", "path": p, "success": False, "error": str(e)} for p in deleted)
if self.persist and results:
await self.file_store.dump()
self.context.response.answer = results
self.context.response.success = all(r["success"] for r in results) if results else True
return self.context.response