ReMe/reme4/steps/index/traverse.py
jinliyl a4efc0f776
refactor(reme4): restructure steps packages (#258)
* fix(bm25_index): 修正BM25索引计算中的文档长度归一化问题

修复了在计算BM25相似度时对文档长度进行不正确归一化的bug,确保所有查询都能得到准确的相关性评分。

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* up

* refactor(steps): Rename and adjust indexing step logic

- Rename `scan_changes.py` and `reindex.py` to `clear_and_scan.py`
- Update implementation details of `ScanChangesStep` and `ClearAndScanStep`
- Modify the scheduling mechanism in `WatchChangesStep`
- Adjust step registration and parameter configuration in config files
- Update related tests to align with the new interface changes

* up

* feat(daily): replace daily CRUD operations with slug provisioning approach

* refactor(tests): migrate CRUD step tests from HTTP server to direct LocalFileStore

* up

* up

* up

* up

---------

Co-authored-by: huangsen <huangsen.huang@alibaba-inc.com>
2026-05-28 14:30:30 +08:00

111 lines
4.1 KiB
Python

"""BFS over wikilink edges from one or more seed files.
One record per traversed *edge* (not per node): the same target can repeat
if reached via different predicates or paths. Each record carries the
predecessor plus the link's predicate/anchor so callers can reconstruct
the path. Adjacency is built once via a single ``file_store.get_nodes()``
call — BFS then runs purely in memory with no per-frontier round-trips.
"""
from collections import deque
from pathlib import Path
from ..base_step import BaseStep
from ...components import R
from ...schema import FileLink
_OUT = {"out", "forward", "both"}
_IN = {"in", "backward", "both"}
_VALID = _OUT | _IN
# source path -> list of (neighbor path, link)
Adjacency = dict[str, list[tuple[str, FileLink]]]
async def _build_adjacency(file_store) -> tuple[Adjacency, Adjacency]:
"""Single ``get_nodes()`` pass → (outbound, inbound) adjacency maps.
Inbound stores the source path next to each link so BFS can attribute
inbound edges back to their origin — ``get_inlinks`` alone returns
target-shaped FileLinks without source attribution.
"""
outbound: Adjacency = {}
inbound: Adjacency = {}
for node in await file_store.get_nodes():
for link in node.links:
if link.target_path:
outbound.setdefault(node.path, []).append((link.target_path, link))
inbound.setdefault(link.target_path, []).append((node.path, link))
return outbound, inbound
def _bfs(
seeds: list[str],
max_depth: int,
direction: str,
outbound: Adjacency,
inbound: Adjacency,
) -> list[dict]:
"""In-memory BFS; emits one record per unique (src, dst, predicate) edge."""
sources: list[Adjacency] = []
if direction in _OUT:
sources.append(outbound)
if direction in _IN:
sources.append(inbound)
visited: set[tuple[str, str, str | None]] = set()
results: list[dict] = []
queue: deque[tuple[str, int]] = deque((s, 0) for s in seeds)
while queue:
current, depth = queue.popleft()
if depth >= max_depth:
continue
for src in sources:
for next_path, link in src.get(current, ()):
key = (current, next_path, link.predicate)
if key in visited:
continue
visited.add(key)
results.append(
{
"path": next_path,
"depth": depth + 1,
"via": current,
"predicate": link.predicate,
"anchor": link.target_anchor,
},
)
if depth + 1 < max_depth:
queue.append((next_path, depth + 1))
return results
@R.register("traverse_step")
class TraverseStep(BaseStep):
"""BFS from one or more seed files to explore wikilink relationships.
Parameters:
path — single seed (str) or list of seeds (vault-relative).
direction — ``forward`` / ``backward`` / ``both`` (or ``out`` / ``in`` / ``both``).
depth — hop limit (default 1 = immediate neighbors).
"""
async def execute(self):
assert self.context is not None
raw = self.context.get("path")
items = [raw] if isinstance(raw, (str, Path)) else list(raw or [])
seeds = [str(p) for p in items if p]
assert seeds, "path is required"
depth = int(self.context.get("depth") or 1)
direction = (self.context.get("direction") or "both").lower()
assert direction in _VALID, f"direction must be one of {sorted(_VALID)}, got {direction!r}"
outbound, inbound = await _build_adjacency(self.file_store)
results = _bfs(seeds, depth, direction, outbound, inbound)
label = seeds[0] if len(seeds) == 1 else f"{len(seeds)} seeds"
self.context.response.success = True
self.context.response.answer = f"Traversed {len(results)} edge(s) from {label}"
self.context.response.metadata.update({"edges": results, "count": len(results)})
return self.context.response