mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-30 01:52:29 +00:00
* fix(bm25_index): 修正BM25索引计算中的文档长度归一化问题 修复了在计算BM25相似度时对文档长度进行不正确归一化的bug,确保所有查询都能得到准确的相关性评分。 * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * refactor(steps): Rename and adjust indexing step logic - Rename `scan_changes.py` and `reindex.py` to `clear_and_scan.py` - Update implementation details of `ScanChangesStep` and `ClearAndScanStep` - Modify the scheduling mechanism in `WatchChangesStep` - Adjust step registration and parameter configuration in config files - Update related tests to align with the new interface changes * up * feat(daily): replace daily CRUD operations with slug provisioning approach * refactor(tests): migrate CRUD step tests from HTTP server to direct LocalFileStore * up * up * up * up --------- Co-authored-by: huangsen <huangsen.huang@alibaba-inc.com>
111 lines
4.1 KiB
Python
111 lines
4.1 KiB
Python
"""BFS over wikilink edges from one or more seed files.
|
|
|
|
One record per traversed *edge* (not per node): the same target can repeat
|
|
if reached via different predicates or paths. Each record carries the
|
|
predecessor plus the link's predicate/anchor so callers can reconstruct
|
|
the path. Adjacency is built once via a single ``file_store.get_nodes()``
|
|
call — BFS then runs purely in memory with no per-frontier round-trips.
|
|
"""
|
|
|
|
from collections import deque
|
|
from pathlib import Path
|
|
|
|
from ..base_step import BaseStep
|
|
from ...components import R
|
|
from ...schema import FileLink
|
|
|
|
_OUT = {"out", "forward", "both"}
|
|
_IN = {"in", "backward", "both"}
|
|
_VALID = _OUT | _IN
|
|
|
|
# source path -> list of (neighbor path, link)
|
|
Adjacency = dict[str, list[tuple[str, FileLink]]]
|
|
|
|
|
|
async def _build_adjacency(file_store) -> tuple[Adjacency, Adjacency]:
|
|
"""Single ``get_nodes()`` pass → (outbound, inbound) adjacency maps.
|
|
|
|
Inbound stores the source path next to each link so BFS can attribute
|
|
inbound edges back to their origin — ``get_inlinks`` alone returns
|
|
target-shaped FileLinks without source attribution.
|
|
"""
|
|
outbound: Adjacency = {}
|
|
inbound: Adjacency = {}
|
|
for node in await file_store.get_nodes():
|
|
for link in node.links:
|
|
if link.target_path:
|
|
outbound.setdefault(node.path, []).append((link.target_path, link))
|
|
inbound.setdefault(link.target_path, []).append((node.path, link))
|
|
return outbound, inbound
|
|
|
|
|
|
def _bfs(
|
|
seeds: list[str],
|
|
max_depth: int,
|
|
direction: str,
|
|
outbound: Adjacency,
|
|
inbound: Adjacency,
|
|
) -> list[dict]:
|
|
"""In-memory BFS; emits one record per unique (src, dst, predicate) edge."""
|
|
sources: list[Adjacency] = []
|
|
if direction in _OUT:
|
|
sources.append(outbound)
|
|
if direction in _IN:
|
|
sources.append(inbound)
|
|
|
|
visited: set[tuple[str, str, str | None]] = set()
|
|
results: list[dict] = []
|
|
queue: deque[tuple[str, int]] = deque((s, 0) for s in seeds)
|
|
|
|
while queue:
|
|
current, depth = queue.popleft()
|
|
if depth >= max_depth:
|
|
continue
|
|
for src in sources:
|
|
for next_path, link in src.get(current, ()):
|
|
key = (current, next_path, link.predicate)
|
|
if key in visited:
|
|
continue
|
|
visited.add(key)
|
|
results.append(
|
|
{
|
|
"path": next_path,
|
|
"depth": depth + 1,
|
|
"via": current,
|
|
"predicate": link.predicate,
|
|
"anchor": link.target_anchor,
|
|
},
|
|
)
|
|
if depth + 1 < max_depth:
|
|
queue.append((next_path, depth + 1))
|
|
return results
|
|
|
|
|
|
@R.register("traverse_step")
|
|
class TraverseStep(BaseStep):
|
|
"""BFS from one or more seed files to explore wikilink relationships.
|
|
|
|
Parameters:
|
|
path — single seed (str) or list of seeds (vault-relative).
|
|
direction — ``forward`` / ``backward`` / ``both`` (or ``out`` / ``in`` / ``both``).
|
|
depth — hop limit (default 1 = immediate neighbors).
|
|
"""
|
|
|
|
async def execute(self):
|
|
assert self.context is not None
|
|
raw = self.context.get("path")
|
|
items = [raw] if isinstance(raw, (str, Path)) else list(raw or [])
|
|
seeds = [str(p) for p in items if p]
|
|
assert seeds, "path is required"
|
|
depth = int(self.context.get("depth") or 1)
|
|
direction = (self.context.get("direction") or "both").lower()
|
|
assert direction in _VALID, f"direction must be one of {sorted(_VALID)}, got {direction!r}"
|
|
|
|
outbound, inbound = await _build_adjacency(self.file_store)
|
|
results = _bfs(seeds, depth, direction, outbound, inbound)
|
|
|
|
label = seeds[0] if len(seeds) == 1 else f"{len(seeds)} seeds"
|
|
self.context.response.success = True
|
|
self.context.response.answer = f"Traversed {len(results)} edge(s) from {label}"
|
|
self.context.response.metadata.update({"edges": results, "count": len(results)})
|
|
return self.context.response
|