mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-30 01:52:29 +00:00
* feat(config): add comprehensive job definitions for vault operations - Add utility jobs like version, search, traverse, list, read, stat - Include file operations like move, delete, upload, download - Add daily workspace management jobs: daily_list, daily_resolve, daily_reindex - Update descriptions to reflect vault-based operations instead of working_dir - Add proper section headers and documentation for each job category refactor(steps): reorganize step modules and remove demo steps - Move steps into categorized packages: common, crud, frontmatter, daily, jobs - Remove demo steps (DemoEchoStep1, DemoEchoStep2, StreamDemoStep1, StreamDemoStep2) - Add new steps: InitStep for vault initialization, TraverseStep for graph traversal - Update __init__.py to auto-import all step modules - Organize imports by functionality (common, CRUD operations, frontmatter, daily) feat(vault): implement vault-centric file operations and configuration - Change default config to use vault_dir instead of working_dir - Add environment variable support for embedding configuration - Implement file watcher with lite backend for daily/digest directories - Update search step to use 'name' instead of 'title' from frontmatter - Create ResourceEntry schema for tracking uploaded assets docs(steps): add comprehensive documentation for all step categories - Document file-I/O split by blast radius (crud vs frontmatter packages) - Add detailed descriptions for each step category and functionality - Explain the purpose and usage patterns for different types of file operations - Provide clear parameter documentation for all new job configurations * fix(config): correct vault directory path and remove unused job configurations - Fix vault_dir from 'vaultd' to 'vault' in default configuration - Remove deprecated traverse and list job configurations - Remove unused tag tooling configurations - Remove background watch_file job configuration refactor(steps): remove unused jobs module import - Comment out jobs module import in steps/__init__.py - This removes unused synchronizer and digester step registrations refactor(tests): update import path and add pylint directive - Update ResourceEntry import from reme4.schema to reme4.schema.resource_meta - Add pylint disable directive for unused argument in test datetime mocks * efactor(steps): remove unused modules from __all__ - Remove "background" module from __all__ list - Remove "jobs" module from __all__ list - These modules were no longer being used in the steps package * feat(config): update vault directory structure and remove file watcher - Change vault_dir reference from ./vault to ./vault in CLI example - Add daily_dir, digest_dir, and resource_dir configuration options - Remove file_watcher component configuration as it's no longer needed - Update comment to reflect correct module name (reme4vault) refactor(steps): add background step and remove deprecated init step - Import and register background step module - Remove deprecated InitStep from common steps - Update __all__ export list to include background step refactor(reindex): improve reindex step to scan vault directly - Update docstring to reflect vault scanning instead of watcher sync - Replace file watcher stop/start logic with direct vault path walking - Add support for suffix filtering during reindex operation - Use index_changes job to process found files refactor(wikilink_utils): enhance inbound source lookup with link scope - Import LinkScopeEnum for proper type handling - Update get_inlinks call to use ALL scope for virtual targets - Improve documentation for reverse-index lookup behavior test(refactor): clean up test suite removing deprecated functionality - Remove test_init_job and test_demo_job unit tests - Update help job assertion to check for literal command format - Change test directory from .reme to vault in CRUD tests - Remove init and demo job calls from integration test BREAKING CHANGE: Removes file_watcher component and init step * style(steps): fix import formatting in __init__.py Add proper spacing in the background module import statement to maintain consistent code style and readability. * refactor(config): change default vault directory from vault to .reme Default dev config now points vault_dir at ./.reme so `python -m reme4 start` can be run from the repo root and exercise the full atomic-tool surface against the seeded test data. BREAKING CHANGE: The default vault directory has been changed from 'vault' to '.reme' in the configuration. * docs(reme4_report): fix markdown formatting and remove extra content * refactor(file_parser): delegate wikilink extraction to WikilinkHandler * fix(search): handle empty query case gracefully - Replace assertion with conditional check for empty query - Set response success to false when query is empty - Return error message instead of throwing assertion error - Maintain existing validation for other parameters
228 lines
9.2 KiB
Python
228 lines
9.2 KiB
Python
"""Hybrid search over file_store using RRF fusion of vector + keyword results."""
|
|
|
|
import asyncio
|
|
|
|
from ..base_step import BaseStep
|
|
from ...components import R
|
|
from ...schema import FileChunk, FileLink, FileNode
|
|
|
|
_RRF_K = 60
|
|
_MAX_CANDIDATES = 200
|
|
|
|
|
|
@R.register("search_step")
|
|
class SearchStep(BaseStep):
|
|
"""Hybrid search: run vector + keyword in parallel, fuse via RRF, filter, truncate."""
|
|
|
|
@staticmethod
|
|
def _rrf_merge(
|
|
vector: list[FileChunk],
|
|
keyword: list[FileChunk],
|
|
vector_weight: float,
|
|
) -> list[FileChunk]:
|
|
"""Fuse two ranked lists with Reciprocal Rank Fusion, keyed by chunk.id."""
|
|
text_weight = 1.0 - vector_weight
|
|
merged: dict[str, FileChunk] = {}
|
|
|
|
for rank, chunk in enumerate(vector, start=1):
|
|
contrib = vector_weight / (_RRF_K + rank)
|
|
c = chunk.model_copy(deep=False)
|
|
c.scores = {**chunk.scores, "vector": chunk.scores.get("vector", chunk.score), "score": contrib}
|
|
merged[c.id] = c
|
|
|
|
for rank, chunk in enumerate(keyword, start=1):
|
|
contrib = text_weight / (_RRF_K + rank)
|
|
existing = merged.get(chunk.id)
|
|
if existing is not None:
|
|
existing.scores = {
|
|
**existing.scores,
|
|
"keyword": chunk.scores.get("keyword", chunk.score),
|
|
"score": existing.scores["score"] + contrib,
|
|
}
|
|
else:
|
|
c = chunk.model_copy(deep=False)
|
|
c.scores = {**chunk.scores, "keyword": chunk.scores.get("keyword", chunk.score), "score": contrib}
|
|
merged[c.id] = c
|
|
|
|
results = list(merged.values())
|
|
results.sort(key=lambda r: r.score, reverse=True)
|
|
return results
|
|
|
|
@staticmethod
|
|
def _format_scores(scores: dict[str, float], hybrid: bool) -> str:
|
|
"""Format scores for the answer line: always show fused; show per-branch when hybrid."""
|
|
parts = [f"score={scores.get('score', 0.0):.4f}"]
|
|
if hybrid:
|
|
for k in ("vector", "keyword"):
|
|
v = scores.get(k)
|
|
parts.append(f"{k}={v:.4f}" if v is not None else f"{k}=-")
|
|
return " ".join(parts)
|
|
|
|
@staticmethod
|
|
def _group_by_neighbor(links: list[FileLink], key_attr: str) -> dict[str, list[dict]]:
|
|
"""Group edges by neighbor path (insertion-ordered), each value a list of {predicate, anchor}."""
|
|
out: dict[str, list[dict]] = {}
|
|
for lnk in links:
|
|
neighbor = getattr(lnk, key_attr)
|
|
if not neighbor:
|
|
continue
|
|
out.setdefault(neighbor, []).append(
|
|
{"predicate": lnk.predicate, "anchor": lnk.target_anchor},
|
|
)
|
|
return out
|
|
|
|
@staticmethod
|
|
def _node_meta(node: FileNode | None) -> dict:
|
|
"""Extract a compact meta dict (name/description) from a FileNode."""
|
|
if node is None:
|
|
return {}
|
|
fm = node.front_matter
|
|
meta: dict = {}
|
|
if fm.name:
|
|
meta["name"] = fm.name
|
|
if fm.description:
|
|
meta["description"] = fm.description
|
|
return meta
|
|
|
|
@staticmethod
|
|
def _format_meta_inline(meta: dict) -> str:
|
|
"""One-line render of node meta for the answer; '(no meta)' when empty."""
|
|
parts = []
|
|
if "name" in meta:
|
|
parts.append(f'name="{meta["name"]}"')
|
|
if "description" in meta:
|
|
parts.append(f'description="{meta["description"]}"')
|
|
return " ".join(parts) if parts else "(no meta)"
|
|
|
|
@staticmethod
|
|
def _format_via(edge: dict) -> str:
|
|
"""Render a single (predicate, anchor) edge as a 'via ...' descriptor."""
|
|
bits = []
|
|
if edge.get("predicate"):
|
|
bits.append(f"predicate={edge['predicate']}")
|
|
if edge.get("anchor"):
|
|
bits.append(f"anchor=#{edge['anchor']}")
|
|
return ", ".join(bits) if bits else "plain"
|
|
|
|
async def _expand_links(
|
|
self,
|
|
chunk_paths: list[str],
|
|
max_per_direction: int,
|
|
) -> dict[str, dict]:
|
|
"""Fetch out/in links for each chunk path; attach neighbor meta. Returns per-path expansion."""
|
|
if not chunk_paths:
|
|
return {}
|
|
|
|
out_lists, in_lists = await asyncio.gather(
|
|
asyncio.gather(*(self.file_store.get_outlinks(p) for p in chunk_paths)),
|
|
asyncio.gather(*(self.file_store.get_inlinks(p) for p in chunk_paths)),
|
|
)
|
|
|
|
# Pre-group + cap per direction so we only fetch meta for displayed neighbors.
|
|
out_grouped = [
|
|
dict(list(self._group_by_neighbor(outs, "target_path").items())[:max_per_direction]) for outs in out_lists
|
|
]
|
|
in_grouped = [
|
|
dict(list(self._group_by_neighbor(ins, "source_path").items())[:max_per_direction]) for ins in in_lists
|
|
]
|
|
|
|
neighbor_paths = sorted({n for g in out_grouped for n in g} | {n for g in in_grouped for n in g})
|
|
nodes = await self.file_store.get_nodes(neighbor_paths) if neighbor_paths else []
|
|
meta_by_path = {n.path: self._node_meta(n) for n in nodes}
|
|
|
|
def _attach(grouped: dict[str, list[dict]]) -> list[dict]:
|
|
return [
|
|
{"path": npath, "meta": meta_by_path.get(npath, {}), "edges": edges} for npath, edges in grouped.items()
|
|
]
|
|
|
|
return {
|
|
cp: {"outlinks": _attach(og), "inlinks": _attach(ig)}
|
|
for cp, og, ig in zip(chunk_paths, out_grouped, in_grouped)
|
|
}
|
|
|
|
@classmethod
|
|
def _render_expansion_lines(cls, expansion: dict) -> list[str]:
|
|
"""Render outlinks/inlinks blocks for one chunk path; return zero or more indented lines."""
|
|
lines: list[str] = []
|
|
for direction, arrow, items in (
|
|
("outlinks", "→", expansion.get("outlinks") or []),
|
|
("inlinks", "←", expansion.get("inlinks") or []),
|
|
):
|
|
if not items:
|
|
continue
|
|
lines.append(f" {direction} ({len(items)}):")
|
|
for item in items:
|
|
lines.append(f" {arrow} {item['path']} {cls._format_meta_inline(item['meta'])}")
|
|
for edge in item["edges"]:
|
|
lines.append(f" via {cls._format_via(edge)}")
|
|
return lines
|
|
|
|
async def execute(self):
|
|
assert self.context is not None
|
|
query: str = (self.context.get("query", "") or "").strip()
|
|
limit: int = int(self.context.get("limit", 5))
|
|
min_score: float = float(self.context.get("min_score", 0.0))
|
|
vector_weight: float = float(self.kwargs.get("vector_weight", 0.7))
|
|
candidate_multiplier: float = float(self.kwargs.get("candidate_multiplier", 3.0))
|
|
expand_links: bool = bool(self.kwargs.get("expand_links", True))
|
|
max_links_per_direction: int = int(self.kwargs.get("max_links_per_direction", 10))
|
|
|
|
if not query:
|
|
self.context.response.success = False
|
|
self.context.response.answer = "Error: query cannot be empty"
|
|
return self.context.response
|
|
assert 0.0 <= vector_weight <= 1.0, f"vector_weight must be in [0, 1], got {vector_weight}"
|
|
assert limit > 0, f"limit must be positive, got {limit}"
|
|
|
|
candidates = min(_MAX_CANDIDATES, max(1, int(limit * candidate_multiplier)))
|
|
search_filter: dict = self.context.get("search_filter", {}) or {}
|
|
|
|
vector_results, keyword_results = await asyncio.gather(
|
|
self.file_store.vector_search(query, candidates, search_filter),
|
|
self.file_store.keyword_search(query, candidates, search_filter),
|
|
)
|
|
|
|
self.logger.info(
|
|
f"[{self.name}] query={query!r} candidates={candidates} "
|
|
f"vector_hits={len(vector_results)} keyword_hits={len(keyword_results)}",
|
|
)
|
|
|
|
hybrid = bool(vector_results) and bool(keyword_results)
|
|
if not vector_results and not keyword_results:
|
|
fused: list[FileChunk] = []
|
|
elif not keyword_results:
|
|
fused = vector_results
|
|
elif not vector_results:
|
|
fused = keyword_results
|
|
else:
|
|
fused = self._rrf_merge(vector_results, keyword_results, vector_weight)
|
|
|
|
if min_score > 0.0:
|
|
fused = [c for c in fused if c.score >= min_score]
|
|
fused = fused[:limit]
|
|
|
|
unique_paths = list(dict.fromkeys(c.path for c in fused))
|
|
link_expansion: dict[str, dict] = (
|
|
await self._expand_links(unique_paths, max_links_per_direction) if expand_links else {}
|
|
)
|
|
|
|
answer_lines: list[str] = []
|
|
for c in fused:
|
|
answer_lines.append(
|
|
f"========== {c.path}:{c.start_line}-{c.end_line} "
|
|
f"[{self._format_scores(c.scores, hybrid)}] ==========\n{c.text}",
|
|
)
|
|
answer_lines.extend(self._render_expansion_lines(link_expansion.get(c.path, {})))
|
|
|
|
self.context.response.answer = "\n".join(answer_lines)
|
|
self.context.response.metadata["results"] = [
|
|
c.model_dump(exclude_none=True, exclude={"embedding"}) for c in fused
|
|
]
|
|
self.context.response.metadata["link_expansion"] = link_expansion
|
|
self.context.response.metadata["counts"] = {
|
|
"vector": len(vector_results),
|
|
"keyword": len(keyword_results),
|
|
"returned": len(fused),
|
|
"hybrid": hybrid,
|
|
}
|
|
return self.context.response
|