mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-10-09 03:20:54 +00:00
docs(reme4): update report with detailed architecture sections (#252)
- Add comprehensive Markdown kernel section covering Obsidian compatibility - Include detailed explanation of YAML front matter and wikilink formats - Document smart slicing mechanism using Markdown AST instead of fixed tokens - Explain graph indexing with bidirectional links and multiple backends - Restructure sections with proper numbering from 4 to 7 - Move Markdown kernel section to appear before self-evolution features - Add detailed explanations of auto-memory, auto-dream, and auto-link processes - Document three-way hybrid search with RRF fusion and progressive expansion - Include engineering value explanations for keyword indexing in Chinese context
This commit is contained in:
parent
0dff85b9b5
commit
ee94d3ec8b
11 changed files with 259 additions and 242 deletions
|
|
@ -120,7 +120,7 @@ V4更加高效的底层记忆索引
|
|||
- 不支持关键词检索,这里需要Keyword倒排索引,对中文的支持较差
|
||||
- [📌 历史] 描述 V3 痛点,不需要代码。
|
||||
- V4版本我们重写了file parser,file store,file graph,file watcher,手写了支持增量更新倒排索引
|
||||
- file parser → [✅ `reme4/components/file_parser/`(base/bare/default/linked 四种)]
|
||||
- file parser → [✅ `reme4/components/file_parser/`(base/default/chunked/linked 四种)]
|
||||
- file store → [✅ `reme4/components/file_store/local_file_store.py`]
|
||||
- file graph → [✅ `reme4/components/file_graph/`(local/nx/neo4j)]
|
||||
- file watcher → [✅ `reme4/components/file_watcher/lite_file_watcher.py` 基于 watchfiles awatch;`base_file_watcher.py` 抽象接口]
|
||||
|
|
|
|||
|
|
@ -86,9 +86,9 @@
|
|||
| 类 | 文件 | 说明 |
|
||||
| --- | --- | --- |
|
||||
| `BaseFileParser` | `base_file_parser.py` | 抽象接口:`parse(path) -> (FileNode, list[FileChunk])`,提供 `_get_relative_path`。 |
|
||||
| `BareFileParser` (`@R "bare"`) | `bare_file_parser.py` | 仅 stat:附件/二进制不读内容、不切块、不抽链接。 |
|
||||
| `DefaultFileParser` (`@R "default"`) | `default_file_parser.py` | 字节级带 overlap 切片 + YAML front matter + wikilink 抽取(含 Dataview `predicate::`)。 |
|
||||
| `LinkedFileParser` (`@R "md"`) | `linked_file_parser.py` | Markdown 专用:mistletoe AST → MdNode 树 → 章节递归分块;每个 chunk 携带完整 heading skeleton(TOC);wikilink 解析支持隐式 `.md`、folder-note、短路径歧义扇出,需注入 `file_graph` 解析目标。 |
|
||||
| `DefaultFileParser` (`@R "default"`) | `default_file_parser.py` | 仅 stat:附件/二进制不读内容、不切块、不抽链接;作为无 `supported_extensions` 命中时的兜底 parser。 |
|
||||
| `ChunkedFileParser` (`@R "chunked"`) | `chunked_file_parser.py` | 字节级带 overlap 切片 + YAML front matter + wikilink 抽取(含 Dataview `predicate::`)。 |
|
||||
| `LinkedFileParser` (`@R "linked"`) | `linked_file_parser.py` | Markdown 专用:mistletoe AST → MdNode 树 → 章节递归分块;每个 chunk 携带完整 heading skeleton(TOC);wikilink 解析支持隐式 `.md`、folder-note、短路径歧义扇出,需注入 `file_graph` 解析目标。 |
|
||||
|
||||
### 4.6 File Store — `reme4/components/file_store/`
|
||||
|
||||
|
|
|
|||
|
|
@ -82,20 +82,35 @@ reme4 search query="..." backend=mcp
|
|||
| crud | upload/download | 其他文件 |
|
||||
| file | stat | path |
|
||||
| file | list | path |
|
||||
| property | property:read | |
|
||||
| property | property:update | path="My Note" status=done xx=xxx |
|
||||
| property | property:delete | keys="[xxxx, xxxx]" |
|
||||
|
|
||||
| graph | traverse | path="My Note" directtion=forward/backward depth=1 predicat=xxx |
|
||||
|
||||
@wangce
|
||||
| crud | write | path="New Note" name="xxx" description="xxx" content="# Hello" (4 字段都必填,frontmatter 只写 name/description) |
|
||||
| crud | write | path="New Note" name="xxx" description="xxx" metadata={}, content="# Hello" (4 字段都必填,frontmatter 只写 name/description) |
|
||||
| crud | read | path="Templates/Recipe.md" |
|
||||
| crud | edit | path="Templates/Recipe.md" old="xxx" new="xxx" |
|
||||
| crud | append | path="My Note" content="New line" |
|
||||
| crud | prepend | path="My Note" content="New line" |
|
||||
|
||||
| crud | delete | path="My Note
|
||||
| daily:crud | daily:xxx | 与 crud 参数保持一致 |
|
||||
|
||||
- daily:resolve name=xxxx (符合一定规范 win下要求)
|
||||
- daily:list date=xxxx 返回path
|
||||
- daily:index
|
||||
|
||||
frontmatter read path
|
||||
frontmatter update path metadata={}
|
||||
frontmatter delete path keys=[]
|
||||
|
||||
delete path
|
||||
download path=xxx(内部相对路径)download_path=(外部绝对路径,可选)
|
||||
upload path=xxx(外部绝对路径)description="xxx" metadata=xxx 返回内部相对路径 加metadata
|
||||
stat path
|
||||
list path
|
||||
mv path=xxx new_path=xxx
|
||||
|
||||
traverse path=xxx direction=xxx depth=xxx
|
||||
|
||||
# 日记类型
|
||||
|
||||
| 类型 | 路径 | 说明 |
|
||||
|
|
|
|||
|
|
@ -1,8 +1,8 @@
|
|||
"""File parser components."""
|
||||
|
||||
from .bare_file_parser import BareFileParser
|
||||
from .base_file_parser import BaseFileParser
|
||||
from .chunked_file_parser import ChunkedFileParser
|
||||
from .default_file_parser import DefaultFileParser
|
||||
from .linked_file_parser import LinkedFileParser
|
||||
|
||||
__all__ = ["BareFileParser", "BaseFileParser", "DefaultFileParser", "LinkedFileParser"]
|
||||
__all__ = ["BaseFileParser", "ChunkedFileParser", "DefaultFileParser", "LinkedFileParser"]
|
||||
|
|
|
|||
|
|
@ -1,22 +0,0 @@
|
|||
"""Stat-only parser for attachment/binary files."""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from .base_file_parser import BaseFileParser
|
||||
from ..component_registry import R
|
||||
from ...schema import FileChunk, FileNode
|
||||
|
||||
|
||||
@R.register("bare")
|
||||
class BareFileParser(BaseFileParser):
|
||||
"""Stat-only parser for attachment/binary files.
|
||||
|
||||
No content read, no chunking, no link extraction. The resulting FileNode
|
||||
has empty links and chunk_ids; front_matter carries mime and size so
|
||||
retrieval can filter by file type without reopening the file.
|
||||
"""
|
||||
|
||||
async def parse(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]:
|
||||
file_path = Path(path)
|
||||
stat = file_path.stat()
|
||||
return FileNode(path=self.to_vault_relative(path), st_mtime=stat.st_mtime, links=[], chunk_ids=[]), []
|
||||
180
reme4/components/file_parser/chunked_file_parser.py
Normal file
180
reme4/components/file_parser/chunked_file_parser.py
Normal file
|
|
@ -0,0 +1,180 @@
|
|||
"""File parser with byte-based overlapping chunking."""
|
||||
|
||||
import re
|
||||
from bisect import bisect_right
|
||||
from pathlib import Path
|
||||
|
||||
import aiofiles
|
||||
import yaml
|
||||
|
||||
from .base_file_parser import BaseFileParser
|
||||
from ..component_registry import R
|
||||
from ...schema import FileChunk, FileFrontMatter, FileLink, FileNode
|
||||
|
||||
# Single-pass wikilink + optional dataview predicate.
|
||||
# Covers: [[X]] / ![[X]] / [[X#h]] / [[X|alias]] / pred:: [[X]] / [pred:: [[X]]]
|
||||
# - predicate group: optional leading '[' (dataview inline-bracket form), an identifier,
|
||||
# then '::' — the whole prefix is non-capturing-optional so bare wikilinks still match.
|
||||
# - optional '!' prefix matches the embed form (![[X]]).
|
||||
# - target / anchor / alias all forbid '\n' so a wikilink cannot span lines.
|
||||
# - alias '|...': consumed but not captured (we don't need display text).
|
||||
_LINK_RE = re.compile(
|
||||
r"(?:\[?\s*(?P<predicate>[A-Za-z][\w-]*)\s*::\s*)?"
|
||||
r"!?\[\[\s*(?P<target>[^\[\]|#\n]+?)"
|
||||
r"(?:#(?P<anchor>[^\[\]|\n]+?))?"
|
||||
r"\s*(?:\|[^\[\]\n]*?)?\s*]]",
|
||||
)
|
||||
|
||||
|
||||
@R.register("chunked")
|
||||
class ChunkedFileParser(BaseFileParser):
|
||||
"""Parser that splits files into byte-based overlapping chunks."""
|
||||
|
||||
def __init__(self, encoding: str = "utf-8", chunk_byte_size: int = 10000, overlap_byte_size: int = 100, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
self.encoding = encoding
|
||||
self.chunk_byte_size = max(100, chunk_byte_size)
|
||||
self.overlap_byte_size = max(4, overlap_byte_size)
|
||||
|
||||
@staticmethod
|
||||
def parse_links(content: str, source_path: str) -> list[FileLink]:
|
||||
"""Extract wikilinks with optional dataview predicate as outgoing FileLinks."""
|
||||
links: list[FileLink] = []
|
||||
for m in _LINK_RE.finditer(content):
|
||||
target = m["target"].strip()
|
||||
if not target:
|
||||
continue
|
||||
anchor = m["anchor"]
|
||||
links.append(
|
||||
FileLink(
|
||||
source_path=source_path,
|
||||
target_path=target,
|
||||
target_anchor=anchor.strip() if anchor else None,
|
||||
predicate=m["predicate"],
|
||||
),
|
||||
)
|
||||
return links
|
||||
|
||||
@staticmethod
|
||||
def _parse_front_matter(text: str) -> tuple[FileFrontMatter, str]:
|
||||
"""Parse YAML front matter delimited by ---, return (front_matter, remaining)."""
|
||||
if not text.startswith("---"):
|
||||
return FileFrontMatter(), text
|
||||
end_idx = text.find("\n---", 3)
|
||||
if end_idx == -1:
|
||||
return FileFrontMatter(), text
|
||||
try:
|
||||
data = yaml.safe_load(text[3:end_idx].strip()) or {}
|
||||
front_matter = FileFrontMatter(**(data if isinstance(data, dict) else {}))
|
||||
except yaml.YAMLError:
|
||||
front_matter = FileFrontMatter()
|
||||
return front_matter, text[end_idx + 4 :].lstrip("\n")
|
||||
|
||||
async def parse(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]:
|
||||
file_path = Path(path)
|
||||
stat = file_path.stat()
|
||||
rel_path = self.to_vault_relative(path)
|
||||
|
||||
async with aiofiles.open(file_path, encoding=self.encoding) as f:
|
||||
text = await f.read()
|
||||
|
||||
if not text:
|
||||
return FileNode(path=rel_path, st_mtime=stat.st_mtime), []
|
||||
|
||||
is_markdown = file_path.suffix.lower() == ".md"
|
||||
if is_markdown:
|
||||
front_matter, content = self._parse_front_matter(text)
|
||||
if not content:
|
||||
return FileNode(path=rel_path, st_mtime=stat.st_mtime, front_matter=front_matter), []
|
||||
links = self.parse_links(content, rel_path)
|
||||
else:
|
||||
front_matter = FileFrontMatter()
|
||||
content = text
|
||||
links = []
|
||||
|
||||
chunks = self._chunk_content(content, rel_path, parse_links=is_markdown)
|
||||
chunk_ids = [c.id for c in chunks]
|
||||
return (
|
||||
FileNode(
|
||||
path=rel_path,
|
||||
st_mtime=stat.st_mtime,
|
||||
front_matter=front_matter,
|
||||
links=links,
|
||||
chunk_ids=chunk_ids,
|
||||
),
|
||||
chunks,
|
||||
)
|
||||
|
||||
def _link_byte_spans(self, content: str) -> list[tuple[int, int]]:
|
||||
"""Return [start, end) byte spans of every wikilink in content."""
|
||||
spans: list[tuple[int, int]] = []
|
||||
last_char, last_byte = 0, 0
|
||||
for m in _LINK_RE.finditer(content):
|
||||
last_byte += len(content[last_char : m.start()].encode(self.encoding))
|
||||
match_bytes = len(m.group(0).encode(self.encoding))
|
||||
spans.append((last_byte, last_byte + match_bytes))
|
||||
last_byte += match_bytes
|
||||
last_char = m.end()
|
||||
return spans
|
||||
|
||||
@staticmethod
|
||||
def _span_containing(
|
||||
pos: int,
|
||||
spans: list[tuple[int, int]],
|
||||
starts: list[int],
|
||||
) -> tuple[int, int] | None:
|
||||
"""Return the span strictly containing pos (s < pos < e), or None."""
|
||||
idx = bisect_right(starts, pos) - 1
|
||||
if idx < 0:
|
||||
return None
|
||||
s, e = spans[idx]
|
||||
return (s, e) if s < pos < e else None
|
||||
|
||||
def _chunk_content(self, content: str, rel_path: str, parse_links: bool = True) -> list[FileChunk]:
|
||||
"""Split content into overlapping byte-range chunks, avoiding cuts inside wikilinks.
|
||||
|
||||
When ``parse_links`` is False, skip wikilink span computation and boundary checks
|
||||
— used for non-markdown files where wikilink semantics don't apply.
|
||||
"""
|
||||
content_bytes = content.encode(self.encoding)
|
||||
n = len(content_bytes)
|
||||
newline_positions = [i for i, b in enumerate(content_bytes) if b == ord("\n")]
|
||||
if parse_links:
|
||||
link_spans = self._link_byte_spans(content)
|
||||
link_starts = [s for s, _ in link_spans]
|
||||
else:
|
||||
link_spans: list[tuple[int, int]] = []
|
||||
link_starts: list[int] = []
|
||||
# Refuse to retreat past half of chunk_byte_size; falls back to hard cut
|
||||
# for pathologically long links so we always make forward progress.
|
||||
min_chunk = self.chunk_byte_size // 2
|
||||
chunks: list[FileChunk] = []
|
||||
start = 0
|
||||
|
||||
while start < n:
|
||||
end = min(start + self.chunk_byte_size, n)
|
||||
if end < n:
|
||||
span = self._span_containing(end, link_spans, link_starts)
|
||||
if span is not None and span[0] - start >= min_chunk:
|
||||
end = span[0]
|
||||
|
||||
chunk_text = content_bytes[start:end].decode(self.encoding, errors="ignore")
|
||||
start_line = bisect_right(newline_positions, start - 1) + 1
|
||||
end_line = bisect_right(newline_positions, end - 1) + 1
|
||||
if content_bytes[end - 1] == ord("\n"):
|
||||
end_line -= 1
|
||||
chunks.append(
|
||||
FileChunk(path=rel_path, start_line=start_line, end_line=end_line, text=chunk_text).set_hash_id(),
|
||||
)
|
||||
if end >= n:
|
||||
break
|
||||
|
||||
next_start = end - self.overlap_byte_size
|
||||
span = self._span_containing(next_start, link_spans, link_starts)
|
||||
if span is not None:
|
||||
next_start = span[1]
|
||||
if next_start <= start:
|
||||
next_start = end
|
||||
start = next_start
|
||||
|
||||
return chunks
|
||||
|
|
@ -1,180 +1,22 @@
|
|||
"""Default file parser with byte-based overlapping chunking."""
|
||||
"""Stat-only parser for attachment/binary files."""
|
||||
|
||||
import re
|
||||
from bisect import bisect_right
|
||||
from pathlib import Path
|
||||
|
||||
import aiofiles
|
||||
import yaml
|
||||
|
||||
from .base_file_parser import BaseFileParser
|
||||
from ..component_registry import R
|
||||
from ...schema import FileChunk, FileFrontMatter, FileLink, FileNode
|
||||
|
||||
# Single-pass wikilink + optional dataview predicate.
|
||||
# Covers: [[X]] / ![[X]] / [[X#h]] / [[X|alias]] / pred:: [[X]] / [pred:: [[X]]]
|
||||
# - predicate group: optional leading '[' (dataview inline-bracket form), an identifier,
|
||||
# then '::' — the whole prefix is non-capturing-optional so bare wikilinks still match.
|
||||
# - optional '!' prefix matches the embed form (![[X]]).
|
||||
# - target / anchor / alias all forbid '\n' so a wikilink cannot span lines.
|
||||
# - alias '|...': consumed but not captured (we don't need display text).
|
||||
_LINK_RE = re.compile(
|
||||
r"(?:\[?\s*(?P<predicate>[A-Za-z][\w-]*)\s*::\s*)?"
|
||||
r"!?\[\[\s*(?P<target>[^\[\]|#\n]+?)"
|
||||
r"(?:#(?P<anchor>[^\[\]|\n]+?))?"
|
||||
r"\s*(?:\|[^\[\]\n]*?)?\s*]]",
|
||||
)
|
||||
from ...schema import FileChunk, FileNode
|
||||
|
||||
|
||||
@R.register("default")
|
||||
class DefaultFileParser(BaseFileParser):
|
||||
"""Parser that splits files into byte-based overlapping chunks."""
|
||||
"""Stat-only parser for attachment/binary files.
|
||||
|
||||
def __init__(self, encoding: str = "utf-8", chunk_byte_size: int = 10000, overlap_byte_size: int = 100, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
self.encoding = encoding
|
||||
self.chunk_byte_size = max(100, chunk_byte_size)
|
||||
self.overlap_byte_size = max(4, overlap_byte_size)
|
||||
|
||||
@staticmethod
|
||||
def parse_links(content: str, source_path: str) -> list[FileLink]:
|
||||
"""Extract wikilinks with optional dataview predicate as outgoing FileLinks."""
|
||||
links: list[FileLink] = []
|
||||
for m in _LINK_RE.finditer(content):
|
||||
target = m["target"].strip()
|
||||
if not target:
|
||||
continue
|
||||
anchor = m["anchor"]
|
||||
links.append(
|
||||
FileLink(
|
||||
source_path=source_path,
|
||||
target_path=target,
|
||||
target_anchor=anchor.strip() if anchor else None,
|
||||
predicate=m["predicate"],
|
||||
),
|
||||
)
|
||||
return links
|
||||
|
||||
@staticmethod
|
||||
def _parse_front_matter(text: str) -> tuple[FileFrontMatter, str]:
|
||||
"""Parse YAML front matter delimited by ---, return (front_matter, remaining)."""
|
||||
if not text.startswith("---"):
|
||||
return FileFrontMatter(), text
|
||||
end_idx = text.find("\n---", 3)
|
||||
if end_idx == -1:
|
||||
return FileFrontMatter(), text
|
||||
try:
|
||||
data = yaml.safe_load(text[3:end_idx].strip()) or {}
|
||||
front_matter = FileFrontMatter(**(data if isinstance(data, dict) else {}))
|
||||
except yaml.YAMLError:
|
||||
front_matter = FileFrontMatter()
|
||||
return front_matter, text[end_idx + 4 :].lstrip("\n")
|
||||
No content read, no chunking, no link extraction. The resulting FileNode
|
||||
has empty links and chunk_ids; front_matter carries mime and size so
|
||||
retrieval can filter by file type without reopening the file.
|
||||
"""
|
||||
|
||||
async def parse(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]:
|
||||
file_path = Path(path)
|
||||
stat = file_path.stat()
|
||||
rel_path = self.to_vault_relative(path)
|
||||
|
||||
async with aiofiles.open(file_path, encoding=self.encoding) as f:
|
||||
text = await f.read()
|
||||
|
||||
if not text:
|
||||
return FileNode(path=rel_path, st_mtime=stat.st_mtime), []
|
||||
|
||||
is_markdown = file_path.suffix.lower() == ".md"
|
||||
if is_markdown:
|
||||
front_matter, content = self._parse_front_matter(text)
|
||||
if not content:
|
||||
return FileNode(path=rel_path, st_mtime=stat.st_mtime, front_matter=front_matter), []
|
||||
links = self.parse_links(content, rel_path)
|
||||
else:
|
||||
front_matter = FileFrontMatter()
|
||||
content = text
|
||||
links = []
|
||||
|
||||
chunks = self._chunk_content(content, rel_path, parse_links=is_markdown)
|
||||
chunk_ids = [c.id for c in chunks]
|
||||
return (
|
||||
FileNode(
|
||||
path=rel_path,
|
||||
st_mtime=stat.st_mtime,
|
||||
front_matter=front_matter,
|
||||
links=links,
|
||||
chunk_ids=chunk_ids,
|
||||
),
|
||||
chunks,
|
||||
)
|
||||
|
||||
def _link_byte_spans(self, content: str) -> list[tuple[int, int]]:
|
||||
"""Return [start, end) byte spans of every wikilink in content."""
|
||||
spans: list[tuple[int, int]] = []
|
||||
last_char, last_byte = 0, 0
|
||||
for m in _LINK_RE.finditer(content):
|
||||
last_byte += len(content[last_char : m.start()].encode(self.encoding))
|
||||
match_bytes = len(m.group(0).encode(self.encoding))
|
||||
spans.append((last_byte, last_byte + match_bytes))
|
||||
last_byte += match_bytes
|
||||
last_char = m.end()
|
||||
return spans
|
||||
|
||||
@staticmethod
|
||||
def _span_containing(
|
||||
pos: int,
|
||||
spans: list[tuple[int, int]],
|
||||
starts: list[int],
|
||||
) -> tuple[int, int] | None:
|
||||
"""Return the span strictly containing pos (s < pos < e), or None."""
|
||||
idx = bisect_right(starts, pos) - 1
|
||||
if idx < 0:
|
||||
return None
|
||||
s, e = spans[idx]
|
||||
return (s, e) if s < pos < e else None
|
||||
|
||||
def _chunk_content(self, content: str, rel_path: str, parse_links: bool = True) -> list[FileChunk]:
|
||||
"""Split content into overlapping byte-range chunks, avoiding cuts inside wikilinks.
|
||||
|
||||
When ``parse_links`` is False, skip wikilink span computation and boundary checks
|
||||
— used for non-markdown files where wikilink semantics don't apply.
|
||||
"""
|
||||
content_bytes = content.encode(self.encoding)
|
||||
n = len(content_bytes)
|
||||
newline_positions = [i for i, b in enumerate(content_bytes) if b == ord("\n")]
|
||||
if parse_links:
|
||||
link_spans = self._link_byte_spans(content)
|
||||
link_starts = [s for s, _ in link_spans]
|
||||
else:
|
||||
link_spans: list[tuple[int, int]] = []
|
||||
link_starts: list[int] = []
|
||||
# Refuse to retreat past half of chunk_byte_size; falls back to hard cut
|
||||
# for pathologically long links so we always make forward progress.
|
||||
min_chunk = self.chunk_byte_size // 2
|
||||
chunks: list[FileChunk] = []
|
||||
start = 0
|
||||
|
||||
while start < n:
|
||||
end = min(start + self.chunk_byte_size, n)
|
||||
if end < n:
|
||||
span = self._span_containing(end, link_spans, link_starts)
|
||||
if span is not None and span[0] - start >= min_chunk:
|
||||
end = span[0]
|
||||
|
||||
chunk_text = content_bytes[start:end].decode(self.encoding, errors="ignore")
|
||||
start_line = bisect_right(newline_positions, start - 1) + 1
|
||||
end_line = bisect_right(newline_positions, end - 1) + 1
|
||||
if content_bytes[end - 1] == ord("\n"):
|
||||
end_line -= 1
|
||||
chunks.append(
|
||||
FileChunk(path=rel_path, start_line=start_line, end_line=end_line, text=chunk_text).set_hash_id(),
|
||||
)
|
||||
if end >= n:
|
||||
break
|
||||
|
||||
next_start = end - self.overlap_byte_size
|
||||
span = self._span_containing(next_start, link_spans, link_starts)
|
||||
if span is not None:
|
||||
next_start = span[1]
|
||||
if next_start <= start:
|
||||
next_start = end
|
||||
start = next_start
|
||||
|
||||
return chunks
|
||||
return FileNode(path=self.to_vault_relative(path), st_mtime=stat.st_mtime, links=[], chunk_ids=[]), []
|
||||
|
|
|
|||
|
|
@ -264,20 +264,20 @@ components:
|
|||
backend: local
|
||||
|
||||
file_parser:
|
||||
default:
|
||||
backend: default
|
||||
linked:
|
||||
backend: linked
|
||||
supported_extensions:
|
||||
- md
|
||||
chunked:
|
||||
backend: chunked
|
||||
supported_extensions:
|
||||
- txt
|
||||
- html
|
||||
- json
|
||||
- yaml
|
||||
- py
|
||||
bare:
|
||||
backend: bare
|
||||
linked:
|
||||
backend: linked
|
||||
supported_extensions:
|
||||
- md
|
||||
default:
|
||||
backend: default
|
||||
|
||||
keyword_index:
|
||||
default:
|
||||
|
|
|
|||
|
|
@ -138,7 +138,7 @@ class BaseStep(ABC):
|
|||
"""Parse ``path`` with the parser whose ``supported_extensions`` claims its suffix.
|
||||
|
||||
First registered match wins (config insertion order). Falls back to the
|
||||
``bare`` parser (stat-only) when no parser claims the suffix — that's
|
||||
``default`` parser (stat-only) when no parser claims the suffix — that's
|
||||
how attachments / binaries / unknown types still produce a FileNode.
|
||||
"""
|
||||
assert self.app_context is not None
|
||||
|
|
@ -154,10 +154,12 @@ class BaseStep(ABC):
|
|||
break
|
||||
|
||||
if parser is None:
|
||||
parser = file_parser_dict.get("bare")
|
||||
parser = file_parser_dict.get("default")
|
||||
|
||||
if parser is None:
|
||||
raise RuntimeError(f"No file parser supports {path} (suffix={suffix!r}) and no 'bare' parser is configured")
|
||||
raise RuntimeError(
|
||||
f"No file parser supports {path} (suffix={suffix!r}) and no 'default' parser is configured",
|
||||
)
|
||||
|
||||
return await parser.parse(path)
|
||||
|
||||
|
|
|
|||
|
|
@ -18,7 +18,7 @@ from typing import Any
|
|||
|
||||
from watchfiles import Change
|
||||
|
||||
from reme4.components.file_parser import DefaultFileParser
|
||||
from reme4.components.file_parser import ChunkedFileParser
|
||||
from reme4.components.file_store import LocalFileStore
|
||||
from reme4.components.runtime_context import RuntimeContext
|
||||
from reme4.schema import Response
|
||||
|
|
@ -86,9 +86,9 @@ async def _make_update_step(
|
|||
suffix_filters: list[str] | None = None,
|
||||
recursive: bool = True,
|
||||
dump: bool = True,
|
||||
) -> tuple[_RecordingUpdateStoreStep, RuntimeContext, LocalFileStore, DefaultFileParser]:
|
||||
) -> tuple[_RecordingUpdateStoreStep, RuntimeContext, LocalFileStore, ChunkedFileParser]:
|
||||
fs = LocalFileStore(store_name="test_store", embedding_model="")
|
||||
parser = DefaultFileParser()
|
||||
parser = ChunkedFileParser()
|
||||
await fs.start()
|
||||
await parser.start()
|
||||
step = _RecordingUpdateStoreStep(
|
||||
|
|
@ -105,7 +105,7 @@ async def _make_update_step(
|
|||
return step, context, fs, parser
|
||||
|
||||
|
||||
async def _teardown(fs: LocalFileStore, parser: DefaultFileParser) -> None:
|
||||
async def _teardown(fs: LocalFileStore, parser: ChunkedFileParser) -> None:
|
||||
await parser.close()
|
||||
await fs.close()
|
||||
|
||||
|
|
|
|||
|
|
@ -1,10 +1,10 @@
|
|||
"""Tests for DefaultFileParser."""
|
||||
"""Tests for ChunkedFileParser."""
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
from reme4.components.file_parser import DefaultFileParser
|
||||
from reme4.components.file_parser import ChunkedFileParser
|
||||
|
||||
|
||||
# Add parent path for import
|
||||
|
|
@ -18,7 +18,7 @@ def test_parse_empty_file():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser()
|
||||
parser = ChunkedFileParser()
|
||||
file_node, chunks = await parser.parse(temp_path)
|
||||
assert file_node.path == temp_path
|
||||
assert len(chunks) == 0
|
||||
|
|
@ -39,7 +39,7 @@ def test_parse_small_file():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser(chunk_byte_size=10000)
|
||||
parser = ChunkedFileParser(chunk_byte_size=10000)
|
||||
_, chunks = await parser.parse(temp_path)
|
||||
assert len(chunks) == 1
|
||||
assert chunks[0].start_line == 1
|
||||
|
|
@ -63,7 +63,7 @@ def test_parse_multiline_file():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser(chunk_byte_size=10000)
|
||||
parser = ChunkedFileParser(chunk_byte_size=10000)
|
||||
_, chunks = await parser.parse(temp_path)
|
||||
assert len(chunks) == 1
|
||||
assert chunks[0].start_line == 1
|
||||
|
|
@ -87,7 +87,7 @@ def test_parse_chunked_file():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser(chunk_byte_size=5000, overlap_byte_size=100)
|
||||
parser = ChunkedFileParser(chunk_byte_size=5000, overlap_byte_size=100)
|
||||
_, chunks = await parser.parse(temp_path)
|
||||
assert len(chunks) > 1, f"Expected multiple chunks, got {len(chunks)}"
|
||||
# Verify overlap by checking that consecutive chunks share some content
|
||||
|
|
@ -109,7 +109,7 @@ def test_parse_with_custom_encoding():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser(encoding="utf-8")
|
||||
parser = ChunkedFileParser(encoding="utf-8")
|
||||
_, chunks = await parser.parse(temp_path)
|
||||
assert len(chunks) >= 1
|
||||
assert "你好世界" in chunks[0].text
|
||||
|
|
@ -130,7 +130,7 @@ def test_file_node_properties():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser()
|
||||
parser = ChunkedFileParser()
|
||||
file_node, _ = await parser.parse(temp_path)
|
||||
assert hasattr(file_node, "path")
|
||||
assert hasattr(file_node, "st_mtime")
|
||||
|
|
@ -152,7 +152,7 @@ def test_file_chunk_properties():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser()
|
||||
parser = ChunkedFileParser()
|
||||
_, chunks = await parser.parse(temp_path)
|
||||
chunk = chunks[0]
|
||||
assert hasattr(chunk, "path")
|
||||
|
|
@ -171,7 +171,7 @@ def test_file_chunk_properties():
|
|||
|
||||
def test_parse_links_bare():
|
||||
"""Bare wikilink: [[target]]."""
|
||||
links = DefaultFileParser.parse_links("see [[note]]", "src.md")
|
||||
links = ChunkedFileParser.parse_links("see [[note]]", "src.md")
|
||||
assert len(links) == 1
|
||||
link = links[0]
|
||||
assert link.source_path == "src.md"
|
||||
|
|
@ -183,7 +183,7 @@ def test_parse_links_bare():
|
|||
|
||||
def test_parse_links_with_anchor():
|
||||
"""Wikilink with anchor: [[target#anchor]]."""
|
||||
links = DefaultFileParser.parse_links("see [[note#section A]]", "src.md")
|
||||
links = ChunkedFileParser.parse_links("see [[note#section A]]", "src.md")
|
||||
assert len(links) == 1
|
||||
assert links[0].target_path == "note"
|
||||
assert links[0].target_anchor == "section A"
|
||||
|
|
@ -193,7 +193,7 @@ def test_parse_links_with_anchor():
|
|||
|
||||
def test_parse_links_alias_dropped():
|
||||
"""Alias after '|' is consumed but not captured as anchor."""
|
||||
links = DefaultFileParser.parse_links("see [[note|display text]]", "src.md")
|
||||
links = ChunkedFileParser.parse_links("see [[note|display text]]", "src.md")
|
||||
assert len(links) == 1
|
||||
assert links[0].target_path == "note"
|
||||
assert links[0].target_anchor is None
|
||||
|
|
@ -202,7 +202,7 @@ def test_parse_links_alias_dropped():
|
|||
|
||||
def test_parse_links_anchor_and_alias():
|
||||
"""[[target#anchor|alias]] — anchor captured, alias dropped."""
|
||||
links = DefaultFileParser.parse_links("see [[note#sec|disp]]", "src.md")
|
||||
links = ChunkedFileParser.parse_links("see [[note#sec|disp]]", "src.md")
|
||||
assert len(links) == 1
|
||||
assert links[0].target_path == "note"
|
||||
assert links[0].target_anchor == "sec"
|
||||
|
|
@ -211,7 +211,7 @@ def test_parse_links_anchor_and_alias():
|
|||
|
||||
def test_parse_links_predicate_simple():
|
||||
"""Dataview inline: predicate:: [[target]]."""
|
||||
links = DefaultFileParser.parse_links("author:: [[Alice]]", "src.md")
|
||||
links = ChunkedFileParser.parse_links("author:: [[Alice]]", "src.md")
|
||||
assert len(links) == 1
|
||||
assert links[0].predicate == "author"
|
||||
assert links[0].target_path == "Alice"
|
||||
|
|
@ -221,7 +221,7 @@ def test_parse_links_predicate_simple():
|
|||
|
||||
def test_parse_links_predicate_bracketed():
|
||||
"""Dataview inline-bracket: [predicate:: [[target]]]."""
|
||||
links = DefaultFileParser.parse_links("text [author:: [[Alice]]] more", "src.md")
|
||||
links = ChunkedFileParser.parse_links("text [author:: [[Alice]]] more", "src.md")
|
||||
assert len(links) == 1
|
||||
assert links[0].predicate == "author"
|
||||
assert links[0].target_path == "Alice"
|
||||
|
|
@ -230,7 +230,7 @@ def test_parse_links_predicate_bracketed():
|
|||
|
||||
def test_parse_links_predicate_bracketed_with_anchor():
|
||||
"""[predicate:: [[target_path#target_anchor]]] — combined form."""
|
||||
links = DefaultFileParser.parse_links(
|
||||
links = ChunkedFileParser.parse_links(
|
||||
"[predicate:: [[target_path#target_anchor]]]",
|
||||
"src.md",
|
||||
)
|
||||
|
|
@ -245,7 +245,7 @@ def test_parse_links_predicate_bracketed_with_anchor():
|
|||
|
||||
def test_parse_links_predicate_sticks_to_first():
|
||||
"""Predicate attaches only to the immediately following wikilink."""
|
||||
links = DefaultFileParser.parse_links("pred:: [[a]] and bare [[b]]", "src.md")
|
||||
links = ChunkedFileParser.parse_links("pred:: [[a]] and bare [[b]]", "src.md")
|
||||
assert len(links) == 2
|
||||
assert links[0].predicate == "pred" and links[0].target_path == "a"
|
||||
assert links[1].predicate is None and links[1].target_path == "b"
|
||||
|
|
@ -254,7 +254,7 @@ def test_parse_links_predicate_sticks_to_first():
|
|||
|
||||
def test_parse_links_multiple_on_one_line():
|
||||
"""Multiple bare wikilinks on the same line are all captured."""
|
||||
links = DefaultFileParser.parse_links("see [[x]] and [[y#h]]", "src.md")
|
||||
links = ChunkedFileParser.parse_links("see [[x]] and [[y#h]]", "src.md")
|
||||
assert [(link.target_path, link.target_anchor) for link in links] == [
|
||||
("x", None),
|
||||
("y", "h"),
|
||||
|
|
@ -264,15 +264,15 @@ def test_parse_links_multiple_on_one_line():
|
|||
|
||||
def test_parse_links_no_match():
|
||||
"""Strings without [[]] yield no links, even if '::' appears."""
|
||||
assert len(DefaultFileParser.parse_links("no link here :: foo", "src.md")) == 0
|
||||
assert len(DefaultFileParser.parse_links("plain text without brackets", "src.md")) == 0
|
||||
assert len(DefaultFileParser.parse_links("", "src.md")) == 0
|
||||
assert len(ChunkedFileParser.parse_links("no link here :: foo", "src.md")) == 0
|
||||
assert len(ChunkedFileParser.parse_links("plain text without brackets", "src.md")) == 0
|
||||
assert len(ChunkedFileParser.parse_links("", "src.md")) == 0
|
||||
print("✓ test_parse_links_no_match passed")
|
||||
|
||||
|
||||
def test_parse_links_predicate_with_dash_and_digits():
|
||||
"""Predicate identifier accepts letters, digits, underscore, dash."""
|
||||
links = DefaultFileParser.parse_links("see-also-2:: [[target]]", "src.md")
|
||||
links = ChunkedFileParser.parse_links("see-also-2:: [[target]]", "src.md")
|
||||
assert len(links) == 1
|
||||
assert links[0].predicate == "see-also-2"
|
||||
assert links[0].target_path == "target"
|
||||
|
|
@ -297,7 +297,7 @@ def test_parse_links_in_file():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser()
|
||||
parser = ChunkedFileParser()
|
||||
file_node, _ = await parser.parse(temp_path)
|
||||
triples = {(link.predicate, link.target_path, link.target_anchor) for link in file_node.links}
|
||||
assert (None, "alpha", None) in triples
|
||||
|
|
@ -325,7 +325,7 @@ def test_parse_links_empty_when_no_content():
|
|||
fm_only_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser()
|
||||
parser = ChunkedFileParser()
|
||||
node1, _ = await parser.parse(empty_path)
|
||||
node2, _ = await parser.parse(fm_only_path)
|
||||
assert node1.links == []
|
||||
|
|
@ -353,7 +353,7 @@ def test_chunk_does_not_split_wikilink_at_boundary():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser(chunk_byte_size=100, overlap_byte_size=10)
|
||||
parser = ChunkedFileParser(chunk_byte_size=100, overlap_byte_size=10)
|
||||
_, chunks = await parser.parse(temp_path)
|
||||
# The first chunk must NOT contain a partial link.
|
||||
first = chunks[0].text
|
||||
|
|
@ -382,7 +382,7 @@ def test_chunk_does_not_split_wikilink_in_overlap():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser(chunk_byte_size=100, overlap_byte_size=20)
|
||||
parser = ChunkedFileParser(chunk_byte_size=100, overlap_byte_size=20)
|
||||
_, chunks = await parser.parse(temp_path)
|
||||
# No chunk should start mid-link.
|
||||
for c in chunks:
|
||||
|
|
@ -412,7 +412,7 @@ def test_chunk_falls_back_for_oversize_link():
|
|||
temp_path = f.name
|
||||
|
||||
try:
|
||||
parser = DefaultFileParser(chunk_byte_size=100, overlap_byte_size=10)
|
||||
parser = ChunkedFileParser(chunk_byte_size=100, overlap_byte_size=10)
|
||||
_, chunks = await parser.parse(temp_path)
|
||||
# Must terminate (not hang) and cover the whole file.
|
||||
assert len(chunks) >= 2
|
||||
|
|
@ -428,7 +428,7 @@ def test_min_chunk_and_overlap_size():
|
|||
|
||||
async def run():
|
||||
# These values should be clamped to minimums
|
||||
parser = DefaultFileParser(chunk_byte_size=1, overlap_byte_size=0)
|
||||
parser = ChunkedFileParser(chunk_byte_size=1, overlap_byte_size=0)
|
||||
assert parser.chunk_byte_size == 100 # minimum
|
||||
assert parser.overlap_byte_size == 4 # minimum
|
||||
|
||||
Loading…
Add table
Reference in a new issue