diff --git a/docs4/background_backup.md b/docs4/background_backup.md index 788c5168..1fde4440 100644 --- a/docs4/background_backup.md +++ b/docs4/background_backup.md @@ -120,7 +120,7 @@ V4更加高效的底层记忆索引 - 不支持关键词检索,这里需要Keyword倒排索引,对中文的支持较差 - [📌 历史] 描述 V3 痛点,不需要代码。 - V4版本我们重写了file parser,file store,file graph,file watcher,手写了支持增量更新倒排索引 - - file parser → [✅ `reme4/components/file_parser/`(base/bare/default/linked 四种)] + - file parser → [✅ `reme4/components/file_parser/`(base/default/chunked/linked 四种)] - file store → [✅ `reme4/components/file_store/local_file_store.py`] - file graph → [✅ `reme4/components/file_graph/`(local/nx/neo4j)] - file watcher → [✅ `reme4/components/file_watcher/lite_file_watcher.py` 基于 watchfiles awatch;`base_file_watcher.py` 抽象接口] diff --git a/docs4/reme4_index.md b/docs4/reme4_index.md index 8257ab8a..18ed6f36 100644 --- a/docs4/reme4_index.md +++ b/docs4/reme4_index.md @@ -86,9 +86,9 @@ | 类 | 文件 | 说明 | | --- | --- | --- | | `BaseFileParser` | `base_file_parser.py` | 抽象接口:`parse(path) -> (FileNode, list[FileChunk])`,提供 `_get_relative_path`。 | -| `BareFileParser` (`@R "bare"`) | `bare_file_parser.py` | 仅 stat:附件/二进制不读内容、不切块、不抽链接。 | -| `DefaultFileParser` (`@R "default"`) | `default_file_parser.py` | 字节级带 overlap 切片 + YAML front matter + wikilink 抽取(含 Dataview `predicate::`)。 | -| `LinkedFileParser` (`@R "md"`) | `linked_file_parser.py` | Markdown 专用:mistletoe AST → MdNode 树 → 章节递归分块;每个 chunk 携带完整 heading skeleton(TOC);wikilink 解析支持隐式 `.md`、folder-note、短路径歧义扇出,需注入 `file_graph` 解析目标。 | +| `DefaultFileParser` (`@R "default"`) | `default_file_parser.py` | 仅 stat:附件/二进制不读内容、不切块、不抽链接;作为无 `supported_extensions` 命中时的兜底 parser。 | +| `ChunkedFileParser` (`@R "chunked"`) | `chunked_file_parser.py` | 字节级带 overlap 切片 + YAML front matter + wikilink 抽取(含 Dataview `predicate::`)。 | +| `LinkedFileParser` (`@R "linked"`) | `linked_file_parser.py` | Markdown 专用:mistletoe AST → MdNode 树 → 章节递归分块;每个 chunk 携带完整 heading skeleton(TOC);wikilink 解析支持隐式 `.md`、folder-note、短路径歧义扇出,需注入 `file_graph` 解析目标。 | ### 4.6 File Store — `reme4/components/file_store/` diff --git a/docs4/reme_design.md b/docs4/reme_design.md index f96ca3ab..9f57f336 100644 --- a/docs4/reme_design.md +++ b/docs4/reme_design.md @@ -82,20 +82,35 @@ reme4 search query="..." backend=mcp | crud | upload/download | 其他文件 | | file | stat | path | | file | list | path | -| property | property:read | | -| property | property:update | path="My Note" status=done xx=xxx | -| property | property:delete | keys="[xxxx, xxxx]" | + | | graph | traverse | path="My Note" directtion=forward/backward depth=1 predicat=xxx | @wangce -| crud | write | path="New Note" name="xxx" description="xxx" content="# Hello" (4 字段都必填,frontmatter 只写 name/description) | +| crud | write | path="New Note" name="xxx" description="xxx" metadata={}, content="# Hello" (4 字段都必填,frontmatter 只写 name/description) | | crud | read | path="Templates/Recipe.md" | | crud | edit | path="Templates/Recipe.md" old="xxx" new="xxx" | | crud | append | path="My Note" content="New line" | -| crud | prepend | path="My Note" content="New line" | + | crud | delete | path="My Note | daily:crud | daily:xxx | 与 crud 参数保持一致 | +- daily:resolve name=xxxx (符合一定规范 win下要求) +- daily:list date=xxxx 返回path +- daily:index + +frontmatter read path +frontmatter update path metadata={} +frontmatter delete path keys=[] + +delete path +download path=xxx(内部相对路径)download_path=(外部绝对路径,可选) +upload path=xxx(外部绝对路径)description="xxx" metadata=xxx 返回内部相对路径 加metadata +stat path +list path +mv path=xxx new_path=xxx + +traverse path=xxx direction=xxx depth=xxx + # 日记类型 | 类型 | 路径 | 说明 | diff --git a/reme4/components/file_parser/__init__.py b/reme4/components/file_parser/__init__.py index c5b62bd9..bf70a0cd 100644 --- a/reme4/components/file_parser/__init__.py +++ b/reme4/components/file_parser/__init__.py @@ -1,8 +1,8 @@ """File parser components.""" -from .bare_file_parser import BareFileParser from .base_file_parser import BaseFileParser +from .chunked_file_parser import ChunkedFileParser from .default_file_parser import DefaultFileParser from .linked_file_parser import LinkedFileParser -__all__ = ["BareFileParser", "BaseFileParser", "DefaultFileParser", "LinkedFileParser"] +__all__ = ["BaseFileParser", "ChunkedFileParser", "DefaultFileParser", "LinkedFileParser"] diff --git a/reme4/components/file_parser/bare_file_parser.py b/reme4/components/file_parser/bare_file_parser.py deleted file mode 100644 index 55b12e5c..00000000 --- a/reme4/components/file_parser/bare_file_parser.py +++ /dev/null @@ -1,22 +0,0 @@ -"""Stat-only parser for attachment/binary files.""" - -from pathlib import Path - -from .base_file_parser import BaseFileParser -from ..component_registry import R -from ...schema import FileChunk, FileNode - - -@R.register("bare") -class BareFileParser(BaseFileParser): - """Stat-only parser for attachment/binary files. - - No content read, no chunking, no link extraction. The resulting FileNode - has empty links and chunk_ids; front_matter carries mime and size so - retrieval can filter by file type without reopening the file. - """ - - async def parse(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]: - file_path = Path(path) - stat = file_path.stat() - return FileNode(path=self.to_vault_relative(path), st_mtime=stat.st_mtime, links=[], chunk_ids=[]), [] diff --git a/reme4/components/file_parser/chunked_file_parser.py b/reme4/components/file_parser/chunked_file_parser.py new file mode 100644 index 00000000..2d892157 --- /dev/null +++ b/reme4/components/file_parser/chunked_file_parser.py @@ -0,0 +1,180 @@ +"""File parser with byte-based overlapping chunking.""" + +import re +from bisect import bisect_right +from pathlib import Path + +import aiofiles +import yaml + +from .base_file_parser import BaseFileParser +from ..component_registry import R +from ...schema import FileChunk, FileFrontMatter, FileLink, FileNode + +# Single-pass wikilink + optional dataview predicate. +# Covers: [[X]] / ![[X]] / [[X#h]] / [[X|alias]] / pred:: [[X]] / [pred:: [[X]]] +# - predicate group: optional leading '[' (dataview inline-bracket form), an identifier, +# then '::' — the whole prefix is non-capturing-optional so bare wikilinks still match. +# - optional '!' prefix matches the embed form (![[X]]). +# - target / anchor / alias all forbid '\n' so a wikilink cannot span lines. +# - alias '|...': consumed but not captured (we don't need display text). +_LINK_RE = re.compile( + r"(?:\[?\s*(?P[A-Za-z][\w-]*)\s*::\s*)?" + r"!?\[\[\s*(?P[^\[\]|#\n]+?)" + r"(?:#(?P[^\[\]|\n]+?))?" + r"\s*(?:\|[^\[\]\n]*?)?\s*]]", +) + + +@R.register("chunked") +class ChunkedFileParser(BaseFileParser): + """Parser that splits files into byte-based overlapping chunks.""" + + def __init__(self, encoding: str = "utf-8", chunk_byte_size: int = 10000, overlap_byte_size: int = 100, **kwargs): + super().__init__(**kwargs) + self.encoding = encoding + self.chunk_byte_size = max(100, chunk_byte_size) + self.overlap_byte_size = max(4, overlap_byte_size) + + @staticmethod + def parse_links(content: str, source_path: str) -> list[FileLink]: + """Extract wikilinks with optional dataview predicate as outgoing FileLinks.""" + links: list[FileLink] = [] + for m in _LINK_RE.finditer(content): + target = m["target"].strip() + if not target: + continue + anchor = m["anchor"] + links.append( + FileLink( + source_path=source_path, + target_path=target, + target_anchor=anchor.strip() if anchor else None, + predicate=m["predicate"], + ), + ) + return links + + @staticmethod + def _parse_front_matter(text: str) -> tuple[FileFrontMatter, str]: + """Parse YAML front matter delimited by ---, return (front_matter, remaining).""" + if not text.startswith("---"): + return FileFrontMatter(), text + end_idx = text.find("\n---", 3) + if end_idx == -1: + return FileFrontMatter(), text + try: + data = yaml.safe_load(text[3:end_idx].strip()) or {} + front_matter = FileFrontMatter(**(data if isinstance(data, dict) else {})) + except yaml.YAMLError: + front_matter = FileFrontMatter() + return front_matter, text[end_idx + 4 :].lstrip("\n") + + async def parse(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]: + file_path = Path(path) + stat = file_path.stat() + rel_path = self.to_vault_relative(path) + + async with aiofiles.open(file_path, encoding=self.encoding) as f: + text = await f.read() + + if not text: + return FileNode(path=rel_path, st_mtime=stat.st_mtime), [] + + is_markdown = file_path.suffix.lower() == ".md" + if is_markdown: + front_matter, content = self._parse_front_matter(text) + if not content: + return FileNode(path=rel_path, st_mtime=stat.st_mtime, front_matter=front_matter), [] + links = self.parse_links(content, rel_path) + else: + front_matter = FileFrontMatter() + content = text + links = [] + + chunks = self._chunk_content(content, rel_path, parse_links=is_markdown) + chunk_ids = [c.id for c in chunks] + return ( + FileNode( + path=rel_path, + st_mtime=stat.st_mtime, + front_matter=front_matter, + links=links, + chunk_ids=chunk_ids, + ), + chunks, + ) + + def _link_byte_spans(self, content: str) -> list[tuple[int, int]]: + """Return [start, end) byte spans of every wikilink in content.""" + spans: list[tuple[int, int]] = [] + last_char, last_byte = 0, 0 + for m in _LINK_RE.finditer(content): + last_byte += len(content[last_char : m.start()].encode(self.encoding)) + match_bytes = len(m.group(0).encode(self.encoding)) + spans.append((last_byte, last_byte + match_bytes)) + last_byte += match_bytes + last_char = m.end() + return spans + + @staticmethod + def _span_containing( + pos: int, + spans: list[tuple[int, int]], + starts: list[int], + ) -> tuple[int, int] | None: + """Return the span strictly containing pos (s < pos < e), or None.""" + idx = bisect_right(starts, pos) - 1 + if idx < 0: + return None + s, e = spans[idx] + return (s, e) if s < pos < e else None + + def _chunk_content(self, content: str, rel_path: str, parse_links: bool = True) -> list[FileChunk]: + """Split content into overlapping byte-range chunks, avoiding cuts inside wikilinks. + + When ``parse_links`` is False, skip wikilink span computation and boundary checks + — used for non-markdown files where wikilink semantics don't apply. + """ + content_bytes = content.encode(self.encoding) + n = len(content_bytes) + newline_positions = [i for i, b in enumerate(content_bytes) if b == ord("\n")] + if parse_links: + link_spans = self._link_byte_spans(content) + link_starts = [s for s, _ in link_spans] + else: + link_spans: list[tuple[int, int]] = [] + link_starts: list[int] = [] + # Refuse to retreat past half of chunk_byte_size; falls back to hard cut + # for pathologically long links so we always make forward progress. + min_chunk = self.chunk_byte_size // 2 + chunks: list[FileChunk] = [] + start = 0 + + while start < n: + end = min(start + self.chunk_byte_size, n) + if end < n: + span = self._span_containing(end, link_spans, link_starts) + if span is not None and span[0] - start >= min_chunk: + end = span[0] + + chunk_text = content_bytes[start:end].decode(self.encoding, errors="ignore") + start_line = bisect_right(newline_positions, start - 1) + 1 + end_line = bisect_right(newline_positions, end - 1) + 1 + if content_bytes[end - 1] == ord("\n"): + end_line -= 1 + chunks.append( + FileChunk(path=rel_path, start_line=start_line, end_line=end_line, text=chunk_text).set_hash_id(), + ) + if end >= n: + break + + next_start = end - self.overlap_byte_size + span = self._span_containing(next_start, link_spans, link_starts) + if span is not None: + next_start = span[1] + if next_start <= start: + next_start = end + start = next_start + + return chunks diff --git a/reme4/components/file_parser/default_file_parser.py b/reme4/components/file_parser/default_file_parser.py index dfb68440..c799a9c3 100644 --- a/reme4/components/file_parser/default_file_parser.py +++ b/reme4/components/file_parser/default_file_parser.py @@ -1,180 +1,22 @@ -"""Default file parser with byte-based overlapping chunking.""" +"""Stat-only parser for attachment/binary files.""" -import re -from bisect import bisect_right from pathlib import Path -import aiofiles -import yaml - from .base_file_parser import BaseFileParser from ..component_registry import R -from ...schema import FileChunk, FileFrontMatter, FileLink, FileNode - -# Single-pass wikilink + optional dataview predicate. -# Covers: [[X]] / ![[X]] / [[X#h]] / [[X|alias]] / pred:: [[X]] / [pred:: [[X]]] -# - predicate group: optional leading '[' (dataview inline-bracket form), an identifier, -# then '::' — the whole prefix is non-capturing-optional so bare wikilinks still match. -# - optional '!' prefix matches the embed form (![[X]]). -# - target / anchor / alias all forbid '\n' so a wikilink cannot span lines. -# - alias '|...': consumed but not captured (we don't need display text). -_LINK_RE = re.compile( - r"(?:\[?\s*(?P[A-Za-z][\w-]*)\s*::\s*)?" - r"!?\[\[\s*(?P[^\[\]|#\n]+?)" - r"(?:#(?P[^\[\]|\n]+?))?" - r"\s*(?:\|[^\[\]\n]*?)?\s*]]", -) +from ...schema import FileChunk, FileNode @R.register("default") class DefaultFileParser(BaseFileParser): - """Parser that splits files into byte-based overlapping chunks.""" + """Stat-only parser for attachment/binary files. - def __init__(self, encoding: str = "utf-8", chunk_byte_size: int = 10000, overlap_byte_size: int = 100, **kwargs): - super().__init__(**kwargs) - self.encoding = encoding - self.chunk_byte_size = max(100, chunk_byte_size) - self.overlap_byte_size = max(4, overlap_byte_size) - - @staticmethod - def parse_links(content: str, source_path: str) -> list[FileLink]: - """Extract wikilinks with optional dataview predicate as outgoing FileLinks.""" - links: list[FileLink] = [] - for m in _LINK_RE.finditer(content): - target = m["target"].strip() - if not target: - continue - anchor = m["anchor"] - links.append( - FileLink( - source_path=source_path, - target_path=target, - target_anchor=anchor.strip() if anchor else None, - predicate=m["predicate"], - ), - ) - return links - - @staticmethod - def _parse_front_matter(text: str) -> tuple[FileFrontMatter, str]: - """Parse YAML front matter delimited by ---, return (front_matter, remaining).""" - if not text.startswith("---"): - return FileFrontMatter(), text - end_idx = text.find("\n---", 3) - if end_idx == -1: - return FileFrontMatter(), text - try: - data = yaml.safe_load(text[3:end_idx].strip()) or {} - front_matter = FileFrontMatter(**(data if isinstance(data, dict) else {})) - except yaml.YAMLError: - front_matter = FileFrontMatter() - return front_matter, text[end_idx + 4 :].lstrip("\n") + No content read, no chunking, no link extraction. The resulting FileNode + has empty links and chunk_ids; front_matter carries mime and size so + retrieval can filter by file type without reopening the file. + """ async def parse(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]: file_path = Path(path) stat = file_path.stat() - rel_path = self.to_vault_relative(path) - - async with aiofiles.open(file_path, encoding=self.encoding) as f: - text = await f.read() - - if not text: - return FileNode(path=rel_path, st_mtime=stat.st_mtime), [] - - is_markdown = file_path.suffix.lower() == ".md" - if is_markdown: - front_matter, content = self._parse_front_matter(text) - if not content: - return FileNode(path=rel_path, st_mtime=stat.st_mtime, front_matter=front_matter), [] - links = self.parse_links(content, rel_path) - else: - front_matter = FileFrontMatter() - content = text - links = [] - - chunks = self._chunk_content(content, rel_path, parse_links=is_markdown) - chunk_ids = [c.id for c in chunks] - return ( - FileNode( - path=rel_path, - st_mtime=stat.st_mtime, - front_matter=front_matter, - links=links, - chunk_ids=chunk_ids, - ), - chunks, - ) - - def _link_byte_spans(self, content: str) -> list[tuple[int, int]]: - """Return [start, end) byte spans of every wikilink in content.""" - spans: list[tuple[int, int]] = [] - last_char, last_byte = 0, 0 - for m in _LINK_RE.finditer(content): - last_byte += len(content[last_char : m.start()].encode(self.encoding)) - match_bytes = len(m.group(0).encode(self.encoding)) - spans.append((last_byte, last_byte + match_bytes)) - last_byte += match_bytes - last_char = m.end() - return spans - - @staticmethod - def _span_containing( - pos: int, - spans: list[tuple[int, int]], - starts: list[int], - ) -> tuple[int, int] | None: - """Return the span strictly containing pos (s < pos < e), or None.""" - idx = bisect_right(starts, pos) - 1 - if idx < 0: - return None - s, e = spans[idx] - return (s, e) if s < pos < e else None - - def _chunk_content(self, content: str, rel_path: str, parse_links: bool = True) -> list[FileChunk]: - """Split content into overlapping byte-range chunks, avoiding cuts inside wikilinks. - - When ``parse_links`` is False, skip wikilink span computation and boundary checks - — used for non-markdown files where wikilink semantics don't apply. - """ - content_bytes = content.encode(self.encoding) - n = len(content_bytes) - newline_positions = [i for i, b in enumerate(content_bytes) if b == ord("\n")] - if parse_links: - link_spans = self._link_byte_spans(content) - link_starts = [s for s, _ in link_spans] - else: - link_spans: list[tuple[int, int]] = [] - link_starts: list[int] = [] - # Refuse to retreat past half of chunk_byte_size; falls back to hard cut - # for pathologically long links so we always make forward progress. - min_chunk = self.chunk_byte_size // 2 - chunks: list[FileChunk] = [] - start = 0 - - while start < n: - end = min(start + self.chunk_byte_size, n) - if end < n: - span = self._span_containing(end, link_spans, link_starts) - if span is not None and span[0] - start >= min_chunk: - end = span[0] - - chunk_text = content_bytes[start:end].decode(self.encoding, errors="ignore") - start_line = bisect_right(newline_positions, start - 1) + 1 - end_line = bisect_right(newline_positions, end - 1) + 1 - if content_bytes[end - 1] == ord("\n"): - end_line -= 1 - chunks.append( - FileChunk(path=rel_path, start_line=start_line, end_line=end_line, text=chunk_text).set_hash_id(), - ) - if end >= n: - break - - next_start = end - self.overlap_byte_size - span = self._span_containing(next_start, link_spans, link_starts) - if span is not None: - next_start = span[1] - if next_start <= start: - next_start = end - start = next_start - - return chunks + return FileNode(path=self.to_vault_relative(path), st_mtime=stat.st_mtime, links=[], chunk_ids=[]), [] diff --git a/reme4/config/default.yaml b/reme4/config/default.yaml index 1e7ff46c..01464a34 100644 --- a/reme4/config/default.yaml +++ b/reme4/config/default.yaml @@ -264,20 +264,20 @@ components: backend: local file_parser: - default: - backend: default + linked: + backend: linked + supported_extensions: + - md + chunked: + backend: chunked supported_extensions: - txt - html - json - yaml - py - bare: - backend: bare - linked: - backend: linked - supported_extensions: - - md + default: + backend: default keyword_index: default: diff --git a/reme4/steps/base_step.py b/reme4/steps/base_step.py index ce1cf675..b3a50aa6 100644 --- a/reme4/steps/base_step.py +++ b/reme4/steps/base_step.py @@ -138,7 +138,7 @@ class BaseStep(ABC): """Parse ``path`` with the parser whose ``supported_extensions`` claims its suffix. First registered match wins (config insertion order). Falls back to the - ``bare`` parser (stat-only) when no parser claims the suffix — that's + ``default`` parser (stat-only) when no parser claims the suffix — that's how attachments / binaries / unknown types still produce a FileNode. """ assert self.app_context is not None @@ -154,10 +154,12 @@ class BaseStep(ABC): break if parser is None: - parser = file_parser_dict.get("bare") + parser = file_parser_dict.get("default") if parser is None: - raise RuntimeError(f"No file parser supports {path} (suffix={suffix!r}) and no 'bare' parser is configured") + raise RuntimeError( + f"No file parser supports {path} (suffix={suffix!r}) and no 'default' parser is configured", + ) return await parser.parse(path) diff --git a/tests4/unittest/test_background_steps.py b/tests4/unittest/test_background_steps.py index 9ac1787b..7a0af518 100644 --- a/tests4/unittest/test_background_steps.py +++ b/tests4/unittest/test_background_steps.py @@ -18,7 +18,7 @@ from typing import Any from watchfiles import Change -from reme4.components.file_parser import DefaultFileParser +from reme4.components.file_parser import ChunkedFileParser from reme4.components.file_store import LocalFileStore from reme4.components.runtime_context import RuntimeContext from reme4.schema import Response @@ -86,9 +86,9 @@ async def _make_update_step( suffix_filters: list[str] | None = None, recursive: bool = True, dump: bool = True, -) -> tuple[_RecordingUpdateStoreStep, RuntimeContext, LocalFileStore, DefaultFileParser]: +) -> tuple[_RecordingUpdateStoreStep, RuntimeContext, LocalFileStore, ChunkedFileParser]: fs = LocalFileStore(store_name="test_store", embedding_model="") - parser = DefaultFileParser() + parser = ChunkedFileParser() await fs.start() await parser.start() step = _RecordingUpdateStoreStep( @@ -105,7 +105,7 @@ async def _make_update_step( return step, context, fs, parser -async def _teardown(fs: LocalFileStore, parser: DefaultFileParser) -> None: +async def _teardown(fs: LocalFileStore, parser: ChunkedFileParser) -> None: await parser.close() await fs.close() diff --git a/tests4/unittest/test_default_file_parser.py b/tests4/unittest/test_chunked_file_parser.py similarity index 90% rename from tests4/unittest/test_default_file_parser.py rename to tests4/unittest/test_chunked_file_parser.py index c069b888..6529beb0 100644 --- a/tests4/unittest/test_default_file_parser.py +++ b/tests4/unittest/test_chunked_file_parser.py @@ -1,10 +1,10 @@ -"""Tests for DefaultFileParser.""" +"""Tests for ChunkedFileParser.""" import asyncio import os import tempfile -from reme4.components.file_parser import DefaultFileParser +from reme4.components.file_parser import ChunkedFileParser # Add parent path for import @@ -18,7 +18,7 @@ def test_parse_empty_file(): temp_path = f.name try: - parser = DefaultFileParser() + parser = ChunkedFileParser() file_node, chunks = await parser.parse(temp_path) assert file_node.path == temp_path assert len(chunks) == 0 @@ -39,7 +39,7 @@ def test_parse_small_file(): temp_path = f.name try: - parser = DefaultFileParser(chunk_byte_size=10000) + parser = ChunkedFileParser(chunk_byte_size=10000) _, chunks = await parser.parse(temp_path) assert len(chunks) == 1 assert chunks[0].start_line == 1 @@ -63,7 +63,7 @@ def test_parse_multiline_file(): temp_path = f.name try: - parser = DefaultFileParser(chunk_byte_size=10000) + parser = ChunkedFileParser(chunk_byte_size=10000) _, chunks = await parser.parse(temp_path) assert len(chunks) == 1 assert chunks[0].start_line == 1 @@ -87,7 +87,7 @@ def test_parse_chunked_file(): temp_path = f.name try: - parser = DefaultFileParser(chunk_byte_size=5000, overlap_byte_size=100) + parser = ChunkedFileParser(chunk_byte_size=5000, overlap_byte_size=100) _, chunks = await parser.parse(temp_path) assert len(chunks) > 1, f"Expected multiple chunks, got {len(chunks)}" # Verify overlap by checking that consecutive chunks share some content @@ -109,7 +109,7 @@ def test_parse_with_custom_encoding(): temp_path = f.name try: - parser = DefaultFileParser(encoding="utf-8") + parser = ChunkedFileParser(encoding="utf-8") _, chunks = await parser.parse(temp_path) assert len(chunks) >= 1 assert "你好世界" in chunks[0].text @@ -130,7 +130,7 @@ def test_file_node_properties(): temp_path = f.name try: - parser = DefaultFileParser() + parser = ChunkedFileParser() file_node, _ = await parser.parse(temp_path) assert hasattr(file_node, "path") assert hasattr(file_node, "st_mtime") @@ -152,7 +152,7 @@ def test_file_chunk_properties(): temp_path = f.name try: - parser = DefaultFileParser() + parser = ChunkedFileParser() _, chunks = await parser.parse(temp_path) chunk = chunks[0] assert hasattr(chunk, "path") @@ -171,7 +171,7 @@ def test_file_chunk_properties(): def test_parse_links_bare(): """Bare wikilink: [[target]].""" - links = DefaultFileParser.parse_links("see [[note]]", "src.md") + links = ChunkedFileParser.parse_links("see [[note]]", "src.md") assert len(links) == 1 link = links[0] assert link.source_path == "src.md" @@ -183,7 +183,7 @@ def test_parse_links_bare(): def test_parse_links_with_anchor(): """Wikilink with anchor: [[target#anchor]].""" - links = DefaultFileParser.parse_links("see [[note#section A]]", "src.md") + links = ChunkedFileParser.parse_links("see [[note#section A]]", "src.md") assert len(links) == 1 assert links[0].target_path == "note" assert links[0].target_anchor == "section A" @@ -193,7 +193,7 @@ def test_parse_links_with_anchor(): def test_parse_links_alias_dropped(): """Alias after '|' is consumed but not captured as anchor.""" - links = DefaultFileParser.parse_links("see [[note|display text]]", "src.md") + links = ChunkedFileParser.parse_links("see [[note|display text]]", "src.md") assert len(links) == 1 assert links[0].target_path == "note" assert links[0].target_anchor is None @@ -202,7 +202,7 @@ def test_parse_links_alias_dropped(): def test_parse_links_anchor_and_alias(): """[[target#anchor|alias]] — anchor captured, alias dropped.""" - links = DefaultFileParser.parse_links("see [[note#sec|disp]]", "src.md") + links = ChunkedFileParser.parse_links("see [[note#sec|disp]]", "src.md") assert len(links) == 1 assert links[0].target_path == "note" assert links[0].target_anchor == "sec" @@ -211,7 +211,7 @@ def test_parse_links_anchor_and_alias(): def test_parse_links_predicate_simple(): """Dataview inline: predicate:: [[target]].""" - links = DefaultFileParser.parse_links("author:: [[Alice]]", "src.md") + links = ChunkedFileParser.parse_links("author:: [[Alice]]", "src.md") assert len(links) == 1 assert links[0].predicate == "author" assert links[0].target_path == "Alice" @@ -221,7 +221,7 @@ def test_parse_links_predicate_simple(): def test_parse_links_predicate_bracketed(): """Dataview inline-bracket: [predicate:: [[target]]].""" - links = DefaultFileParser.parse_links("text [author:: [[Alice]]] more", "src.md") + links = ChunkedFileParser.parse_links("text [author:: [[Alice]]] more", "src.md") assert len(links) == 1 assert links[0].predicate == "author" assert links[0].target_path == "Alice" @@ -230,7 +230,7 @@ def test_parse_links_predicate_bracketed(): def test_parse_links_predicate_bracketed_with_anchor(): """[predicate:: [[target_path#target_anchor]]] — combined form.""" - links = DefaultFileParser.parse_links( + links = ChunkedFileParser.parse_links( "[predicate:: [[target_path#target_anchor]]]", "src.md", ) @@ -245,7 +245,7 @@ def test_parse_links_predicate_bracketed_with_anchor(): def test_parse_links_predicate_sticks_to_first(): """Predicate attaches only to the immediately following wikilink.""" - links = DefaultFileParser.parse_links("pred:: [[a]] and bare [[b]]", "src.md") + links = ChunkedFileParser.parse_links("pred:: [[a]] and bare [[b]]", "src.md") assert len(links) == 2 assert links[0].predicate == "pred" and links[0].target_path == "a" assert links[1].predicate is None and links[1].target_path == "b" @@ -254,7 +254,7 @@ def test_parse_links_predicate_sticks_to_first(): def test_parse_links_multiple_on_one_line(): """Multiple bare wikilinks on the same line are all captured.""" - links = DefaultFileParser.parse_links("see [[x]] and [[y#h]]", "src.md") + links = ChunkedFileParser.parse_links("see [[x]] and [[y#h]]", "src.md") assert [(link.target_path, link.target_anchor) for link in links] == [ ("x", None), ("y", "h"), @@ -264,15 +264,15 @@ def test_parse_links_multiple_on_one_line(): def test_parse_links_no_match(): """Strings without [[]] yield no links, even if '::' appears.""" - assert len(DefaultFileParser.parse_links("no link here :: foo", "src.md")) == 0 - assert len(DefaultFileParser.parse_links("plain text without brackets", "src.md")) == 0 - assert len(DefaultFileParser.parse_links("", "src.md")) == 0 + assert len(ChunkedFileParser.parse_links("no link here :: foo", "src.md")) == 0 + assert len(ChunkedFileParser.parse_links("plain text without brackets", "src.md")) == 0 + assert len(ChunkedFileParser.parse_links("", "src.md")) == 0 print("✓ test_parse_links_no_match passed") def test_parse_links_predicate_with_dash_and_digits(): """Predicate identifier accepts letters, digits, underscore, dash.""" - links = DefaultFileParser.parse_links("see-also-2:: [[target]]", "src.md") + links = ChunkedFileParser.parse_links("see-also-2:: [[target]]", "src.md") assert len(links) == 1 assert links[0].predicate == "see-also-2" assert links[0].target_path == "target" @@ -297,7 +297,7 @@ def test_parse_links_in_file(): temp_path = f.name try: - parser = DefaultFileParser() + parser = ChunkedFileParser() file_node, _ = await parser.parse(temp_path) triples = {(link.predicate, link.target_path, link.target_anchor) for link in file_node.links} assert (None, "alpha", None) in triples @@ -325,7 +325,7 @@ def test_parse_links_empty_when_no_content(): fm_only_path = f.name try: - parser = DefaultFileParser() + parser = ChunkedFileParser() node1, _ = await parser.parse(empty_path) node2, _ = await parser.parse(fm_only_path) assert node1.links == [] @@ -353,7 +353,7 @@ def test_chunk_does_not_split_wikilink_at_boundary(): temp_path = f.name try: - parser = DefaultFileParser(chunk_byte_size=100, overlap_byte_size=10) + parser = ChunkedFileParser(chunk_byte_size=100, overlap_byte_size=10) _, chunks = await parser.parse(temp_path) # The first chunk must NOT contain a partial link. first = chunks[0].text @@ -382,7 +382,7 @@ def test_chunk_does_not_split_wikilink_in_overlap(): temp_path = f.name try: - parser = DefaultFileParser(chunk_byte_size=100, overlap_byte_size=20) + parser = ChunkedFileParser(chunk_byte_size=100, overlap_byte_size=20) _, chunks = await parser.parse(temp_path) # No chunk should start mid-link. for c in chunks: @@ -412,7 +412,7 @@ def test_chunk_falls_back_for_oversize_link(): temp_path = f.name try: - parser = DefaultFileParser(chunk_byte_size=100, overlap_byte_size=10) + parser = ChunkedFileParser(chunk_byte_size=100, overlap_byte_size=10) _, chunks = await parser.parse(temp_path) # Must terminate (not hang) and cover the whole file. assert len(chunks) >= 2 @@ -428,7 +428,7 @@ def test_min_chunk_and_overlap_size(): async def run(): # These values should be clamped to minimums - parser = DefaultFileParser(chunk_byte_size=1, overlap_byte_size=0) + parser = ChunkedFileParser(chunk_byte_size=1, overlap_byte_size=0) assert parser.chunk_byte_size == 100 # minimum assert parser.overlap_byte_size == 4 # minimum