diff --git a/reme/components/file_chunker/default_file_chunker.py b/reme/components/file_chunker/default_file_chunker.py index 96b7a6eb..a2c57f3c 100644 --- a/reme/components/file_chunker/default_file_chunker.py +++ b/reme/components/file_chunker/default_file_chunker.py @@ -59,7 +59,7 @@ class DefaultFileChunker(BaseFileChunker): content = text links = [] - chunks = self._chunk_content(content, rel_path, parse_links=is_markdown) + chunks = self.chunk_content(content, rel_path, parse_links=is_markdown) chunk_ids = [c.id for c in chunks] return ( FileNode( @@ -97,7 +97,7 @@ class DefaultFileChunker(BaseFileChunker): s, e = spans[idx] return (s, e) if s < pos < e else None - def _chunk_content(self, content: str, rel_path: str, parse_links: bool = True) -> list[FileChunk]: + def chunk_content(self, content: str, rel_path: str, parse_links: bool = True) -> list[FileChunk]: """Split content into overlapping byte-range chunks, avoiding cuts inside wikilinks. When ``parse_links`` is False, skip wikilink span computation and boundary checks diff --git a/reme/components/file_chunker/markdown_file_chunker.py b/reme/components/file_chunker/markdown_file_chunker.py index 9847f669..373af3c6 100644 --- a/reme/components/file_chunker/markdown_file_chunker.py +++ b/reme/components/file_chunker/markdown_file_chunker.py @@ -1,19 +1,22 @@ """Markdown file chunker — frontmatter + wikilink graph + AST tree chunks. -Each chunk carries the **complete heading skeleton** of the document -with its content inlined under the section that owns it; other sections -appear as bare headings so the reader always sees a full document map. +When ``embed_toc`` is enabled, a chunk that starts inside a section carries +only that section's ancestor heading breadcrumb. Sibling headings remain in +the document stream and are stored once, avoiding the quadratic growth caused +by repeating the complete document outline in every chunk. -Pipeline: build mistletoe AST → ``MdNode`` tree (sections nest by -heading level) → recursive chunk (try whole subtree; on overflow walk -children — body siblings pack as a run, subsections recurse). Leaf -blocks (table / code / list / paragraph) split on internal boundaries -and each piece is annotated ``[Part X/N]``. Wikilink extraction is +Pipeline: count headings without an AST → use plain-text byte chunks when the +configured section limit is exceeded; otherwise build a mistletoe AST → +``MdNode`` tree (sections nest by heading level) → recursively chunk children +and merge adjacent small subtrees at their parent. Leaf blocks (table / code / +list / paragraph) split on internal boundaries and each piece is annotated +``[Part X/N]``. Wikilink extraction is delegated to :class:`reme.utils.wikilink_handler.WikilinkHandler` — the single source of truth for ``[[...]]`` syntax (including Dataview-style typed predicates). """ +import re from dataclasses import dataclass, field from pathlib import Path from typing import Any @@ -23,6 +26,7 @@ from pydantic import ValidationError from .base_file_chunker import BaseFileChunker +from .default_file_chunker import DefaultFileChunker from ..component_registry import R from ...schema import ( FileChunk, @@ -33,6 +37,10 @@ from ...utils.wikilink_handler import WikilinkHandler # -- AST node + helpers --------------------------------------------------- +_ATX_HEADING_RE = re.compile(r"^ {0,3}#{1,6}(?:[ \t]+|$)") +_SETEXT_HEADING_RE = re.compile(r"^ {0,3}(?:=+|-+)[ \t]*$") +_FENCE_RE = re.compile(r"^ {0,3}(`{3,}|~{3,})") + @dataclass class MdNode: @@ -40,9 +48,7 @@ class MdNode: heading) / ``body`` (one mistletoe block; ``block`` keeps the original). ``text`` is the rendered subtree (own heading excluded for sections). - ``desc_toc`` caches the section-only DFS outline of descendants - (own heading excluded), used as the TOC suffix when emitting chunks - inside a section. Line ranges span the full subtree. + Line ranges span the full subtree. """ kind: str # "root" | "section" | "body" @@ -53,7 +59,6 @@ class MdNode: text: str = "" start_line: int = 0 end_line: int = 0 - desc_toc: str = "" def _heading_text(node: Any, renderer) -> str: @@ -66,16 +71,13 @@ def _heading_text(node: Any, renderer) -> str: def _finalize(n: MdNode) -> None: """Bottom-up pass: propagate line ranges, populate ``n.text`` (rendered - subtree, own heading excluded for sections) and ``n.desc_toc`` (DFS - section outline of descendants).""" + subtree, own heading excluded for sections).""" parts: list[str] = [] - desc_lines: list[str] = [] for c in n.children: _finalize(c) if c.kind == "section": heading = f"{'#' * c.level} {c.heading or ''}" parts.append(f"{heading}\n\n{c.text}" if c.text else heading) - desc_lines.append(f"{heading}\n\n{c.desc_toc}" if c.desc_toc else heading) elif c.text: parts.append(c.text) if n.children: @@ -86,7 +88,6 @@ def _finalize(n: MdNode) -> None: n.end_line = n.start_line if n.kind != "body": n.text = "\n\n".join(parts) - n.desc_toc = "\n\n".join(desc_lines) def _toc_join(*parts: str) -> str: @@ -94,27 +95,20 @@ def _toc_join(*parts: str) -> str: return "\n\n".join(p for p in parts if p) -def _subtree_toc(n: MdNode) -> str: - """Section's heading + descendants TOC — its contribution to a parent's - ``desc_toc``. For root (no own heading) this is just ``desc_toc``.""" - if n.kind != "section" or n.heading is None: - return n.desc_toc - heading = f"{'#' * n.level} {n.heading}" - return f"{heading}\n\n{n.desc_toc}" if n.desc_toc else heading - - # -- Chunker -------------------------------------------------------------- @R.register("markdown") class MarkdownFileChunker(BaseFileChunker): - """Markdown chunker: frontmatter + wikilink edges + full-skeleton chunks.""" + """Markdown chunker with breadcrumb context and adjacent-section packing.""" def __init__( self, encoding: str = "utf-8", chunk_chars: int = 10000, embed_toc: bool = True, + max_ast_sections: int | None = 100, + default_chunker: str = "default", include_frontmatter_in_metadata: bool = False, include_frontmatter_keys_in_metadata: list[str] | None = None, **kwargs, @@ -123,22 +117,32 @@ class MarkdownFileChunker(BaseFileChunker): self.encoding = encoding self.chunk_chars = max(100, chunk_chars) self.embed_toc = embed_toc + self.max_ast_sections = max(0, max_ast_sections) if max_ast_sections is not None else None + self.default_chunker = self.bind(default_chunker, BaseFileChunker, optional=False) self.include_frontmatter_in_metadata = include_frontmatter_in_metadata self.include_frontmatter_keys_in_metadata = list(include_frontmatter_keys_in_metadata or []) async def chunk(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]: - from mistletoe.markdown_renderer import MarkdownRenderer - from mistletoe.block_token import Document - file_path = Path(path) rel_path = self.to_workspace_relative(path) front_matter, content, line_offset = self._parse_front_matter(file_path.read_text(encoding=self.encoding)) chunks: list[FileChunk] = [] if content and content.strip(): - with MarkdownRenderer() as renderer: - tree = self._build_tree(Document(content), renderer, line_offset=line_offset) - chunks = self._chunk_node(tree, "", "", rel_path, renderer) + section_count = self._count_sections(content, stop_after=self.max_ast_sections) + if self.max_ast_sections is not None and section_count > self.max_ast_sections: + self.logger.info( + f"Markdown AST skipped for {rel_path}: sections>{self.max_ast_sections}; " + "using plain-text chunking", + ) + chunks = self._chunk_plain_text(content, rel_path, line_offset) + else: + from mistletoe.markdown_renderer import MarkdownRenderer + from mistletoe.block_token import Document + + with MarkdownRenderer() as renderer: + tree = self._build_tree(Document(content), renderer, line_offset=line_offset) + chunks = self._chunk_node(tree, (), rel_path, renderer) if self.include_frontmatter_in_metadata: chunk_metadata = self._chunk_metadata( front_matter, @@ -158,6 +162,66 @@ class MarkdownFileChunker(BaseFileChunker): ) return node, chunks + @staticmethod + def _count_sections(content: str, stop_after: int | None = None) -> int: + """Count block-level ATX/Setext headings without building a Markdown AST. + + Fenced code content is ignored. When ``stop_after`` is set, return as + soon as the count exceeds it because the caller only needs to choose + between AST and plain-text chunking. + """ + count = 0 + fence_char = "" + fence_length = 0 + previous_text = False + + line_start = 0 + while line_start <= len(content): + newline = content.find("\n", line_start) + if newline < 0: + line = content[line_start:] + else: + line = content[line_start:newline] + if line.endswith("\r"): + line = line[:-1] + + fence_match = _FENCE_RE.match(line) + if fence_match: + marker = fence_match.group(1) + if not fence_char: + fence_char = marker[0] + fence_length = len(marker) + elif marker[0] == fence_char and len(marker) >= fence_length and not line[fence_match.end() :].strip(): + fence_char = "" + fence_length = 0 + previous_text = False + elif not fence_char: + is_atx = _ATX_HEADING_RE.match(line) is not None + is_setext = previous_text and _SETEXT_HEADING_RE.match(line) is not None + if is_atx or is_setext: + count += 1 + if stop_after is not None and count > stop_after: + return count + previous_text = bool(line.strip()) and not is_atx and not is_setext + + if newline < 0: + break + line_start = newline + 1 + + return count + + def _chunk_plain_text(self, content: str, path: str, line_offset: int) -> list[FileChunk]: + """Use the default byte chunker while preserving original Markdown lines.""" + if not isinstance(self.default_chunker, DefaultFileChunker): + raise RuntimeError("DefaultFileChunker dependency is unavailable; call start() before chunk().") + chunks = self.default_chunker.chunk_content(content, path, parse_links=True) + if line_offset: + for chunk in chunks: + chunk.start_line += line_offset + chunk.end_line += line_offset + chunk.set_hash_id() + return chunks + @staticmethod def _parse_front_matter(text: str) -> tuple[FileFrontMatter, str, int]: """Parse YAML frontmatter, returning 1-based line offset for body AST lines. @@ -242,146 +306,126 @@ class MarkdownFileChunker(BaseFileChunker): _finalize(root) return root - # -- Recursive chunker ------------------------------------------------ + # -- Recursive subtree chunking -------------------------------------- def _chunk_node( self, node: MdNode, - before: str, - after: str, + ancestors: tuple[str, ...], path: str, renderer, ) -> list[FileChunk]: - """Try the whole subtree; on overflow split (leaf) or descend. - ``before``/``after`` are TOC fragments that bracket each emitted - chunk's content (chunk text = ``before + content + after``). - As we descend, the prefix grows with section headings already - passed and the suffix shrinks correspondingly. + """Greedily assemble child subtrees, recursing only on oversized ones. + + A fitting subtree is returned immediately. For an oversized container, + fitting children are appended directly to a local ``FileChunk`` cache; + only an oversized child calls ``_chunk_node`` recursively. A full cache + is finalized before assembly continues in document order. """ - if not node.text: - return [] + prefix = _toc_join(*ancestors) + heading = "" + subtree_text = node.text if node.kind == "section": - heading_line = f"{'#' * node.level} {node.heading or ''}" - before_self = _toc_join(before, heading_line) - else: - before_self = before - if len(node.text) <= self.chunk_chars: + heading = f"{'#' * node.level} {node.heading or ''}" + subtree_text = _toc_join(heading, node.text) + + if subtree_text and len(subtree_text) <= self.chunk_chars: return [ self._make_chunk( - before_self, - node.text, - after, + prefix, + subtree_text, + "", node.start_line, node.end_line, path, ), ] + if node.kind == "body": - return self._split_leaf(node, before, after, path, renderer) - after_inside = _toc_join(node.desc_toc, after) - sub_tocs = [_subtree_toc(c) for c in node.children if c.kind == "section"] - chunks: list[FileChunk] = [] - accumulated = before_self - sec_idx = 0 - run: list[MdNode] = [] - for c in node.children: - if c.kind == "section": - if run: - chunks.extend( - self._chunk_body_run( - run, - before_self, - after_inside, - path, - renderer, - ), - ) - run = [] - remaining = "\n\n".join(sub_tocs[sec_idx + 1 :]) - chunks.extend( - self._chunk_node( - c, - accumulated, - _toc_join(remaining, after), - path, - renderer, - ), - ) - accumulated = _toc_join(accumulated, sub_tocs[sec_idx]) - sec_idx += 1 - else: - run.append(c) - if run: - chunks.extend( - self._chunk_body_run( - run, - before_self, - after_inside, - path, - renderer, - ), - ) - return chunks + return self._split_leaf(node, prefix, "", path, renderer) - def _chunk_body_run( - self, - run: list[MdNode], - before: str, - after: str, - path: str, - renderer, - ) -> list[FileChunk]: - """Greedy-pack consecutive body siblings under the same TOC slot. - No ``[Part X/N]`` markers — distinct blocks, not a leaf split. - Oversized single body recurses to ``_split_leaf``.""" - composite_size = sum(len(b.text) for b in run) + 2 * max(0, len(run) - 1) - if composite_size <= self.chunk_chars: - return [ - self._make_chunk( - before, - "\n\n".join(b.text for b in run), - after, - run[0].start_line, - run[-1].end_line, - path, - ), - ] + child_ancestors = ancestors + if node.kind == "section": + child_ancestors = (*ancestors, heading) chunks: list[FileChunk] = [] - bucket: list[MdNode] = [] - bucket_chars = 0 + cache: FileChunk | None = None + child_prefix = _toc_join(*child_ancestors) - def flush() -> None: - nonlocal bucket, bucket_chars - if not bucket: + def flush_cache() -> None: + nonlocal cache + if cache is None: return - chunks.append( - self._make_chunk( - before, - "\n\n".join(b.text for b in bucket), - after, - bucket[0].start_line, - bucket[-1].end_line, - path, - ), - ) - bucket = [] - bucket_chars = 0 + chunks.append(cache.set_hash_id()) + cache = None - for body in run: - if len(body.text) > self.chunk_chars: - flush() - chunks.extend(self._split_leaf(body, before, after, path, renderer)) + def append_to_cache(text: str, start_line: int, end_line: int, new_prefix: str) -> None: + nonlocal cache + if not text: + return + if cache is not None: + candidate = _toc_join(cache.text, text) + if len(candidate) <= self.chunk_chars: + cache.text = candidate + cache.end_line = end_line + if len(candidate) == self.chunk_chars: + flush_cache() + return + flush_cache() + + cache_text = _toc_join(new_prefix, text) if self.embed_toc else text + cache = FileChunk( + path=path, + start_line=start_line, + end_line=end_line, + text=cache_text, + ) + if len(cache_text) >= self.chunk_chars: + flush_cache() + + for child in node.children: + child_heading = f"{'#' * child.level} {child.heading or ''}" if child.kind == "section" else "" + child_text = _toc_join(child_heading, child.text) if child_heading else child.text + if len(child_text) <= self.chunk_chars: + append_to_cache(child_text, child.start_line, child.end_line, child_prefix) continue - sep = 2 if bucket else 0 - if bucket_chars + sep + len(body.text) > self.chunk_chars: - flush() - sep = 0 - bucket.append(body) - bucket_chars += sep + len(body.text) - flush() + + flush_cache() + chunks.extend(self._chunk_node(child, child_ancestors, path, renderer)) + + flush_cache() + if node.kind == "section": + if not chunks: + return [ + self._make_chunk( + prefix, + heading, + "", + node.start_line, + node.start_line, + path, + ), + ] + first_content = self._without_prefix(chunks[0].text, child_prefix) + chunks[0] = self._make_chunk( + prefix, + _toc_join(heading, first_content), + "", + node.start_line, + chunks[0].end_line, + path, + ) return chunks + def _without_prefix(self, text: str, prefix: str) -> str: + """Remove a breadcrumb that this chunker prepended to ``text``.""" + if not self.embed_toc or not prefix: + return text + if text == prefix: + return "" + marker = f"{prefix}\n\n" + return text[len(marker) :] if text.startswith(marker) else text + # -- Leaf splitters: build (text, start, end) units, hand off to packer def _split_leaf( diff --git a/reme/config/default.yaml b/reme/config/default.yaml index 517c9d04..6ce0e287 100644 --- a/reme/config/default.yaml +++ b/reme/config/default.yaml @@ -705,6 +705,9 @@ components: markdown: backend: markdown supported_extensions: [ "md" ] + embed_toc: true + max_ast_sections: 100 + default_chunker: default include_frontmatter_in_metadata: false include_frontmatter_keys_in_metadata: [] # empty = all non-empty frontmatter keys json: diff --git a/reme/config/jinli_lme.yaml b/reme/config/jinli_lme.yaml index df7906aa..8f3708e0 100644 --- a/reme/config/jinli_lme.yaml +++ b/reme/config/jinli_lme.yaml @@ -397,6 +397,9 @@ components: markdown: backend: markdown supported_extensions: [ "md" ] + embed_toc: true + max_ast_sections: 100 + default_chunker: default default: backend: default supported_extensions: [ "json", "jsonl" ] diff --git a/tests/unit/test_config_parser.py b/tests/unit/test_config_parser.py index 9f1c39c1..7a6207ee 100644 --- a/tests/unit/test_config_parser.py +++ b/tests/unit/test_config_parser.py @@ -54,6 +54,9 @@ def test_default_config_keeps_frontmatter_chunk_metadata_opt_in(): cfg = _load_config("default.yaml") markdown = cfg["components"]["file_chunker"]["markdown"] + assert markdown["embed_toc"] is True + assert markdown["max_ast_sections"] == 100 + assert markdown["default_chunker"] == "default" assert markdown["include_frontmatter_in_metadata"] is False # Allow-list defaults to empty; combined with the False above, chunk metadata stays empty. assert markdown["include_frontmatter_keys_in_metadata"] == [] or markdown.get( diff --git a/tests/unit/test_markdown_file_chunker.py b/tests/unit/test_markdown_file_chunker.py index 7db22fd4..f60950f8 100644 --- a/tests/unit/test_markdown_file_chunker.py +++ b/tests/unit/test_markdown_file_chunker.py @@ -11,8 +11,11 @@ a markdown-to-FileNode transformer. import asyncio import os import tempfile +from unittest.mock import patch -from reme.components.file_chunker import MarkdownFileChunker +from reme.components import ApplicationContext +from reme.components.file_chunker import DefaultFileChunker, MarkdownFileChunker +from reme.enumeration import ComponentEnum class temp_chdir: @@ -183,6 +186,64 @@ def test_parse_small_body_one_chunk(): asyncio.run(run()) +def test_parse_small_children_are_cached_without_recursive_calls(): + """An oversized parent greedily caches fitting children without recursing into them.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + sections = "\n\n".join(f"# Section-{i}\n\n{'x' * 30}" for i in range(6)) + path = _write_md(tmp, "small-children.md", sections) + chunker = MarkdownFileChunker(chunk_chars=100, embed_toc=False) + with patch.object( + chunker, + "_chunk_node", + wraps=chunker._chunk_node, + ) as chunk_node: + _, chunks = await chunker.chunk(path) + + assert chunk_node.call_count == 1 + assert 1 < len(chunks) < 6 + assert all(len(chunk.text) <= chunker.chunk_chars for chunk in chunks) + + asyncio.run(run()) + + +def test_parse_child_cache_flushes_at_exact_limit(): + """A cache reaching ``chunk_chars`` is finalized before the next child.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + path = _write_md(tmp, "exact-cache.md", f"# A\n\n{'x' * 95}\n\n# B\n\ny") + chunker = MarkdownFileChunker(chunk_chars=100) + _, chunks = await chunker.chunk(path) + + assert len(chunks) == 2 + assert len(chunks[0].text) == chunker.chunk_chars + assert chunks[0].text.startswith("# A") + assert chunks[1].text == "# B\n\ny" + + asyncio.run(run()) + + +def test_parse_oversized_child_flushes_parent_cache(): + """Recursive child chunks do not merge across the parent cache boundary.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + leaves = "\n\n".join(f"## L{i}\n\n{'x' * 10}" for i in range(6)) + body = f"# A\n\na\n\n# Large\n\n{leaves}\n\n# C\n\nc" + path = _write_md(tmp, "recursive-boundary.md", body) + chunker = MarkdownFileChunker(chunk_chars=100, embed_toc=False) + _, chunks = await chunker.chunk(path) + + assert len(chunks) == 4 + assert chunks[0].text == "# A\n\na" + assert chunks[-2].text == f"## L5\n\n{'x' * 10}" + assert chunks[-1].text == "# C\n\nc" + + asyncio.run(run()) + + def test_parse_oversized_body_splits(): """A body exceeding chunk_chars triggers multiple chunks.""" @@ -312,6 +373,172 @@ def test_parse_embed_toc_prefixes_chunk_text(): asyncio.run(run()) +def test_parse_embed_toc_is_enabled_by_default(): + """The default adds bounded ancestor breadcrumbs without a full outline.""" + chunker = MarkdownFileChunker() + assert chunker.embed_toc is True + assert chunker.max_ast_sections == 100 + + +def test_count_sections_ignores_fenced_headings_and_supports_setext(): + """The AST preflight counts real headings without parsing fenced examples.""" + content = "# Real\n\n```markdown\n# Fake\nFake too\n---\n```\n\nSetext\n===\n" + assert MarkdownFileChunker._count_sections(content) == 2 + assert MarkdownFileChunker._count_sections(content, stop_after=1) == 2 + + +def test_parse_excessive_sections_uses_plain_text_without_ast(): + """Too many sections preserve Markdown metadata while bypassing mistletoe.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + app_context = ApplicationContext(workspace_dir=tmp) + default_chunker = DefaultFileChunker( + name="default", + app_context=app_context, + chunk_byte_size=160, + overlap_byte_size=20, + ) + app_context.components[ComponentEnum.FILE_CHUNKER] = {"default": default_chunker} + sections = "\n\n".join(f"# Section-{i}\n\n{'x' * 40} [[target.md]]" for i in range(4)) + body = f"---\nname: fallback\n---\n{sections}" + _write_md(tmp, "fallback.md", body) + path = os.path.join(tmp, "fallback.md") + chunker = MarkdownFileChunker( + app_context=app_context, + chunk_chars=100, + embed_toc=True, + max_ast_sections=2, + include_frontmatter_in_metadata=True, + ) + await default_chunker.start() + await chunker.start() + try: + with ( + patch( + "mistletoe.block_token.Document", + side_effect=AssertionError("fallback must not construct an AST"), + ), + patch.object( + default_chunker, + "chunk_content", + wraps=default_chunker.chunk_content, + ) as chunk_content, + ): + node, chunks = await chunker.chunk(path) + assert chunker.default_chunker is default_chunker + assert chunk_content.call_count == 1 + finally: + await chunker.close() + await default_chunker.close() + + assert len(chunks) > 1 + assert chunks[0].start_line == 4 + assert node.front_matter.name == "fallback" + assert node.chunk_ids == [chunk.id for chunk in chunks] + assert {(link.target_path, link.source_path) for link in node.links} == { + ("target.md", "fallback.md"), + } + assert all(chunk.metadata == {"name": "fallback"} for chunk in chunks) + for i in range(4): + assert any(f"# Section-{i}" in chunk.text for chunk in chunks) + + asyncio.run(run()) + + +def test_parse_section_limit_is_inclusive_for_ast(): + """A document at the configured section limit still takes the AST path.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + path = _write_md(tmp, "at-limit.md", "# A\n\na\n\n# B\n\nb") + chunker = MarkdownFileChunker(max_ast_sections=2) + with patch.object(chunker, "_build_tree", wraps=chunker._build_tree) as build_tree: + _, chunks = await chunker.chunk(path) + + assert build_tree.call_count == 1 + assert len(chunks) == 1 + + asyncio.run(run()) + + +def test_parse_small_sections_are_merged_and_headings_preserved(): + """Adjacent small sections share chunks without losing their headings.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + sections = "\n\n".join(f"## Section-{i:03d}\n\nfact-{i:03d}" for i in range(40)) + path = _write_md(tmp, "sections.md", f"# Root\n\n{sections}") + chunker = MarkdownFileChunker(chunk_chars=200, embed_toc=False) + _, chunks = await chunker.chunk(path) + + assert 1 < len(chunks) < 40 + for i in range(40): + heading = f"## Section-{i:03d}" + assert sum(chunk.text.count(heading) for chunk in chunks) == 1 + + asyncio.run(run()) + + +def test_parse_embed_toc_uses_breadcrumbs_without_sibling_duplication(): + """TOC context repeats ancestors, not parallel section headings.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + sections = "\n\n".join(f"## Parallel-{i:03d}\n\nfact-{i:03d}" for i in range(40)) + path = _write_md(tmp, "breadcrumbs.md", f"# Root\n\n{sections}") + chunker = MarkdownFileChunker(chunk_chars=200, embed_toc=True) + _, chunks = await chunker.chunk(path) + + assert len(chunks) > 1 + assert all("# Root" in chunk.text for chunk in chunks) + for i in range(40): + heading = f"## Parallel-{i:03d}" + assert sum(chunk.text.count(heading) for chunk in chunks) == 1 + + asyncio.run(run()) + + +def test_parse_heading_heavy_output_grows_linearly(): + """A thousand parallel sections remain a small, linear number of chunks.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + sections = "\n\n".join(f"## Observation-{i:04d}\n\nsynthetic fact" for i in range(1000)) + body = f"# Root\n\n{sections}" + path = _write_md(tmp, "linear.md", body) + chunker = MarkdownFileChunker(chunk_chars=10000, embed_toc=True, max_ast_sections=None) + _, chunks = await chunker.chunk(path) + + assert len(chunks) < 10 + assert sum(len(chunk.text) for chunk in chunks) < 2 * len(body) + + asyncio.run(run()) + + +def test_parse_nested_sections_merge_within_recursive_context(): + """Nested continuations retain their branch breadcrumb after recursive merging.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + branches = [] + for branch, fact in (("A", "a"), ("B", "b")): + leaves = "\n\n".join(f"### {branch}{i}\n\n{fact * 45}" for i in range(4)) + branches.append(f"## {branch}\n\n{leaves}") + branch_text = "\n\n".join(branches) + path = _write_md(tmp, "nested.md", f"# Root\n\n{branch_text}") + chunker = MarkdownFileChunker(chunk_chars=150, embed_toc=True) + _, chunks = await chunker.chunk(path) + + assert len(chunks) == 4 + assert all(len(chunk.text) <= chunker.chunk_chars for chunk in chunks) + assert chunks[1].text.startswith("# Root\n\n## A\n\n### A2") + assert chunks[2].text.startswith("# Root\n\n## B\n\n### B0") + assert "## B" not in chunks[1].text + + asyncio.run(run()) + + def test_parse_frontmatter_preserves_original_line_numbers(): """Chunk line ranges are 1-based and refer to the original file, including frontmatter.""" @@ -371,6 +598,9 @@ if __name__ == "__main__": test_parse_frontmatter_only() test_parse_frontmatter_metadata_is_opt_in() test_parse_small_body_one_chunk() + test_parse_small_children_are_cached_without_recursive_calls() + test_parse_child_cache_flushes_at_exact_limit() + test_parse_oversized_child_flushes_parent_cache() test_parse_oversized_body_splits() test_parse_chunk_ids_match_node_chunk_ids() test_parse_links_literal_targets() @@ -379,6 +609,14 @@ if __name__ == "__main__": test_parse_links_deduped() test_parse_min_chunk_chars_clamped() test_parse_embed_toc_prefixes_chunk_text() + test_parse_embed_toc_is_enabled_by_default() + test_count_sections_ignores_fenced_headings_and_supports_setext() + test_parse_excessive_sections_uses_plain_text_without_ast() + test_parse_section_limit_is_inclusive_for_ast() + test_parse_small_sections_are_merged_and_headings_preserved() + test_parse_embed_toc_uses_breadcrumbs_without_sibling_duplication() + test_parse_heading_heavy_output_grows_linearly() + test_parse_nested_sections_merge_within_recursive_context() test_parse_frontmatter_preserves_original_line_numbers() test_parse_frontmatter_offsets_split_table_rows() test_parse_bad_frontmatter_does_not_abort_chunking()