mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-10-11 03:40:03 +00:00
fix: bound markdown chunking for large section trees (#369)
This commit is contained in:
parent
9c9b040d42
commit
987f275985
6 changed files with 441 additions and 150 deletions
|
|
@ -59,7 +59,7 @@ class DefaultFileChunker(BaseFileChunker):
|
|||
content = text
|
||||
links = []
|
||||
|
||||
chunks = self._chunk_content(content, rel_path, parse_links=is_markdown)
|
||||
chunks = self.chunk_content(content, rel_path, parse_links=is_markdown)
|
||||
chunk_ids = [c.id for c in chunks]
|
||||
return (
|
||||
FileNode(
|
||||
|
|
@ -97,7 +97,7 @@ class DefaultFileChunker(BaseFileChunker):
|
|||
s, e = spans[idx]
|
||||
return (s, e) if s < pos < e else None
|
||||
|
||||
def _chunk_content(self, content: str, rel_path: str, parse_links: bool = True) -> list[FileChunk]:
|
||||
def chunk_content(self, content: str, rel_path: str, parse_links: bool = True) -> list[FileChunk]:
|
||||
"""Split content into overlapping byte-range chunks, avoiding cuts inside wikilinks.
|
||||
|
||||
When ``parse_links`` is False, skip wikilink span computation and boundary checks
|
||||
|
|
|
|||
|
|
@ -1,19 +1,22 @@
|
|||
"""Markdown file chunker — frontmatter + wikilink graph + AST tree chunks.
|
||||
|
||||
Each chunk carries the **complete heading skeleton** of the document
|
||||
with its content inlined under the section that owns it; other sections
|
||||
appear as bare headings so the reader always sees a full document map.
|
||||
When ``embed_toc`` is enabled, a chunk that starts inside a section carries
|
||||
only that section's ancestor heading breadcrumb. Sibling headings remain in
|
||||
the document stream and are stored once, avoiding the quadratic growth caused
|
||||
by repeating the complete document outline in every chunk.
|
||||
|
||||
Pipeline: build mistletoe AST → ``MdNode`` tree (sections nest by
|
||||
heading level) → recursive chunk (try whole subtree; on overflow walk
|
||||
children — body siblings pack as a run, subsections recurse). Leaf
|
||||
blocks (table / code / list / paragraph) split on internal boundaries
|
||||
and each piece is annotated ``[Part X/N]``. Wikilink extraction is
|
||||
Pipeline: count headings without an AST → use plain-text byte chunks when the
|
||||
configured section limit is exceeded; otherwise build a mistletoe AST →
|
||||
``MdNode`` tree (sections nest by heading level) → recursively chunk children
|
||||
and merge adjacent small subtrees at their parent. Leaf blocks (table / code /
|
||||
list / paragraph) split on internal boundaries and each piece is annotated
|
||||
``[Part X/N]``. Wikilink extraction is
|
||||
delegated to :class:`reme.utils.wikilink_handler.WikilinkHandler` —
|
||||
the single source of truth for ``[[...]]`` syntax (including
|
||||
Dataview-style typed predicates).
|
||||
"""
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
|
@ -23,6 +26,7 @@ from pydantic import ValidationError
|
|||
|
||||
|
||||
from .base_file_chunker import BaseFileChunker
|
||||
from .default_file_chunker import DefaultFileChunker
|
||||
from ..component_registry import R
|
||||
from ...schema import (
|
||||
FileChunk,
|
||||
|
|
@ -33,6 +37,10 @@ from ...utils.wikilink_handler import WikilinkHandler
|
|||
|
||||
# -- AST node + helpers ---------------------------------------------------
|
||||
|
||||
_ATX_HEADING_RE = re.compile(r"^ {0,3}#{1,6}(?:[ \t]+|$)")
|
||||
_SETEXT_HEADING_RE = re.compile(r"^ {0,3}(?:=+|-+)[ \t]*$")
|
||||
_FENCE_RE = re.compile(r"^ {0,3}(`{3,}|~{3,})")
|
||||
|
||||
|
||||
@dataclass
|
||||
class MdNode:
|
||||
|
|
@ -40,9 +48,7 @@ class MdNode:
|
|||
heading) / ``body`` (one mistletoe block; ``block`` keeps the original).
|
||||
|
||||
``text`` is the rendered subtree (own heading excluded for sections).
|
||||
``desc_toc`` caches the section-only DFS outline of descendants
|
||||
(own heading excluded), used as the TOC suffix when emitting chunks
|
||||
inside a section. Line ranges span the full subtree.
|
||||
Line ranges span the full subtree.
|
||||
"""
|
||||
|
||||
kind: str # "root" | "section" | "body"
|
||||
|
|
@ -53,7 +59,6 @@ class MdNode:
|
|||
text: str = ""
|
||||
start_line: int = 0
|
||||
end_line: int = 0
|
||||
desc_toc: str = ""
|
||||
|
||||
|
||||
def _heading_text(node: Any, renderer) -> str:
|
||||
|
|
@ -66,16 +71,13 @@ def _heading_text(node: Any, renderer) -> str:
|
|||
|
||||
def _finalize(n: MdNode) -> None:
|
||||
"""Bottom-up pass: propagate line ranges, populate ``n.text`` (rendered
|
||||
subtree, own heading excluded for sections) and ``n.desc_toc`` (DFS
|
||||
section outline of descendants)."""
|
||||
subtree, own heading excluded for sections)."""
|
||||
parts: list[str] = []
|
||||
desc_lines: list[str] = []
|
||||
for c in n.children:
|
||||
_finalize(c)
|
||||
if c.kind == "section":
|
||||
heading = f"{'#' * c.level} {c.heading or ''}"
|
||||
parts.append(f"{heading}\n\n{c.text}" if c.text else heading)
|
||||
desc_lines.append(f"{heading}\n\n{c.desc_toc}" if c.desc_toc else heading)
|
||||
elif c.text:
|
||||
parts.append(c.text)
|
||||
if n.children:
|
||||
|
|
@ -86,7 +88,6 @@ def _finalize(n: MdNode) -> None:
|
|||
n.end_line = n.start_line
|
||||
if n.kind != "body":
|
||||
n.text = "\n\n".join(parts)
|
||||
n.desc_toc = "\n\n".join(desc_lines)
|
||||
|
||||
|
||||
def _toc_join(*parts: str) -> str:
|
||||
|
|
@ -94,27 +95,20 @@ def _toc_join(*parts: str) -> str:
|
|||
return "\n\n".join(p for p in parts if p)
|
||||
|
||||
|
||||
def _subtree_toc(n: MdNode) -> str:
|
||||
"""Section's heading + descendants TOC — its contribution to a parent's
|
||||
``desc_toc``. For root (no own heading) this is just ``desc_toc``."""
|
||||
if n.kind != "section" or n.heading is None:
|
||||
return n.desc_toc
|
||||
heading = f"{'#' * n.level} {n.heading}"
|
||||
return f"{heading}\n\n{n.desc_toc}" if n.desc_toc else heading
|
||||
|
||||
|
||||
# -- Chunker --------------------------------------------------------------
|
||||
|
||||
|
||||
@R.register("markdown")
|
||||
class MarkdownFileChunker(BaseFileChunker):
|
||||
"""Markdown chunker: frontmatter + wikilink edges + full-skeleton chunks."""
|
||||
"""Markdown chunker with breadcrumb context and adjacent-section packing."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
encoding: str = "utf-8",
|
||||
chunk_chars: int = 10000,
|
||||
embed_toc: bool = True,
|
||||
max_ast_sections: int | None = 100,
|
||||
default_chunker: str = "default",
|
||||
include_frontmatter_in_metadata: bool = False,
|
||||
include_frontmatter_keys_in_metadata: list[str] | None = None,
|
||||
**kwargs,
|
||||
|
|
@ -123,22 +117,32 @@ class MarkdownFileChunker(BaseFileChunker):
|
|||
self.encoding = encoding
|
||||
self.chunk_chars = max(100, chunk_chars)
|
||||
self.embed_toc = embed_toc
|
||||
self.max_ast_sections = max(0, max_ast_sections) if max_ast_sections is not None else None
|
||||
self.default_chunker = self.bind(default_chunker, BaseFileChunker, optional=False)
|
||||
self.include_frontmatter_in_metadata = include_frontmatter_in_metadata
|
||||
self.include_frontmatter_keys_in_metadata = list(include_frontmatter_keys_in_metadata or [])
|
||||
|
||||
async def chunk(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]:
|
||||
from mistletoe.markdown_renderer import MarkdownRenderer
|
||||
from mistletoe.block_token import Document
|
||||
|
||||
file_path = Path(path)
|
||||
rel_path = self.to_workspace_relative(path)
|
||||
front_matter, content, line_offset = self._parse_front_matter(file_path.read_text(encoding=self.encoding))
|
||||
|
||||
chunks: list[FileChunk] = []
|
||||
if content and content.strip():
|
||||
with MarkdownRenderer() as renderer:
|
||||
tree = self._build_tree(Document(content), renderer, line_offset=line_offset)
|
||||
chunks = self._chunk_node(tree, "", "", rel_path, renderer)
|
||||
section_count = self._count_sections(content, stop_after=self.max_ast_sections)
|
||||
if self.max_ast_sections is not None and section_count > self.max_ast_sections:
|
||||
self.logger.info(
|
||||
f"Markdown AST skipped for {rel_path}: sections>{self.max_ast_sections}; "
|
||||
"using plain-text chunking",
|
||||
)
|
||||
chunks = self._chunk_plain_text(content, rel_path, line_offset)
|
||||
else:
|
||||
from mistletoe.markdown_renderer import MarkdownRenderer
|
||||
from mistletoe.block_token import Document
|
||||
|
||||
with MarkdownRenderer() as renderer:
|
||||
tree = self._build_tree(Document(content), renderer, line_offset=line_offset)
|
||||
chunks = self._chunk_node(tree, (), rel_path, renderer)
|
||||
if self.include_frontmatter_in_metadata:
|
||||
chunk_metadata = self._chunk_metadata(
|
||||
front_matter,
|
||||
|
|
@ -158,6 +162,66 @@ class MarkdownFileChunker(BaseFileChunker):
|
|||
)
|
||||
return node, chunks
|
||||
|
||||
@staticmethod
|
||||
def _count_sections(content: str, stop_after: int | None = None) -> int:
|
||||
"""Count block-level ATX/Setext headings without building a Markdown AST.
|
||||
|
||||
Fenced code content is ignored. When ``stop_after`` is set, return as
|
||||
soon as the count exceeds it because the caller only needs to choose
|
||||
between AST and plain-text chunking.
|
||||
"""
|
||||
count = 0
|
||||
fence_char = ""
|
||||
fence_length = 0
|
||||
previous_text = False
|
||||
|
||||
line_start = 0
|
||||
while line_start <= len(content):
|
||||
newline = content.find("\n", line_start)
|
||||
if newline < 0:
|
||||
line = content[line_start:]
|
||||
else:
|
||||
line = content[line_start:newline]
|
||||
if line.endswith("\r"):
|
||||
line = line[:-1]
|
||||
|
||||
fence_match = _FENCE_RE.match(line)
|
||||
if fence_match:
|
||||
marker = fence_match.group(1)
|
||||
if not fence_char:
|
||||
fence_char = marker[0]
|
||||
fence_length = len(marker)
|
||||
elif marker[0] == fence_char and len(marker) >= fence_length and not line[fence_match.end() :].strip():
|
||||
fence_char = ""
|
||||
fence_length = 0
|
||||
previous_text = False
|
||||
elif not fence_char:
|
||||
is_atx = _ATX_HEADING_RE.match(line) is not None
|
||||
is_setext = previous_text and _SETEXT_HEADING_RE.match(line) is not None
|
||||
if is_atx or is_setext:
|
||||
count += 1
|
||||
if stop_after is not None and count > stop_after:
|
||||
return count
|
||||
previous_text = bool(line.strip()) and not is_atx and not is_setext
|
||||
|
||||
if newline < 0:
|
||||
break
|
||||
line_start = newline + 1
|
||||
|
||||
return count
|
||||
|
||||
def _chunk_plain_text(self, content: str, path: str, line_offset: int) -> list[FileChunk]:
|
||||
"""Use the default byte chunker while preserving original Markdown lines."""
|
||||
if not isinstance(self.default_chunker, DefaultFileChunker):
|
||||
raise RuntimeError("DefaultFileChunker dependency is unavailable; call start() before chunk().")
|
||||
chunks = self.default_chunker.chunk_content(content, path, parse_links=True)
|
||||
if line_offset:
|
||||
for chunk in chunks:
|
||||
chunk.start_line += line_offset
|
||||
chunk.end_line += line_offset
|
||||
chunk.set_hash_id()
|
||||
return chunks
|
||||
|
||||
@staticmethod
|
||||
def _parse_front_matter(text: str) -> tuple[FileFrontMatter, str, int]:
|
||||
"""Parse YAML frontmatter, returning 1-based line offset for body AST lines.
|
||||
|
|
@ -242,146 +306,126 @@ class MarkdownFileChunker(BaseFileChunker):
|
|||
_finalize(root)
|
||||
return root
|
||||
|
||||
# -- Recursive chunker ------------------------------------------------
|
||||
# -- Recursive subtree chunking --------------------------------------
|
||||
|
||||
def _chunk_node(
|
||||
self,
|
||||
node: MdNode,
|
||||
before: str,
|
||||
after: str,
|
||||
ancestors: tuple[str, ...],
|
||||
path: str,
|
||||
renderer,
|
||||
) -> list[FileChunk]:
|
||||
"""Try the whole subtree; on overflow split (leaf) or descend.
|
||||
``before``/``after`` are TOC fragments that bracket each emitted
|
||||
chunk's content (chunk text = ``before + content + after``).
|
||||
As we descend, the prefix grows with section headings already
|
||||
passed and the suffix shrinks correspondingly.
|
||||
"""Greedily assemble child subtrees, recursing only on oversized ones.
|
||||
|
||||
A fitting subtree is returned immediately. For an oversized container,
|
||||
fitting children are appended directly to a local ``FileChunk`` cache;
|
||||
only an oversized child calls ``_chunk_node`` recursively. A full cache
|
||||
is finalized before assembly continues in document order.
|
||||
"""
|
||||
if not node.text:
|
||||
return []
|
||||
prefix = _toc_join(*ancestors)
|
||||
heading = ""
|
||||
subtree_text = node.text
|
||||
if node.kind == "section":
|
||||
heading_line = f"{'#' * node.level} {node.heading or ''}"
|
||||
before_self = _toc_join(before, heading_line)
|
||||
else:
|
||||
before_self = before
|
||||
if len(node.text) <= self.chunk_chars:
|
||||
heading = f"{'#' * node.level} {node.heading or ''}"
|
||||
subtree_text = _toc_join(heading, node.text)
|
||||
|
||||
if subtree_text and len(subtree_text) <= self.chunk_chars:
|
||||
return [
|
||||
self._make_chunk(
|
||||
before_self,
|
||||
node.text,
|
||||
after,
|
||||
prefix,
|
||||
subtree_text,
|
||||
"",
|
||||
node.start_line,
|
||||
node.end_line,
|
||||
path,
|
||||
),
|
||||
]
|
||||
|
||||
if node.kind == "body":
|
||||
return self._split_leaf(node, before, after, path, renderer)
|
||||
after_inside = _toc_join(node.desc_toc, after)
|
||||
sub_tocs = [_subtree_toc(c) for c in node.children if c.kind == "section"]
|
||||
chunks: list[FileChunk] = []
|
||||
accumulated = before_self
|
||||
sec_idx = 0
|
||||
run: list[MdNode] = []
|
||||
for c in node.children:
|
||||
if c.kind == "section":
|
||||
if run:
|
||||
chunks.extend(
|
||||
self._chunk_body_run(
|
||||
run,
|
||||
before_self,
|
||||
after_inside,
|
||||
path,
|
||||
renderer,
|
||||
),
|
||||
)
|
||||
run = []
|
||||
remaining = "\n\n".join(sub_tocs[sec_idx + 1 :])
|
||||
chunks.extend(
|
||||
self._chunk_node(
|
||||
c,
|
||||
accumulated,
|
||||
_toc_join(remaining, after),
|
||||
path,
|
||||
renderer,
|
||||
),
|
||||
)
|
||||
accumulated = _toc_join(accumulated, sub_tocs[sec_idx])
|
||||
sec_idx += 1
|
||||
else:
|
||||
run.append(c)
|
||||
if run:
|
||||
chunks.extend(
|
||||
self._chunk_body_run(
|
||||
run,
|
||||
before_self,
|
||||
after_inside,
|
||||
path,
|
||||
renderer,
|
||||
),
|
||||
)
|
||||
return chunks
|
||||
return self._split_leaf(node, prefix, "", path, renderer)
|
||||
|
||||
def _chunk_body_run(
|
||||
self,
|
||||
run: list[MdNode],
|
||||
before: str,
|
||||
after: str,
|
||||
path: str,
|
||||
renderer,
|
||||
) -> list[FileChunk]:
|
||||
"""Greedy-pack consecutive body siblings under the same TOC slot.
|
||||
No ``[Part X/N]`` markers — distinct blocks, not a leaf split.
|
||||
Oversized single body recurses to ``_split_leaf``."""
|
||||
composite_size = sum(len(b.text) for b in run) + 2 * max(0, len(run) - 1)
|
||||
if composite_size <= self.chunk_chars:
|
||||
return [
|
||||
self._make_chunk(
|
||||
before,
|
||||
"\n\n".join(b.text for b in run),
|
||||
after,
|
||||
run[0].start_line,
|
||||
run[-1].end_line,
|
||||
path,
|
||||
),
|
||||
]
|
||||
child_ancestors = ancestors
|
||||
if node.kind == "section":
|
||||
child_ancestors = (*ancestors, heading)
|
||||
|
||||
chunks: list[FileChunk] = []
|
||||
bucket: list[MdNode] = []
|
||||
bucket_chars = 0
|
||||
cache: FileChunk | None = None
|
||||
child_prefix = _toc_join(*child_ancestors)
|
||||
|
||||
def flush() -> None:
|
||||
nonlocal bucket, bucket_chars
|
||||
if not bucket:
|
||||
def flush_cache() -> None:
|
||||
nonlocal cache
|
||||
if cache is None:
|
||||
return
|
||||
chunks.append(
|
||||
self._make_chunk(
|
||||
before,
|
||||
"\n\n".join(b.text for b in bucket),
|
||||
after,
|
||||
bucket[0].start_line,
|
||||
bucket[-1].end_line,
|
||||
path,
|
||||
),
|
||||
)
|
||||
bucket = []
|
||||
bucket_chars = 0
|
||||
chunks.append(cache.set_hash_id())
|
||||
cache = None
|
||||
|
||||
for body in run:
|
||||
if len(body.text) > self.chunk_chars:
|
||||
flush()
|
||||
chunks.extend(self._split_leaf(body, before, after, path, renderer))
|
||||
def append_to_cache(text: str, start_line: int, end_line: int, new_prefix: str) -> None:
|
||||
nonlocal cache
|
||||
if not text:
|
||||
return
|
||||
if cache is not None:
|
||||
candidate = _toc_join(cache.text, text)
|
||||
if len(candidate) <= self.chunk_chars:
|
||||
cache.text = candidate
|
||||
cache.end_line = end_line
|
||||
if len(candidate) == self.chunk_chars:
|
||||
flush_cache()
|
||||
return
|
||||
flush_cache()
|
||||
|
||||
cache_text = _toc_join(new_prefix, text) if self.embed_toc else text
|
||||
cache = FileChunk(
|
||||
path=path,
|
||||
start_line=start_line,
|
||||
end_line=end_line,
|
||||
text=cache_text,
|
||||
)
|
||||
if len(cache_text) >= self.chunk_chars:
|
||||
flush_cache()
|
||||
|
||||
for child in node.children:
|
||||
child_heading = f"{'#' * child.level} {child.heading or ''}" if child.kind == "section" else ""
|
||||
child_text = _toc_join(child_heading, child.text) if child_heading else child.text
|
||||
if len(child_text) <= self.chunk_chars:
|
||||
append_to_cache(child_text, child.start_line, child.end_line, child_prefix)
|
||||
continue
|
||||
sep = 2 if bucket else 0
|
||||
if bucket_chars + sep + len(body.text) > self.chunk_chars:
|
||||
flush()
|
||||
sep = 0
|
||||
bucket.append(body)
|
||||
bucket_chars += sep + len(body.text)
|
||||
flush()
|
||||
|
||||
flush_cache()
|
||||
chunks.extend(self._chunk_node(child, child_ancestors, path, renderer))
|
||||
|
||||
flush_cache()
|
||||
if node.kind == "section":
|
||||
if not chunks:
|
||||
return [
|
||||
self._make_chunk(
|
||||
prefix,
|
||||
heading,
|
||||
"",
|
||||
node.start_line,
|
||||
node.start_line,
|
||||
path,
|
||||
),
|
||||
]
|
||||
first_content = self._without_prefix(chunks[0].text, child_prefix)
|
||||
chunks[0] = self._make_chunk(
|
||||
prefix,
|
||||
_toc_join(heading, first_content),
|
||||
"",
|
||||
node.start_line,
|
||||
chunks[0].end_line,
|
||||
path,
|
||||
)
|
||||
return chunks
|
||||
|
||||
def _without_prefix(self, text: str, prefix: str) -> str:
|
||||
"""Remove a breadcrumb that this chunker prepended to ``text``."""
|
||||
if not self.embed_toc or not prefix:
|
||||
return text
|
||||
if text == prefix:
|
||||
return ""
|
||||
marker = f"{prefix}\n\n"
|
||||
return text[len(marker) :] if text.startswith(marker) else text
|
||||
|
||||
# -- Leaf splitters: build (text, start, end) units, hand off to packer
|
||||
|
||||
def _split_leaf(
|
||||
|
|
|
|||
|
|
@ -705,6 +705,9 @@ components:
|
|||
markdown:
|
||||
backend: markdown
|
||||
supported_extensions: [ "md" ]
|
||||
embed_toc: true
|
||||
max_ast_sections: 100
|
||||
default_chunker: default
|
||||
include_frontmatter_in_metadata: false
|
||||
include_frontmatter_keys_in_metadata: [] # empty = all non-empty frontmatter keys
|
||||
json:
|
||||
|
|
|
|||
|
|
@ -397,6 +397,9 @@ components:
|
|||
markdown:
|
||||
backend: markdown
|
||||
supported_extensions: [ "md" ]
|
||||
embed_toc: true
|
||||
max_ast_sections: 100
|
||||
default_chunker: default
|
||||
default:
|
||||
backend: default
|
||||
supported_extensions: [ "json", "jsonl" ]
|
||||
|
|
|
|||
|
|
@ -54,6 +54,9 @@ def test_default_config_keeps_frontmatter_chunk_metadata_opt_in():
|
|||
cfg = _load_config("default.yaml")
|
||||
|
||||
markdown = cfg["components"]["file_chunker"]["markdown"]
|
||||
assert markdown["embed_toc"] is True
|
||||
assert markdown["max_ast_sections"] == 100
|
||||
assert markdown["default_chunker"] == "default"
|
||||
assert markdown["include_frontmatter_in_metadata"] is False
|
||||
# Allow-list defaults to empty; combined with the False above, chunk metadata stays empty.
|
||||
assert markdown["include_frontmatter_keys_in_metadata"] == [] or markdown.get(
|
||||
|
|
|
|||
|
|
@ -11,8 +11,11 @@ a markdown-to-FileNode transformer.
|
|||
import asyncio
|
||||
import os
|
||||
import tempfile
|
||||
from unittest.mock import patch
|
||||
|
||||
from reme.components.file_chunker import MarkdownFileChunker
|
||||
from reme.components import ApplicationContext
|
||||
from reme.components.file_chunker import DefaultFileChunker, MarkdownFileChunker
|
||||
from reme.enumeration import ComponentEnum
|
||||
|
||||
|
||||
class temp_chdir:
|
||||
|
|
@ -183,6 +186,64 @@ def test_parse_small_body_one_chunk():
|
|||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_small_children_are_cached_without_recursive_calls():
|
||||
"""An oversized parent greedily caches fitting children without recursing into them."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
sections = "\n\n".join(f"# Section-{i}\n\n{'x' * 30}" for i in range(6))
|
||||
path = _write_md(tmp, "small-children.md", sections)
|
||||
chunker = MarkdownFileChunker(chunk_chars=100, embed_toc=False)
|
||||
with patch.object(
|
||||
chunker,
|
||||
"_chunk_node",
|
||||
wraps=chunker._chunk_node,
|
||||
) as chunk_node:
|
||||
_, chunks = await chunker.chunk(path)
|
||||
|
||||
assert chunk_node.call_count == 1
|
||||
assert 1 < len(chunks) < 6
|
||||
assert all(len(chunk.text) <= chunker.chunk_chars for chunk in chunks)
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_child_cache_flushes_at_exact_limit():
|
||||
"""A cache reaching ``chunk_chars`` is finalized before the next child."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
path = _write_md(tmp, "exact-cache.md", f"# A\n\n{'x' * 95}\n\n# B\n\ny")
|
||||
chunker = MarkdownFileChunker(chunk_chars=100)
|
||||
_, chunks = await chunker.chunk(path)
|
||||
|
||||
assert len(chunks) == 2
|
||||
assert len(chunks[0].text) == chunker.chunk_chars
|
||||
assert chunks[0].text.startswith("# A")
|
||||
assert chunks[1].text == "# B\n\ny"
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_oversized_child_flushes_parent_cache():
|
||||
"""Recursive child chunks do not merge across the parent cache boundary."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
leaves = "\n\n".join(f"## L{i}\n\n{'x' * 10}" for i in range(6))
|
||||
body = f"# A\n\na\n\n# Large\n\n{leaves}\n\n# C\n\nc"
|
||||
path = _write_md(tmp, "recursive-boundary.md", body)
|
||||
chunker = MarkdownFileChunker(chunk_chars=100, embed_toc=False)
|
||||
_, chunks = await chunker.chunk(path)
|
||||
|
||||
assert len(chunks) == 4
|
||||
assert chunks[0].text == "# A\n\na"
|
||||
assert chunks[-2].text == f"## L5\n\n{'x' * 10}"
|
||||
assert chunks[-1].text == "# C\n\nc"
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_oversized_body_splits():
|
||||
"""A body exceeding chunk_chars triggers multiple chunks."""
|
||||
|
||||
|
|
@ -312,6 +373,172 @@ def test_parse_embed_toc_prefixes_chunk_text():
|
|||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_embed_toc_is_enabled_by_default():
|
||||
"""The default adds bounded ancestor breadcrumbs without a full outline."""
|
||||
chunker = MarkdownFileChunker()
|
||||
assert chunker.embed_toc is True
|
||||
assert chunker.max_ast_sections == 100
|
||||
|
||||
|
||||
def test_count_sections_ignores_fenced_headings_and_supports_setext():
|
||||
"""The AST preflight counts real headings without parsing fenced examples."""
|
||||
content = "# Real\n\n```markdown\n# Fake\nFake too\n---\n```\n\nSetext\n===\n"
|
||||
assert MarkdownFileChunker._count_sections(content) == 2
|
||||
assert MarkdownFileChunker._count_sections(content, stop_after=1) == 2
|
||||
|
||||
|
||||
def test_parse_excessive_sections_uses_plain_text_without_ast():
|
||||
"""Too many sections preserve Markdown metadata while bypassing mistletoe."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
app_context = ApplicationContext(workspace_dir=tmp)
|
||||
default_chunker = DefaultFileChunker(
|
||||
name="default",
|
||||
app_context=app_context,
|
||||
chunk_byte_size=160,
|
||||
overlap_byte_size=20,
|
||||
)
|
||||
app_context.components[ComponentEnum.FILE_CHUNKER] = {"default": default_chunker}
|
||||
sections = "\n\n".join(f"# Section-{i}\n\n{'x' * 40} [[target.md]]" for i in range(4))
|
||||
body = f"---\nname: fallback\n---\n{sections}"
|
||||
_write_md(tmp, "fallback.md", body)
|
||||
path = os.path.join(tmp, "fallback.md")
|
||||
chunker = MarkdownFileChunker(
|
||||
app_context=app_context,
|
||||
chunk_chars=100,
|
||||
embed_toc=True,
|
||||
max_ast_sections=2,
|
||||
include_frontmatter_in_metadata=True,
|
||||
)
|
||||
await default_chunker.start()
|
||||
await chunker.start()
|
||||
try:
|
||||
with (
|
||||
patch(
|
||||
"mistletoe.block_token.Document",
|
||||
side_effect=AssertionError("fallback must not construct an AST"),
|
||||
),
|
||||
patch.object(
|
||||
default_chunker,
|
||||
"chunk_content",
|
||||
wraps=default_chunker.chunk_content,
|
||||
) as chunk_content,
|
||||
):
|
||||
node, chunks = await chunker.chunk(path)
|
||||
assert chunker.default_chunker is default_chunker
|
||||
assert chunk_content.call_count == 1
|
||||
finally:
|
||||
await chunker.close()
|
||||
await default_chunker.close()
|
||||
|
||||
assert len(chunks) > 1
|
||||
assert chunks[0].start_line == 4
|
||||
assert node.front_matter.name == "fallback"
|
||||
assert node.chunk_ids == [chunk.id for chunk in chunks]
|
||||
assert {(link.target_path, link.source_path) for link in node.links} == {
|
||||
("target.md", "fallback.md"),
|
||||
}
|
||||
assert all(chunk.metadata == {"name": "fallback"} for chunk in chunks)
|
||||
for i in range(4):
|
||||
assert any(f"# Section-{i}" in chunk.text for chunk in chunks)
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_section_limit_is_inclusive_for_ast():
|
||||
"""A document at the configured section limit still takes the AST path."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
path = _write_md(tmp, "at-limit.md", "# A\n\na\n\n# B\n\nb")
|
||||
chunker = MarkdownFileChunker(max_ast_sections=2)
|
||||
with patch.object(chunker, "_build_tree", wraps=chunker._build_tree) as build_tree:
|
||||
_, chunks = await chunker.chunk(path)
|
||||
|
||||
assert build_tree.call_count == 1
|
||||
assert len(chunks) == 1
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_small_sections_are_merged_and_headings_preserved():
|
||||
"""Adjacent small sections share chunks without losing their headings."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
sections = "\n\n".join(f"## Section-{i:03d}\n\nfact-{i:03d}" for i in range(40))
|
||||
path = _write_md(tmp, "sections.md", f"# Root\n\n{sections}")
|
||||
chunker = MarkdownFileChunker(chunk_chars=200, embed_toc=False)
|
||||
_, chunks = await chunker.chunk(path)
|
||||
|
||||
assert 1 < len(chunks) < 40
|
||||
for i in range(40):
|
||||
heading = f"## Section-{i:03d}"
|
||||
assert sum(chunk.text.count(heading) for chunk in chunks) == 1
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_embed_toc_uses_breadcrumbs_without_sibling_duplication():
|
||||
"""TOC context repeats ancestors, not parallel section headings."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
sections = "\n\n".join(f"## Parallel-{i:03d}\n\nfact-{i:03d}" for i in range(40))
|
||||
path = _write_md(tmp, "breadcrumbs.md", f"# Root\n\n{sections}")
|
||||
chunker = MarkdownFileChunker(chunk_chars=200, embed_toc=True)
|
||||
_, chunks = await chunker.chunk(path)
|
||||
|
||||
assert len(chunks) > 1
|
||||
assert all("# Root" in chunk.text for chunk in chunks)
|
||||
for i in range(40):
|
||||
heading = f"## Parallel-{i:03d}"
|
||||
assert sum(chunk.text.count(heading) for chunk in chunks) == 1
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_heading_heavy_output_grows_linearly():
|
||||
"""A thousand parallel sections remain a small, linear number of chunks."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
sections = "\n\n".join(f"## Observation-{i:04d}\n\nsynthetic fact" for i in range(1000))
|
||||
body = f"# Root\n\n{sections}"
|
||||
path = _write_md(tmp, "linear.md", body)
|
||||
chunker = MarkdownFileChunker(chunk_chars=10000, embed_toc=True, max_ast_sections=None)
|
||||
_, chunks = await chunker.chunk(path)
|
||||
|
||||
assert len(chunks) < 10
|
||||
assert sum(len(chunk.text) for chunk in chunks) < 2 * len(body)
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_nested_sections_merge_within_recursive_context():
|
||||
"""Nested continuations retain their branch breadcrumb after recursive merging."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
branches = []
|
||||
for branch, fact in (("A", "a"), ("B", "b")):
|
||||
leaves = "\n\n".join(f"### {branch}{i}\n\n{fact * 45}" for i in range(4))
|
||||
branches.append(f"## {branch}\n\n{leaves}")
|
||||
branch_text = "\n\n".join(branches)
|
||||
path = _write_md(tmp, "nested.md", f"# Root\n\n{branch_text}")
|
||||
chunker = MarkdownFileChunker(chunk_chars=150, embed_toc=True)
|
||||
_, chunks = await chunker.chunk(path)
|
||||
|
||||
assert len(chunks) == 4
|
||||
assert all(len(chunk.text) <= chunker.chunk_chars for chunk in chunks)
|
||||
assert chunks[1].text.startswith("# Root\n\n## A\n\n### A2")
|
||||
assert chunks[2].text.startswith("# Root\n\n## B\n\n### B0")
|
||||
assert "## B" not in chunks[1].text
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_frontmatter_preserves_original_line_numbers():
|
||||
"""Chunk line ranges are 1-based and refer to the original file, including frontmatter."""
|
||||
|
||||
|
|
@ -371,6 +598,9 @@ if __name__ == "__main__":
|
|||
test_parse_frontmatter_only()
|
||||
test_parse_frontmatter_metadata_is_opt_in()
|
||||
test_parse_small_body_one_chunk()
|
||||
test_parse_small_children_are_cached_without_recursive_calls()
|
||||
test_parse_child_cache_flushes_at_exact_limit()
|
||||
test_parse_oversized_child_flushes_parent_cache()
|
||||
test_parse_oversized_body_splits()
|
||||
test_parse_chunk_ids_match_node_chunk_ids()
|
||||
test_parse_links_literal_targets()
|
||||
|
|
@ -379,6 +609,14 @@ if __name__ == "__main__":
|
|||
test_parse_links_deduped()
|
||||
test_parse_min_chunk_chars_clamped()
|
||||
test_parse_embed_toc_prefixes_chunk_text()
|
||||
test_parse_embed_toc_is_enabled_by_default()
|
||||
test_count_sections_ignores_fenced_headings_and_supports_setext()
|
||||
test_parse_excessive_sections_uses_plain_text_without_ast()
|
||||
test_parse_section_limit_is_inclusive_for_ast()
|
||||
test_parse_small_sections_are_merged_and_headings_preserved()
|
||||
test_parse_embed_toc_uses_breadcrumbs_without_sibling_duplication()
|
||||
test_parse_heading_heavy_output_grows_linearly()
|
||||
test_parse_nested_sections_merge_within_recursive_context()
|
||||
test_parse_frontmatter_preserves_original_line_numbers()
|
||||
test_parse_frontmatter_offsets_split_table_rows()
|
||||
test_parse_bad_frontmatter_does_not_abort_chunking()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue