diff --git a/docs/en/memory_as_file.md b/docs/en/memory_as_file.md index 5202887a..f519396a 100644 --- a/docs/en/memory_as_file.md +++ b/docs/en/memory_as_file.md @@ -388,3 +388,8 @@ This lets the agent see not only an isolated paragraph but also its structural p Non-Markdown files use `DefaultFileChunker` by default. It splits by byte size and preserves a small overlap. For Markdown, the chunker also avoids cutting `[[wikilinks]]` in the middle. + +`DefaultFileChunker` and `MarkdownFileChunker` decode files with their configured `encoding` and normalize platform +newlines to LF before indexing. Their default `invalid_encoding_policy: replace` keeps decodable content searchable +when a source contains invalid bytes, without modifying the source file. Set `invalid_encoding_policy: strict` on a +chunker component to reject such files instead. diff --git a/docs/zh/memory_as_file.md b/docs/zh/memory_as_file.md index f9532d42..6d792ed9 100644 --- a/docs/zh/memory_as_file.md +++ b/docs/zh/memory_as_file.md @@ -363,3 +363,7 @@ FileChunk[] 这样检索命中时,Agent 不只看到孤立段落,还能看到它在原文件中的结构位置。 非 Markdown 默认走 `DefaultFileChunker`:按字节大小切分,并保留少量 overlap;对 Markdown 则会避免把 `[[wikilink]]` 从中间切开。 + +`DefaultFileChunker` 和 `MarkdownFileChunker` 使用各自配置的 `encoding` 解码文件,并在索引前将平台换行符统一为 +LF。默认的 `invalid_encoding_policy: replace` 会在源文件含无效字节时保留其中可解码的内容用于检索,但不会修改源 +文件;如需拒绝此类文件,可在 chunker 组件上设置 `invalid_encoding_policy: strict`。 diff --git a/reme/components/file_chunker/default_file_chunker.py b/reme/components/file_chunker/default_file_chunker.py index acc545a8..4fddc632 100644 --- a/reme/components/file_chunker/default_file_chunker.py +++ b/reme/components/file_chunker/default_file_chunker.py @@ -22,9 +22,10 @@ class DefaultFileChunker(BaseFileChunker): def __init__( self, encoding: str = "utf-8", - invalid_encoding_policy: InvalidEncodingPolicy = "replace", chunk_byte_size: int = 10000, overlap_byte_size: int = 100, + *, + invalid_encoding_policy: InvalidEncodingPolicy = "replace", **kwargs, ): super().__init__(**kwargs) @@ -57,7 +58,7 @@ class DefaultFileChunker(BaseFileChunker): async with aiofiles.open(file_path, "rb") as f: data = await f.read() try: - return data.decode(self.encoding) + text = data.decode(self.encoding) except UnicodeDecodeError as exc: if self.invalid_encoding_policy == "strict": raise @@ -66,7 +67,14 @@ class DefaultFileChunker(BaseFileChunker): f"Invalid {self.encoding} in {file_path} at byte {exc.start} (bytes: {invalid_bytes}); " "indexed with replacement characters; source file unchanged", ) - return data.decode(self.encoding, errors="replace") + text = data.decode(self.encoding, errors="replace") + # Some codecs (for example ASCII) cannot encode U+FFFD. Convert the + # decoded fallback to that codec's own replacement representation so + # later byte-based chunking remains safe. + text = text.encode(self.encoding, errors="replace").decode(self.encoding) + + # Match the universal-newline behavior of the previous text-mode reads. + return text.replace("\r\n", "\n").replace("\r", "\n") async def chunk(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]: file_path = Path(path) diff --git a/reme/components/file_chunker/markdown_file_chunker.py b/reme/components/file_chunker/markdown_file_chunker.py index a01e3e82..06a07793 100644 --- a/reme/components/file_chunker/markdown_file_chunker.py +++ b/reme/components/file_chunker/markdown_file_chunker.py @@ -103,12 +103,13 @@ class MarkdownFileChunker(DefaultFileChunker): def __init__( self, encoding: str = "utf-8", - invalid_encoding_policy: InvalidEncodingPolicy = "replace", chunk_byte_size: int = 10000, embed_toc: bool = True, max_ast_sections: int | None = 100, include_frontmatter_in_metadata: bool = False, include_frontmatter_keys_in_metadata: list[str] | None = None, + *, + invalid_encoding_policy: InvalidEncodingPolicy = "replace", **kwargs, ): super().__init__( diff --git a/tests/unit/test_default_file_chunker.py b/tests/unit/test_default_file_chunker.py index d5bbdf23..aebd19ba 100644 --- a/tests/unit/test_default_file_chunker.py +++ b/tests/unit/test_default_file_chunker.py @@ -107,6 +107,59 @@ def test_invalid_encoding_policy_is_validated(): raise AssertionError("invalid encoding policy must be rejected") +def test_constructor_preserves_positional_arguments(): + """New decoding options must not reinterpret the established positional API.""" + chunker = DefaultFileChunker("utf-8", 5000, 100) + + assert chunker.encoding == "utf-8" + assert chunker.chunk_byte_size == 5000 + assert chunker.overlap_byte_size == 100 + + +def test_newlines_are_normalized_before_chunking(): + """Binary reads retain the universal-newline behavior of the old text reader.""" + + async def run(): + with tempfile.NamedTemporaryFile(delete=False, suffix=".txt") as f: + f.write(b"alpha\r\nbeta\rgamma\n") + temp_path = f.name + + try: + chunker = DefaultFileChunker() + _, original_chunks = await chunker.chunk(temp_path) + with open(temp_path, "wb") as f: + f.write(b"alpha\nbeta\ngamma\n") + _, normalized_chunks = await chunker.chunk(temp_path) + + assert [chunk.text for chunk in original_chunks] == ["alpha\nbeta\ngamma\n"] + assert [chunk.id for chunk in original_chunks] == [chunk.id for chunk in normalized_chunks] + finally: + os.unlink(temp_path) + + asyncio.run(run()) + + +def test_ascii_replacement_remains_encodable(): + """Replacement mode must survive byte-based chunking with a single-byte codec.""" + + async def run(): + source = b"valid\xffinvalid" + with tempfile.NamedTemporaryFile(delete=False, suffix=".txt") as f: + f.write(source) + temp_path = f.name + + try: + _, chunks = await DefaultFileChunker(encoding="ascii").chunk(temp_path) + + assert [chunk.text for chunk in chunks] == ["valid?invalid"] + with open(temp_path, "rb") as f: + assert f.read() == source + finally: + os.unlink(temp_path) + + asyncio.run(run()) + + def test_parse_links_bare(): """Bare wikilink: [[target]].""" links = WikilinkHandler.extract_links("see [[note]]", "src.md") diff --git a/tests/unit/test_markdown_file_chunker.py b/tests/unit/test_markdown_file_chunker.py index 54676607..fbeeffaf 100644 --- a/tests/unit/test_markdown_file_chunker.py +++ b/tests/unit/test_markdown_file_chunker.py @@ -102,6 +102,59 @@ def test_invalid_utf8_strict_policy_still_raises(): asyncio.run(run()) +def test_constructor_preserves_positional_arguments(): + """The decoding policy must not shift the established positional parameters.""" + chunker = MarkdownFileChunker("utf-8", 5000, False, 10, True, ["name"]) + + assert chunker.encoding == "utf-8" + assert chunker.chunk_byte_size == 5000 + assert chunker.embed_toc is False + assert chunker.max_ast_sections == 10 + assert chunker.include_frontmatter_in_metadata is True + assert chunker.include_frontmatter_keys_in_metadata == ["name"] + + +def test_plain_text_fallback_normalizes_newlines(): + """Markdown fallback chunks stay stable for equivalent platform newlines.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + path = os.path.join(tmp, "fallback.md") + with open(path, "wb") as f: + f.write(b"# One\r\nbody\r# Two\r\nbody\r\n") + + chunker = MarkdownFileChunker(max_ast_sections=0) + _, original_chunks = await chunker.chunk("fallback.md") + with open(path, "wb") as f: + f.write(b"# One\nbody\n# Two\nbody\n") + _, normalized_chunks = await chunker.chunk("fallback.md") + + assert [chunk.text for chunk in original_chunks] == [chunk.text for chunk in normalized_chunks] + assert [chunk.id for chunk in original_chunks] == [chunk.id for chunk in normalized_chunks] + + asyncio.run(run()) + + +def test_invalid_ascii_is_replaced_for_markdown(): + """Markdown byte accounting accepts the configured codec's replacement text.""" + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + source = b"# valid\ncontent before \xff content after\n" + path = os.path.join(tmp, "invalid.md") + with open(path, "wb") as f: + f.write(source) + + chunker = MarkdownFileChunker(encoding="ascii") + _, chunks = await chunker.chunk("invalid.md") + + assert "content before ? content after" in "\n".join(chunk.text for chunk in chunks) + with open(path, "rb") as f: + assert f.read() == source + + asyncio.run(run()) + + def test_parse_frontmatter_only(): """A file with only frontmatter (no body) → no chunks, no links."""