mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-10-10 03:30:56 +00:00
fix(index): preserve text chunker compatibility
This commit is contained in:
parent
d4bfa3aaea
commit
e08837326d
6 changed files with 128 additions and 4 deletions
|
|
@ -388,3 +388,8 @@ This lets the agent see not only an isolated paragraph but also its structural p
|
|||
|
||||
Non-Markdown files use `DefaultFileChunker` by default. It splits by byte size and preserves a small overlap. For
|
||||
Markdown, the chunker also avoids cutting `[[wikilinks]]` in the middle.
|
||||
|
||||
`DefaultFileChunker` and `MarkdownFileChunker` decode files with their configured `encoding` and normalize platform
|
||||
newlines to LF before indexing. Their default `invalid_encoding_policy: replace` keeps decodable content searchable
|
||||
when a source contains invalid bytes, without modifying the source file. Set `invalid_encoding_policy: strict` on a
|
||||
chunker component to reject such files instead.
|
||||
|
|
|
|||
|
|
@ -363,3 +363,7 @@ FileChunk[]
|
|||
这样检索命中时,Agent 不只看到孤立段落,还能看到它在原文件中的结构位置。
|
||||
|
||||
非 Markdown 默认走 `DefaultFileChunker`:按字节大小切分,并保留少量 overlap;对 Markdown 则会避免把 `[[wikilink]]` 从中间切开。
|
||||
|
||||
`DefaultFileChunker` 和 `MarkdownFileChunker` 使用各自配置的 `encoding` 解码文件,并在索引前将平台换行符统一为
|
||||
LF。默认的 `invalid_encoding_policy: replace` 会在源文件含无效字节时保留其中可解码的内容用于检索,但不会修改源
|
||||
文件;如需拒绝此类文件,可在 chunker 组件上设置 `invalid_encoding_policy: strict`。
|
||||
|
|
|
|||
|
|
@ -22,9 +22,10 @@ class DefaultFileChunker(BaseFileChunker):
|
|||
def __init__(
|
||||
self,
|
||||
encoding: str = "utf-8",
|
||||
invalid_encoding_policy: InvalidEncodingPolicy = "replace",
|
||||
chunk_byte_size: int = 10000,
|
||||
overlap_byte_size: int = 100,
|
||||
*,
|
||||
invalid_encoding_policy: InvalidEncodingPolicy = "replace",
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__(**kwargs)
|
||||
|
|
@ -57,7 +58,7 @@ class DefaultFileChunker(BaseFileChunker):
|
|||
async with aiofiles.open(file_path, "rb") as f:
|
||||
data = await f.read()
|
||||
try:
|
||||
return data.decode(self.encoding)
|
||||
text = data.decode(self.encoding)
|
||||
except UnicodeDecodeError as exc:
|
||||
if self.invalid_encoding_policy == "strict":
|
||||
raise
|
||||
|
|
@ -66,7 +67,14 @@ class DefaultFileChunker(BaseFileChunker):
|
|||
f"Invalid {self.encoding} in {file_path} at byte {exc.start} (bytes: {invalid_bytes}); "
|
||||
"indexed with replacement characters; source file unchanged",
|
||||
)
|
||||
return data.decode(self.encoding, errors="replace")
|
||||
text = data.decode(self.encoding, errors="replace")
|
||||
# Some codecs (for example ASCII) cannot encode U+FFFD. Convert the
|
||||
# decoded fallback to that codec's own replacement representation so
|
||||
# later byte-based chunking remains safe.
|
||||
text = text.encode(self.encoding, errors="replace").decode(self.encoding)
|
||||
|
||||
# Match the universal-newline behavior of the previous text-mode reads.
|
||||
return text.replace("\r\n", "\n").replace("\r", "\n")
|
||||
|
||||
async def chunk(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]:
|
||||
file_path = Path(path)
|
||||
|
|
|
|||
|
|
@ -103,12 +103,13 @@ class MarkdownFileChunker(DefaultFileChunker):
|
|||
def __init__(
|
||||
self,
|
||||
encoding: str = "utf-8",
|
||||
invalid_encoding_policy: InvalidEncodingPolicy = "replace",
|
||||
chunk_byte_size: int = 10000,
|
||||
embed_toc: bool = True,
|
||||
max_ast_sections: int | None = 100,
|
||||
include_frontmatter_in_metadata: bool = False,
|
||||
include_frontmatter_keys_in_metadata: list[str] | None = None,
|
||||
*,
|
||||
invalid_encoding_policy: InvalidEncodingPolicy = "replace",
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__(
|
||||
|
|
|
|||
|
|
@ -107,6 +107,59 @@ def test_invalid_encoding_policy_is_validated():
|
|||
raise AssertionError("invalid encoding policy must be rejected")
|
||||
|
||||
|
||||
def test_constructor_preserves_positional_arguments():
|
||||
"""New decoding options must not reinterpret the established positional API."""
|
||||
chunker = DefaultFileChunker("utf-8", 5000, 100)
|
||||
|
||||
assert chunker.encoding == "utf-8"
|
||||
assert chunker.chunk_byte_size == 5000
|
||||
assert chunker.overlap_byte_size == 100
|
||||
|
||||
|
||||
def test_newlines_are_normalized_before_chunking():
|
||||
"""Binary reads retain the universal-newline behavior of the old text reader."""
|
||||
|
||||
async def run():
|
||||
with tempfile.NamedTemporaryFile(delete=False, suffix=".txt") as f:
|
||||
f.write(b"alpha\r\nbeta\rgamma\n")
|
||||
temp_path = f.name
|
||||
|
||||
try:
|
||||
chunker = DefaultFileChunker()
|
||||
_, original_chunks = await chunker.chunk(temp_path)
|
||||
with open(temp_path, "wb") as f:
|
||||
f.write(b"alpha\nbeta\ngamma\n")
|
||||
_, normalized_chunks = await chunker.chunk(temp_path)
|
||||
|
||||
assert [chunk.text for chunk in original_chunks] == ["alpha\nbeta\ngamma\n"]
|
||||
assert [chunk.id for chunk in original_chunks] == [chunk.id for chunk in normalized_chunks]
|
||||
finally:
|
||||
os.unlink(temp_path)
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_ascii_replacement_remains_encodable():
|
||||
"""Replacement mode must survive byte-based chunking with a single-byte codec."""
|
||||
|
||||
async def run():
|
||||
source = b"valid\xffinvalid"
|
||||
with tempfile.NamedTemporaryFile(delete=False, suffix=".txt") as f:
|
||||
f.write(source)
|
||||
temp_path = f.name
|
||||
|
||||
try:
|
||||
_, chunks = await DefaultFileChunker(encoding="ascii").chunk(temp_path)
|
||||
|
||||
assert [chunk.text for chunk in chunks] == ["valid?invalid"]
|
||||
with open(temp_path, "rb") as f:
|
||||
assert f.read() == source
|
||||
finally:
|
||||
os.unlink(temp_path)
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_links_bare():
|
||||
"""Bare wikilink: [[target]]."""
|
||||
links = WikilinkHandler.extract_links("see [[note]]", "src.md")
|
||||
|
|
|
|||
|
|
@ -102,6 +102,59 @@ def test_invalid_utf8_strict_policy_still_raises():
|
|||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_constructor_preserves_positional_arguments():
|
||||
"""The decoding policy must not shift the established positional parameters."""
|
||||
chunker = MarkdownFileChunker("utf-8", 5000, False, 10, True, ["name"])
|
||||
|
||||
assert chunker.encoding == "utf-8"
|
||||
assert chunker.chunk_byte_size == 5000
|
||||
assert chunker.embed_toc is False
|
||||
assert chunker.max_ast_sections == 10
|
||||
assert chunker.include_frontmatter_in_metadata is True
|
||||
assert chunker.include_frontmatter_keys_in_metadata == ["name"]
|
||||
|
||||
|
||||
def test_plain_text_fallback_normalizes_newlines():
|
||||
"""Markdown fallback chunks stay stable for equivalent platform newlines."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
path = os.path.join(tmp, "fallback.md")
|
||||
with open(path, "wb") as f:
|
||||
f.write(b"# One\r\nbody\r# Two\r\nbody\r\n")
|
||||
|
||||
chunker = MarkdownFileChunker(max_ast_sections=0)
|
||||
_, original_chunks = await chunker.chunk("fallback.md")
|
||||
with open(path, "wb") as f:
|
||||
f.write(b"# One\nbody\n# Two\nbody\n")
|
||||
_, normalized_chunks = await chunker.chunk("fallback.md")
|
||||
|
||||
assert [chunk.text for chunk in original_chunks] == [chunk.text for chunk in normalized_chunks]
|
||||
assert [chunk.id for chunk in original_chunks] == [chunk.id for chunk in normalized_chunks]
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_invalid_ascii_is_replaced_for_markdown():
|
||||
"""Markdown byte accounting accepts the configured codec's replacement text."""
|
||||
|
||||
async def run():
|
||||
with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp):
|
||||
source = b"# valid\ncontent before \xff content after\n"
|
||||
path = os.path.join(tmp, "invalid.md")
|
||||
with open(path, "wb") as f:
|
||||
f.write(source)
|
||||
|
||||
chunker = MarkdownFileChunker(encoding="ascii")
|
||||
_, chunks = await chunker.chunk("invalid.md")
|
||||
|
||||
assert "content before ? content after" in "\n".join(chunk.text for chunk in chunks)
|
||||
with open(path, "rb") as f:
|
||||
assert f.read() == source
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_parse_frontmatter_only():
|
||||
"""A file with only frontmatter (no body) → no chunks, no links."""
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue