mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-21 00:22:45 +00:00
refactor(file_parser): simplify chunk ID generation in LinkedFileParser
Remove path and line numbers from chunk ID calculation to improve cache stability. The chunk identity now uses only the content hash instead of including path::start::end coordinates. This change removes the documentation about chunk identity format and updates the hash_text call to use only the text content.
This commit is contained in:
parent
81e8f417b9
commit
48137f7f3f
1 changed files with 1 additions and 6 deletions
|
|
@ -9,11 +9,6 @@ heading level) → recursive chunk (try whole subtree; on overflow walk
|
|||
children — body siblings pack as a run, subsections recurse). Leaf
|
||||
blocks (table / code / list / paragraph) split on internal boundaries
|
||||
and each piece is annotated ``[Part X/N]``.
|
||||
|
||||
``chunk_chars`` constrains content only; the TOC skeleton is additive.
|
||||
Set ``embed_toc=False`` to drop the wrap. Chunk identity is
|
||||
``hash_text(path::start::end::text)`` for cache stability across
|
||||
re-parses of unchanged sections.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
|
@ -437,7 +432,7 @@ class LinkedFileParser(BaseFileParser):
|
|||
when ``embed_toc``, otherwise just ``content``."""
|
||||
text = _toc_join(before, content, after) if self.embed_toc else content
|
||||
return FileChunk(
|
||||
id=hash_text(f"{path}::{start_line}::{end_line}::{text}"),
|
||||
id=hash_text(text),
|
||||
path=path,
|
||||
start_line=start_line,
|
||||
end_line=end_line,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue