ReMe/reme2/component/file_parser/md_file_parser.py
jinli.yl 3fc3fd65e8
Some checks failed
Pre-commit / run (ubuntu-latest) (push) Has been cancelled
feat(parser): add base file parser and concrete implementations
- Introduce BaseFileParser abstract class with component registration
- Add MdFileParser implementation for markdown files with YAML frontmatter
- Create TextFileParser implementation with built-in chunking support
- Implement file suffix enumeration for parser type safety
- Add chunking logic with configurable token size and overlap
- Support text file parsing with error handling for encoding issues
- Include line number tracking and content hashing for file chunks
2026-04-24 21:00:31 +08:00

54 lines
1.6 KiB
Python

"""Markdown file parser."""
import asyncio
from pathlib import Path
import frontmatter
from .base_file_parser import BaseFileParser
from ..component_registry import R
from ...enumeration import FileSuffixEnum
from ...schema import FileChunk, FileMetadata
from ...utils import chunk_markdown
@R.register("md")
class MdFileParser(BaseFileParser):
"""Parser for Markdown files with YAML frontmatter support."""
suffixes = [FileSuffixEnum.MD, FileSuffixEnum.MARKDOWN]
def __init__(self, encoding: str = "utf-8", chunk_tokens: int = 400, chunk_overlap: int = 80, **kwargs):
super().__init__(**kwargs)
self.encoding = encoding
self.chunk_tokens = chunk_tokens
self.chunk_overlap = chunk_overlap
async def parse(self, path: str) -> tuple[FileMetadata, list[FileChunk]]:
file_path = Path(path)
def _read_and_parse():
raw = file_path.read_text(encoding=self.encoding)
post = frontmatter.loads(raw)
stat = file_path.stat()
return stat, dict(post.metadata), post.content
stat, metadata, content = await asyncio.to_thread(_read_and_parse)
file_meta = FileMetadata(
modified_time=stat.st_mtime,
path=str(file_path.absolute()),
metadata=metadata,
)
chunks = (
chunk_markdown(
content,
file_meta.path,
self.chunk_tokens,
self.chunk_overlap,
)
or []
)
return file_meta, chunks