mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-17 23:51:19 +00:00
- Remove abstract methods from base component start/close - Update BaseJob to remove name parameter and simplify initialization - Change file modification time field from mtime_ms to modified_time in seconds - Add type checking imports and improve typing annotations - Implement LocalFileStore with JSONL persistence for file chunks - Add MdFileParser with markdown and frontmatter support - Simplify HttpClient call method with proper kwargs handling - Remove unused ReMe class methods and create backup version - Update StreamJob to use step_components instead of steps attribute
64 lines
1.9 KiB
Python
64 lines
1.9 KiB
Python
"""Default file parser for unknown file types."""
|
|
|
|
import asyncio
|
|
import hashlib
|
|
from pathlib import Path
|
|
|
|
from .base_file_parser import BaseFileParser
|
|
from ..component_registry import R
|
|
from ...schema import FileChunk, FileMetadata
|
|
from ...utils import hash_text, chunk_markdown
|
|
|
|
|
|
@R.register("default")
|
|
class DefaultFileParser(BaseFileParser):
|
|
"""Fallback parser for unknown file types.
|
|
|
|
Attempts to read as text and chunk. If the file is binary,
|
|
stores metadata only with no chunks.
|
|
"""
|
|
|
|
suffixes = []
|
|
|
|
def __init__(self, encoding: str = "utf-8", **kwargs):
|
|
super().__init__(**kwargs)
|
|
self.encoding = encoding
|
|
|
|
async def parse(self, path: str) -> tuple[FileMetadata, list[FileChunk]]:
|
|
file_path = Path(path)
|
|
|
|
def _read_file():
|
|
stat = file_path.stat()
|
|
raw = file_path.read_bytes()
|
|
try:
|
|
content = raw.decode(self.encoding)
|
|
content_hash = hash_text(content)
|
|
return stat, content_hash, content
|
|
except (UnicodeDecodeError, ValueError):
|
|
binary_hash = hashlib.sha256(raw).hexdigest()
|
|
return stat, binary_hash, None
|
|
|
|
stat, file_hash, content = await asyncio.to_thread(_read_file)
|
|
|
|
file_meta = FileMetadata(
|
|
hash=file_hash,
|
|
modified_time=stat.st_mtime,
|
|
size=stat.st_size,
|
|
path=str(file_path.absolute()),
|
|
content=content,
|
|
)
|
|
|
|
chunks: list[FileChunk] = []
|
|
if content:
|
|
chunks = (
|
|
chunk_markdown(
|
|
content,
|
|
file_meta.path,
|
|
self.chunk_tokens,
|
|
self.chunk_overlap,
|
|
)
|
|
or []
|
|
)
|
|
|
|
file_meta.chunk_count = len(chunks)
|
|
return file_meta, chunks
|