ReMe/reme2/component/file_parser/default_file_parser.py
jinli.yl 42a3343cb5 feat(core): add core components and application framework
- Introduce Application class for managing application lifecycle
- Add base component classes for LLM formatters and token counters
- Implement embedding model base with caching and batching support
- Create file watcher base with watchfiles integration
- Add job and step base components for workflow execution
- Update base component with async locks and improved lifecycle management
- Register new component types in component registry
- Add application context and runtime context for dependency injection
2026-04-23 16:25:31 +08:00

64 lines
1.9 KiB
Python

"""Default file parser for unknown file types."""
import asyncio
import hashlib
from pathlib import Path
from .base_file_parser import BaseFileParser
from ..component_registry import R
from ...schema import FileChunk, FileMetadata
from ...utils import hash_text, chunk_markdown
@R.register("default")
class DefaultFileParser(BaseFileParser):
"""Fallback parser for unknown file types.
Attempts to read as text and chunk. If the file is binary,
stores metadata only with no chunks.
"""
suffixes = []
def __init__(self, encoding: str = "utf-8", **kwargs):
super().__init__(**kwargs)
self.encoding = encoding
async def parse(self, path: str) -> tuple[FileMetadata, list[FileChunk]]:
file_path = Path(path)
def _read_file():
stat = file_path.stat()
raw = file_path.read_bytes()
try:
content = raw.decode(self.encoding)
content_hash = hash_text(content)
return stat, content_hash, content
except (UnicodeDecodeError, ValueError):
binary_hash = hashlib.sha256(raw).hexdigest()
return stat, binary_hash, None
stat, file_hash, content = await asyncio.to_thread(_read_file)
file_meta = FileMetadata(
hash=file_hash,
mtime_ms=stat.st_mtime * 1000,
size=stat.st_size,
path=str(file_path.absolute()),
content=content,
)
chunks: list[FileChunk] = []
if content:
chunks = (
chunk_markdown(
content,
file_meta.path,
self.chunk_tokens,
self.chunk_overlap,
)
or []
)
file_meta.chunk_count = len(chunks)
return file_meta, chunks