mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-19 00:01:33 +00:00
Some checks are pending
Pre-commit / run (ubuntu-latest) (push) Waiting to run
- Add file_parser component with default implementation - Introduce SearchFilter schema for path and tag filtering - Implement filter functionality in BaseFileStore and LocalFileStore - Update file watcher to use parser-based filtering instead of suffix filters - Register new FILE_PARSER component enum - Add test_data directory to gitignore refactor: improve component imports and initialization - Fix relative imports in application.py - Add file_parser import to component init - Initialize registry dict when component type doesn't exist - Remove circular import in HttpService by using string annotation - Update config yaml to use proper component names refactor: enhance file watcher architecture - Replace MdFileWatcher with more flexible FullFileWatcher and LightFileWatcher - Remove suffix-based filtering in favor of parser-based approach - Update BaseFileWatcher to resolve parsers from app context - Remove unused watch_filter method refactor: update ReMe core functionality - Remove memory_path creation - Simplify dream and proactive methods to return empty strings - Update config defaults for HTTP service and component backends docs: update component configuration in paw.yaml - Change service backend from cmd to http - Rename components to use correct singular forms - Add default file parser and file watcher configurations - Set up local file store with default settings ``` Co-authored-by: huangsen <huangsen.huang@alibaba-inc.com>
41 lines
1.1 KiB
Python
41 lines
1.1 KiB
Python
"""Abstract base class for file parsers."""
|
|
|
|
from abc import abstractmethod
|
|
|
|
from ..base_component import BaseComponent
|
|
from ...enumeration import ComponentEnum
|
|
from ...schema import FileChunk, FileMetadata
|
|
|
|
|
|
class BaseFileParser(BaseComponent):
|
|
"""Abstract base class for file format parsers.
|
|
|
|
Each parser declares which file suffixes it handles and implements
|
|
the parse method to produce FileMetadata and FileChunks.
|
|
"""
|
|
|
|
component_type = ComponentEnum.FILE_PARSER
|
|
|
|
suffixes: list[str] = []
|
|
|
|
def __init__(self, chunk_tokens: int = 400, chunk_overlap: int = 80, **kwargs):
|
|
super().__init__(**kwargs)
|
|
self.chunk_tokens = chunk_tokens
|
|
self.chunk_overlap = chunk_overlap
|
|
|
|
async def _start(self, app_context=None):
|
|
pass
|
|
|
|
async def _close(self):
|
|
pass
|
|
|
|
@abstractmethod
|
|
async def parse(self, path: str) -> tuple[FileMetadata, list[FileChunk]]:
|
|
"""Parse a file into metadata and chunks.
|
|
|
|
Args:
|
|
path: Absolute path to the file.
|
|
|
|
Returns:
|
|
Tuple of (FileMetadata, list of FileChunks).
|
|
"""
|