mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-10-04 02:33:54 +00:00
Some checks are pending
Pre-commit / run (ubuntu-latest) (push) Waiting to run
- Implement BaseComponent with async lifecycle and context management - Add ApplicationContext for managing component initialization and registry - Create Application class for orchestrating job execution and lifecycle - Add AS LLM components with OpenAI chat model wrapper - Implement AS LLM formatter components with OpenAI formatter - Add client implementations including base, HTTP and ReMe clients - Create embedding model base class with caching and batching support - Implement file store base class with vector and full-text search - Add file watcher components for monitoring file system changes - Create job components for executing workflows - Implement service components for exposing jobs via different protocols - Add configuration schema with ApplicationConfig and ComponentConfig - Include utility modules for case conversion, chunking, logging and similarity - Register component types and create component registry system
53 lines
1.8 KiB
Python
53 lines
1.8 KiB
Python
"""File chunk schema module.
|
|
|
|
This module defines the FileChunk model for representing chunks of file content
|
|
in the document processing and retrieval pipeline.
|
|
"""
|
|
|
|
from pydantic import Field
|
|
|
|
from .base_node import BaseNode
|
|
|
|
|
|
class FileChunk(BaseNode):
|
|
"""A chunk of file content with positional and scoring metadata.
|
|
|
|
Represents a contiguous section of a file that has been extracted
|
|
for processing, embedding, or retrieval. Inherits text and embedding
|
|
capabilities from BaseNode.
|
|
|
|
Attributes:
|
|
path: File path relative to workspace root.
|
|
start_line: Starting line number (1-indexed) in the source file.
|
|
end_line: Ending line number (1-indexed) in the source file.
|
|
hash: Hash of the chunk content for deduplication.
|
|
scores: Search relevance scores indexed by score type.
|
|
|
|
Properties:
|
|
score: Final combined score for search result ranking.
|
|
merge_key: Unique key for merging duplicate search results.
|
|
"""
|
|
|
|
path: str = Field(..., description="File path relative to workspace")
|
|
start_line: int = Field(..., description="Starting line number (1-indexed)")
|
|
end_line: int = Field(..., description="Ending line number (1-indexed)")
|
|
hash: str = Field(..., description="Hash of chunk content for deduplication")
|
|
scores: dict[str, float] = Field(default_factory=dict, description="Search scores by type")
|
|
|
|
@property
|
|
def score(self) -> float:
|
|
"""Get the final score for search result ranking.
|
|
|
|
Returns:
|
|
The combined score, or 0.0 if not set.
|
|
"""
|
|
return self.scores.get("score", 0.0)
|
|
|
|
@property
|
|
def merge_key(self) -> str:
|
|
"""Generate a unique key for merging search results.
|
|
|
|
Returns:
|
|
A string key in format "path:start_line:end_line".
|
|
"""
|
|
return f"{self.path}:{self.start_line}:{self.end_line}"
|