mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-10-07 03:00:27 +00:00
Some checks are pending
Pre-commit / run (ubuntu-latest) (push) Waiting to run
- Introduce BaseAsTokenCounter and EstimatedAsTokenCounter for token estimation - Add AsMsgStat and AsBlockStat schema for message statistics tracking - Implement FileIO class with read/write/append/edit operations - Create file utility functions for safe async file reading and truncation - Add MemorySearch component for semantic search in memory files - Register new component types in ComponentEnum and update imports - Add constants for default host, port, and truncation limits - Create BaseService abstract base class for service implementations - Implement BaseStep with component accessors and lifecycle management - Add proper __all__ exports for all new modules and components
53 lines
1.8 KiB
Python
53 lines
1.8 KiB
Python
"""File chunk schema module.
|
|
|
|
This module defines the FileChunk model for representing chunks of file content
|
|
in the document processing and retrieval pipeline.
|
|
"""
|
|
|
|
from pydantic import Field
|
|
|
|
from .base_node import BaseNode
|
|
|
|
|
|
class FileChunk(BaseNode):
|
|
"""A chunk of file content with positional and scoring metadata.
|
|
|
|
Represents a contiguous section of a file that has been extracted
|
|
for processing, embedding, or retrieval. Inherits text and embedding
|
|
capabilities from BaseNode.
|
|
|
|
Attributes:
|
|
path: File path relative to workspace root.
|
|
start_line: Starting line number (1-indexed) in the source file.
|
|
end_line: Ending line number (1-indexed) in the source file.
|
|
hash: Hash of the chunk content for deduplication.
|
|
scores: Search relevance scores indexed by score type.
|
|
|
|
Properties:
|
|
score: Final combined score for search result ranking.
|
|
merge_key: Unique key for merging duplicate search results.
|
|
"""
|
|
|
|
path: str = Field(..., description="File path relative to workspace")
|
|
start_line: int = Field(..., description="Starting line number (1-indexed)")
|
|
end_line: int = Field(..., description="Ending line number (1-indexed)")
|
|
hash: str = Field(..., description="Hash of chunk content for deduplication")
|
|
scores: dict[str, float] = Field(default_factory=dict, description="Search scores by type")
|
|
|
|
@property
|
|
def score(self) -> float:
|
|
"""Get the final score for search result ranking.
|
|
|
|
Returns:
|
|
The combined score, or 0.0 if not set.
|
|
"""
|
|
return self.scores.get("score", 0.0)
|
|
|
|
@property
|
|
def merge_key(self) -> str:
|
|
"""Generate a unique key for merging search results.
|
|
|
|
Returns:
|
|
A string key in format "path:start_line:end_line".
|
|
"""
|
|
return f"{self.path}:{self.start_line}:{self.end_line}"
|