mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-05 08:06:15 +00:00
* fix(bm25_index): 修正BM25索引计算中的文档长度归一化问题 修复了在计算BM25相似度时对文档长度进行不正确归一化的bug,确保所有查询都能得到准确的相关性评分。 * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * refactor(steps): Rename and adjust indexing step logic - Rename `scan_changes.py` and `reindex.py` to `clear_and_scan.py` - Update implementation details of `ScanChangesStep` and `ClearAndScanStep` - Modify the scheduling mechanism in `WatchChangesStep` - Adjust step registration and parameter configuration in config files - Update related tests to align with the new interface changes * up * feat(daily): replace daily CRUD operations with slug provisioning approach * refactor(tests): migrate CRUD step tests from HTTP server to direct LocalFileStore * up * up * up * up --------- Co-authored-by: huangsen <huangsen.huang@alibaba-inc.com>
64 lines
2.4 KiB
Python
64 lines
2.4 KiB
Python
"""Abstract base class for tokenizers."""
|
|
|
|
from abc import abstractmethod
|
|
from pathlib import Path
|
|
|
|
import aiofiles
|
|
|
|
from ..base_component import BaseComponent
|
|
from ...enumeration import ComponentEnum
|
|
|
|
|
|
class BaseTokenizer(BaseComponent):
|
|
"""Tokenizer base class with shared stopword loading and post-processing.
|
|
|
|
Subclasses implement raw tokenization via `_tokenize_one`; lowercasing and
|
|
stopword filtering are handled here so every backend behaves consistently.
|
|
"""
|
|
|
|
component_type = ComponentEnum.TOKENIZER
|
|
DEFAULT_STOPWORDS_PATH = Path(__file__).parent / "stopwords"
|
|
|
|
def __init__(
|
|
self,
|
|
stopwords_path: str | Path | None = None,
|
|
filter_stopwords: bool = True,
|
|
**kwargs,
|
|
):
|
|
super().__init__(**kwargs)
|
|
self.stopwords_path = Path(stopwords_path) if stopwords_path else self.DEFAULT_STOPWORDS_PATH
|
|
self.filter_stopwords = filter_stopwords
|
|
self._stopwords: set[str] = set()
|
|
|
|
async def _start(self) -> None:
|
|
# A missing file is non-fatal: tokenizers still work, just without filtering.
|
|
if not self.stopwords_path.exists():
|
|
self.logger.warning(f"Stopwords file not found: {self.stopwords_path}")
|
|
return
|
|
async with aiofiles.open(self.stopwords_path, encoding="utf-8") as f:
|
|
content = await f.read()
|
|
self._stopwords = {line.strip().lower() for line in content.splitlines() if line.strip()}
|
|
self.logger.info(f"Loaded {len(self._stopwords)} stopwords from {self.stopwords_path}")
|
|
|
|
async def _close(self) -> None:
|
|
self._stopwords.clear()
|
|
|
|
@property
|
|
def stopwords(self) -> set[str]:
|
|
"""Loaded stopwords (empty set if none were loaded)."""
|
|
return self._stopwords
|
|
|
|
def tokenize(self, texts: list[str], lower: bool = True, **kwargs) -> list[list[str]]:
|
|
"""Tokenize each text and apply shared post-processing."""
|
|
return [self._postprocess(self._tokenize_one(t, **kwargs), lower) for t in texts]
|
|
|
|
def _postprocess(self, tokens: list[str], lower: bool) -> list[str]:
|
|
if lower:
|
|
tokens = [t.lower() for t in tokens]
|
|
if self.filter_stopwords and self._stopwords:
|
|
tokens = [t for t in tokens if t not in self._stopwords]
|
|
return tokens
|
|
|
|
@abstractmethod
|
|
def _tokenize_one(self, text: str, **kwargs) -> list[str]:
|
|
"""Return raw tokens for one text; lowercasing/filtering happen upstream."""
|