mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-07 08:26:06 +00:00
* fix(bm25_index): 修正BM25索引计算中的文档长度归一化问题 修复了在计算BM25相似度时对文档长度进行不正确归一化的bug,确保所有查询都能得到准确的相关性评分。 * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * refactor(steps): Rename and adjust indexing step logic - Rename `scan_changes.py` and `reindex.py` to `clear_and_scan.py` - Update implementation details of `ScanChangesStep` and `ClearAndScanStep` - Modify the scheduling mechanism in `WatchChangesStep` - Adjust step registration and parameter configuration in config files - Update related tests to align with the new interface changes * up * feat(daily): replace daily CRUD operations with slug provisioning approach * refactor(tests): migrate CRUD step tests from HTTP server to direct LocalFileStore * up * up * up * up --------- Co-authored-by: huangsen <huangsen.huang@alibaba-inc.com>
24 lines
880 B
Python
24 lines
880 B
Python
"""Regex tokenizer with Chinese character splitting."""
|
|
|
|
import re
|
|
|
|
from .base_tokenizer import BaseTokenizer
|
|
from ..component_registry import R
|
|
|
|
|
|
@R.register("regex")
|
|
class RegexTokenizer(BaseTokenizer):
|
|
"""Regex tokenizer: each CJK char is its own token, non-CJK uses word boundaries.
|
|
|
|
Treating CJK characters as individual tokens avoids needing a Chinese
|
|
segmenter while still giving BM25-style indexes useful unigrams.
|
|
"""
|
|
|
|
WORD_PATTERN = re.compile(r"(?u)\b\w\w+\b") # non-CJK words, 2+ chars
|
|
CHINESE_PATTERN = re.compile(r"[一-鿿]")
|
|
|
|
def _tokenize_one(self, text: str, **kwargs) -> list[str]:
|
|
# Pull CJK chars first, then strip them out so the word regex only sees the rest.
|
|
tokens = self.CHINESE_PATTERN.findall(text)
|
|
tokens.extend(self.WORD_PATTERN.findall(self.CHINESE_PATTERN.sub(" ", text)))
|
|
return tokens
|