mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-30 01:52:29 +00:00
* fix(bm25_index): 修正BM25索引计算中的文档长度归一化问题 修复了在计算BM25相似度时对文档长度进行不正确归一化的bug,确保所有查询都能得到准确的相关性评分。 * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * up * refactor(steps): Rename and adjust indexing step logic - Rename `scan_changes.py` and `reindex.py` to `clear_and_scan.py` - Update implementation details of `ScanChangesStep` and `ClearAndScanStep` - Modify the scheduling mechanism in `WatchChangesStep` - Adjust step registration and parameter configuration in config files - Update related tests to align with the new interface changes * up * feat(daily): replace daily CRUD operations with slug provisioning approach * refactor(tests): migrate CRUD step tests from HTTP server to direct LocalFileStore * up * up * up * up --------- Co-authored-by: huangsen <huangsen.huang@alibaba-inc.com>
43 lines
1.4 KiB
Python
43 lines
1.4 KiB
Python
"""Jieba tokenizer for Chinese text segmentation."""
|
|
|
|
from typing import Callable
|
|
|
|
from .base_tokenizer import BaseTokenizer
|
|
from ..component_registry import R
|
|
|
|
|
|
@R.register("jieba")
|
|
class JiebaTokenizer(BaseTokenizer):
|
|
"""Tokenizer backed by jieba for Chinese word segmentation.
|
|
|
|
`backend` selects the underlying implementation:
|
|
- "rjieba": Rust binding of jieba-rs, ~10-30x faster than pure Python (default).
|
|
- "jieba": Original pure-Python jieba, slowest but the reference.
|
|
"""
|
|
|
|
SUPPORTED_BACKENDS = ("rjieba", "jieba")
|
|
|
|
def __init__(self, backend: str = "rjieba", **kwargs):
|
|
super().__init__(**kwargs)
|
|
if backend not in self.SUPPORTED_BACKENDS:
|
|
raise ValueError(
|
|
f"Unknown jieba backend {backend!r}; expected one of {self.SUPPORTED_BACKENDS}",
|
|
)
|
|
self.backend = backend
|
|
self._cut: Callable[[str], list[str]] | None = None
|
|
|
|
async def _start(self) -> None:
|
|
await super()._start()
|
|
# Resolve the backend once at startup so per-call overhead is just one attribute lookup.
|
|
if self.backend == "rjieba":
|
|
import rjieba
|
|
|
|
self._cut = rjieba.cut
|
|
else:
|
|
import jieba
|
|
|
|
self._cut = jieba.cut
|
|
self.logger.info(f"JiebaTokenizer using backend: {self.backend}")
|
|
|
|
def _tokenize_one(self, text: str, **kwargs) -> list[str]:
|
|
return list(self._cut(text))
|