feat(components): add token counter and file-based utility components
Some checks are pending
Pre-commit / run (ubuntu-latest) (push) Waiting to run

- Introduce BaseAsTokenCounter and EstimatedAsTokenCounter for token estimation
- Add AsMsgStat and AsBlockStat schema for message statistics tracking
- Implement FileIO class with read/write/append/edit operations
- Create file utility functions for safe async file reading and truncation
- Add MemorySearch component for semantic search in memory files
- Register new component types in ComponentEnum and update imports
- Add constants for default host, port, and truncation limits
- Create BaseService abstract base class for service implementations
- Implement BaseStep with component accessors and lifecycle management
- Add proper __all__ exports for all new modules and components
This commit is contained in:
jinli.yl 2026-04-16 20:21:04 +08:00
parent 819443813b
commit baf110e602
67 changed files with 190 additions and 94 deletions

9
reme2/__init__.py Normal file
View file

@ -0,0 +1,9 @@
"""ReMe CLI package."""
from reme2.application import Application
from reme2.component import BaseComponent
__all__ = [
"BaseComponent",
"Application",
]

View file

@ -14,13 +14,10 @@ class Application(BaseComponent):
"""Application component for managing the main application."""
def __init__(self, **kwargs) -> None:
super().__init__()
self.context = ApplicationContext(**kwargs)
working_path = Path(self.config.working_dir).absolute()
working_path.mkdir(parents=True, exist_ok=True)
memory_path = working_path / "memory"
memory_path.mkdir(parents=True, exist_ok=True)
if self.config.enable_logo:
print_logo(self.config)
@ -32,6 +29,8 @@ class Application(BaseComponent):
)
logger.info(f"Initializing {self.config.app_name} Application")
super().__init__()
from .component import R
# Initialize the service
@ -123,10 +122,10 @@ class Application(BaseComponent):
stream_queue = asyncio.Queue()
task = asyncio.create_task(job(stream_queue=stream_queue, app_context=self.context, **kwargs))
async for chunk in execute_stream_task(
stream_queue=stream_queue,
task=task,
task_name=name,
output_format="chunk",
stream_queue=stream_queue,
task=task,
task_name=name,
output_format="chunk",
):
assert isinstance(chunk, StreamChunk)
yield chunk

View file

@ -1,5 +1,7 @@
"""AgentScope TokenCounter wrappers."""
from agentscope.token import TokenCounterBase
from .estimate_token_counter import EstimatedTokenCounter
from ..base_component import BaseComponent
from ..component_registry import R
@ -7,7 +9,7 @@ from ...enumeration import ComponentEnum
class BaseAsTokenCounter(BaseComponent):
"""Base wrapper for token counters.
"""Base wrapper for AgentScope token counters.
Subclasses should implement _start() to initialize self.token_counter.
"""
@ -17,7 +19,7 @@ class BaseAsTokenCounter(BaseComponent):
def __init__(self, **kwargs) -> None:
"""Initialize with token counter configuration kwargs."""
super().__init__(**kwargs)
self.token_counter: EstimatedTokenCounter | None = None
self.token_counter: TokenCounterBase | None = None
async def _start(self, app_context=None) -> None:
"""Initialize the token counter. Override in subclasses."""
@ -26,23 +28,6 @@ class BaseAsTokenCounter(BaseComponent):
"""Release token counter resources."""
self.token_counter = None
async def count(self, messages: list[dict], **kwargs) -> int:
"""Count tokens in messages.
Args:
messages: List of message dictionaries.
**kwargs: Additional arguments passed to the token counter.
Returns:
Estimated token count.
Raises:
RuntimeError: If token counter is not initialized.
"""
if self.token_counter is None:
raise RuntimeError("Token counter not initialized. Call start() first.")
return await self.token_counter.count(messages, **kwargs)
@R.register("estimated")
class EstimatedAsTokenCounter(BaseAsTokenCounter):
@ -56,5 +41,4 @@ class EstimatedAsTokenCounter(BaseAsTokenCounter):
__all__ = [
"BaseAsTokenCounter",
"EstimatedAsTokenCounter",
"EstimatedTokenCounter",
]

View file

@ -0,0 +1,38 @@
"""Estimated token counter implementation."""
from agentscope.token import TokenCounterBase
class EstimatedTokenCounter(TokenCounterBase):
"""Token counter that estimates tokens using character-based calculation.
This is a lightweight approximation suitable for cases where exact token
counting is not critical. For accurate counts, use tiktoken or the
model's tokenizer directly.
"""
def __init__(self, estimate_divisor: float = 4):
"""Initialize the estimated token counter.
Args:
estimate_divisor: The divisor for character-to-token estimation.
Default 4 assumes roughly 4 characters per token.
Use 2-3 for Chinese/Japanese text, 4-5 for English.
"""
if estimate_divisor == 0:
raise ValueError("estimate_divisor cannot be zero")
self.estimate_divisor: float = estimate_divisor
async def count(self, text: str, **kwargs) -> int:
"""Count tokens in the given messages.
Args:
text: The text to count tokens.
**kwargs: Additional arguments.
Returns:
Estimated number of tokens in all messages.
"""
if not text:
return 0
return int(len(text.encode("utf-8")) / self.estimate_divisor + 0.5)

View file

@ -3,10 +3,11 @@
import copy
from abc import abstractmethod
from agentscope.formatter import FormatterBase
from agentscope.model import ChatModelBase
from agentscope.token import TokenCounterBase
from .application_context import ApplicationContext
from .as_llm import BaseAsLLM
from .as_llm_formatter import BaseAsLLMFormatter
from .as_token_counter import BaseAsTokenCounter
from .base_component import BaseComponent
from .embedding import BaseEmbeddingModel
from .file_store import BaseFileStore
@ -84,40 +85,46 @@ class BaseStep(BaseComponent):
return self.application_context.app_config
@property
def as_llm(self) -> BaseAsLLM:
def as_llm(self) -> ChatModelBase:
"""Get the AsLLM instance by name."""
name: str = self.kwargs.get("as_llm", "default")
llms = self.application_context.components[ComponentEnum.AS_LLM]
if name not in llms:
raise ValueError(f"AsLLM {name} not found.")
llm = llms[name]
if not isinstance(llm, BaseAsLLM):
raise TypeError(f"{name} is not a BaseAsLLM instance.")
return llm
name_or_instance = self.kwargs.get("as_llm", "default")
if isinstance(name_or_instance, ChatModelBase):
return name_or_instance
name = name_or_instance
as_llm_dict = self.application_context.components[ComponentEnum.AS_LLM]
if name not in as_llm_dict:
raise ValueError(f"AsLLM '{name}' not found.")
wrapper = as_llm_dict[name]
return wrapper.model
@property
def as_llm_formatter(self) -> BaseAsLLMFormatter:
def as_llm_formatter(self) -> FormatterBase:
"""Get the AsLLMFormatter instance by name."""
name: str = self.kwargs.get("as_llm_formatter", "default")
formatters = self.application_context.components[ComponentEnum.AS_LLM_FORMATTER]
if name not in formatters:
raise ValueError(f"AsLLMFormatter {name} not found.")
formatter = formatters[name]
if not isinstance(formatter, BaseAsLLMFormatter):
raise TypeError(f"{name} is not a BaseAsLLMFormatter instance.")
return formatter
name_or_instance = self.kwargs.get("as_llm_formatter", "default")
if isinstance(name_or_instance, FormatterBase):
return name_or_instance
name = name_or_instance
formatter_dict = self.application_context.components[ComponentEnum.AS_LLM_FORMATTER]
if name not in formatter_dict:
raise ValueError(f"AsLLMFormatter '{name}' not found.")
wrapper = formatter_dict[name]
return wrapper.formatter
@property
def as_token_counter(self):
def as_token_counter(self) -> TokenCounterBase:
"""Get the TokenCounter instance by name."""
name: str = self.kwargs.get("as_token_counter", "default")
counters = self.application_context.components[ComponentEnum.AS_TOKEN_COUNTER]
if name not in counters:
raise ValueError(f"AsTokenCounter {name} not found.")
counter = counters[name]
if not isinstance(counter, BaseAsTokenCounter):
raise TypeError(f"{name} is not a BaseAsTokenCounter instance.")
return counter
name_or_instance = self.kwargs.get("as_token_counter", "default")
if isinstance(name_or_instance, TokenCounterBase):
return name_or_instance
name = name_or_instance
counter_dict = self.application_context.components[ComponentEnum.AS_TOKEN_COUNTER]
if name not in counter_dict:
raise ValueError(f"AsTokenCounter '{name}' not found.")
wrapper = counter_dict[name]
return wrapper.token_counter
@property
def file_store(self) -> BaseFileStore:

View file

@ -0,0 +1,80 @@
# all @jinli
reme help
reme start
reme restart
reme version
reme vault="My Vault"
# daily
daily:path
reme daily:xxx
# crud
reme create file="New Note" content="# Hello" title="xxx" tags="[]" status=""
reme read file=Recipe/path="Templates/Recipe.md"
reme edit file=Recipe/path="Templates/Recipe.md" old="xxx" new="xxx"
reme append file="My Note" content="New line"
reme prepend file="My Note" content="New line"
reme delete file="My Note"/path
# reme
reme stat file/path
reme list file/path
# search
reme search query="search term" limit=10 tag="[]" score=0.1 copy=true
# property
reme property:read
reme property:update file="My Note" status=done xx=xxx
reme property:delete keys="[xxxx, xxxx]"
# 全局所有标签
reme tags
# show link
reme backlinks file="My Note"
reme links file="My Note"
[[Algorithm Notes#Sorting]]
记忆的类型
1. daily -> daily/xxxx-mm-dd/overview.md -> xxxx.md
2. topic -> topic/personal(agent)/xxxx.md
记忆的被动总结
1. 是否需要:是需要,不会主动触发
2. 什么时候调用:
- freq (every_n_turn、compact) -> daily_summarizer
- topic (/dream ) -> topic_summarizer(daily_xx -> topic_xx)
- proactive -> proactive_summarizer(personal_xxx -> proactive_query)
- pre_query
记忆的搜索
file watch
- start 全量扫描 -> 文件变化结果
- 增量扫描 -> 文件变化结果
如果有变化,检测 增加、删除、修改 cud:
1. file -> metadata 存json
- path
- mtime
- header: tags/kv/title/desc
2. file -> link 存json
- 读全量修改的文件的全量,识别link
3. file -> chunk 存db/json
- 切片 -> file+start+end: preview(100)

View file

@ -11,17 +11,18 @@ from agentscope.token import HuggingFaceTokenCounter, TokenCounterBase
from agentscope.tool import Toolkit
from .application import Application
from .component import R
from .component.runtime_context import RuntimeContext
from .component import R, RuntimeContext
from .config import parse_args
from .enumeration import ComponentEnum
from .file_based.summarizer import Summarizer
from .utils import run_coro_safely
class ReMe(Application):
"""ReMe memory management application."""
async def summary_memory(
memory_path = working_path / "memory"
memory_path.mkdir(parents=True, exist_ok=True)
async def summarize(
self,
messages: list[Msg],
as_llm: str | ChatModelBase = "default",
@ -116,6 +117,14 @@ class ReMe(Application):
"""Generate proactive memory insights."""
class ReMeLight(ReMe):
"""ReMe memory management application."""
def __init__(self, **kwargs) -> None:
super().__init__(**kwargs)
self.context.app_config.service.backend = "http"
def main():
"""Entry point for ReMe CLI."""
action, config = parse_args(sys.argv[1:])
@ -127,7 +136,7 @@ def main():
backend: str = config.pop("backend", "http")
client_cls = R.get(ComponentEnum.CLIENT, backend)
client = client_cls(action=action, **config)
asyncio.run(client())
run_coro_safely(client())
if __name__ == "__main__":

View file

@ -2,7 +2,7 @@
from .case_converter import camel_to_snake, snake_to_camel
from .chunking_utils import chunk_markdown
from .common_utils import hash_text, execute_stream_task
from .common_utils import hash_text, execute_stream_task, run_coro_safely
from .logger_utils import get_logger
from .logo_utils import print_logo
from .similarity_utils import cosine_similarity, batch_cosine_similarity
@ -14,6 +14,7 @@ __all__ = [
"chunk_markdown",
"hash_text",
"execute_stream_task",
"run_coro_safely",
"get_logger",
"print_logo",
"cosine_similarity",

View file

@ -1,9 +0,0 @@
"""ReMe CLI package."""
from reme_cli.application import Application
from reme_cli.component import BaseComponent
__all__ = [
"BaseComponent",
"Application",
]

View file

@ -1,22 +0,0 @@
from typing import Any
from agentscope.token import TokenCounterBase
class EstimatedTokenCounter(TokenCounterBase):
def __init__(self, estimate_divisor: float = 4):
if estimate_divisor == 0:
raise ValueError("estimate_divisor cannot be zero")
self.estimate_divisor: float = estimate_divisor
async def count(
self,
messages: list[dict],
text: str | None = None,
**kwargs: Any,
) -> int:
if not text:
return 0
else:
return int(len(text.encode("utf-8")) / self.estimate_divisor + 0.5)