mirror of
https://github.com/agentscope-ai/ReMe.git
synced 2026-09-29 01:41:38 +00:00
Reframe digest as the abstract memory layer (details stay in the daily/
resource material; digest holds principles, patterns, precedents reachable
via derived_from provenance edges). Replaces the old digester with a
2-phase ReAct workflow + a daily-tick wrapper:
- Phase 1 (Dreamer extract): clusters material into orthogonal memory
sub-units; each sub-unit maps 1:1 to a digest node (no inner atom
enumeration). Biases toward fewer / richer sub-units.
- Phase 2 (Dreamer integrate per sub-unit): cross-bucket recall +
exactly one write decision (CREATE / UPDATE / SKIP); UPDATE shapes
surfaced explicitly (corroborate / refine / correct).
- CronDreamer: scans <daily_dir>/<today>.md + <daily_dir>/<today>/**
+ <resource_dir>/<today>/** and runs dream_one per file.
Write tools are proper subclasses of the canonical file_io WriteStep /
EditStep with only path-shape + bucket + E-1 edge-conservation rules
layered on top:
- DigestWriteStep(WriteStep): path = <digest_dir>/<bucket>/<slug>.md,
must-not-exist, schema mirrors `write` (path / name / description /
content) so frontmatter lands automatically.
- DigestEditStep(EditStep): body-only find-and-replace + must-exist +
E-1 conservation preflight (refuses if any outbound wikilink would
be dropped).
Configuration:
- Bucket vocabulary structured in code (tuple[{name, description}]);
prompt renders the heuristic block at runtime via {buckets}.
- digest_dir / daily_dir / resource_dir come from app config (not tool
params); prompts use {digest_dir} placeholder.
- BaseStep walks class MRO when loading prompts, so subclasses inherit
parent yaml without duplication.
Tooling: agentscope register_tool_function schemas now wrap in the
proper {"type":"function","function":{...}} envelope. OpenAIAsLLM
routes base_url through client_kwargs so non-default endpoints work.
Smoke: tests4/smoke/{_dreamer_fixture.py,test_dreamer_inproc.py,
test_dreamer_cli.sh} drive the end-to-end pipeline.
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
243 lines
9.4 KiB
Python
243 lines
9.4 KiB
Python
"""Base step class for LLM workflow execution."""
|
|
|
|
import copy
|
|
from abc import abstractmethod, ABC
|
|
from pathlib import Path
|
|
from typing import TypeVar, TYPE_CHECKING
|
|
|
|
from agentscope.formatter import FormatterBase
|
|
from agentscope.message import TextBlock
|
|
from agentscope.model import ChatModelBase
|
|
from agentscope.token import TokenCounterBase
|
|
from agentscope.tool import Toolkit, ToolResponse
|
|
|
|
from ..components.base_component import ComponentMixin
|
|
from ..components.embedding import BaseEmbeddingModel
|
|
from ..components.file_parser import BaseFileParser
|
|
from ..components.file_store import BaseFileStore
|
|
from ..components.prompt_handler import PromptHandler
|
|
from ..components.runtime_context import RuntimeContext
|
|
from ..enumeration import ComponentEnum
|
|
from ..schema import FileChunk, FileNode, Response
|
|
|
|
if TYPE_CHECKING:
|
|
from ..components import ApplicationContext
|
|
from ..components.job import BaseJob
|
|
|
|
T = TypeVar("T")
|
|
|
|
_UNSET = object()
|
|
|
|
|
|
class Ref:
|
|
"""Descriptor that lazily resolves a component dependency for Steps.
|
|
|
|
Replaces the ``@property`` + ``_resolve()`` boilerplate with a single
|
|
class-level declaration::
|
|
|
|
as_llm = Ref(ChatModelBase, ComponentEnum.AS_LLM, "model")
|
|
file_store = Ref(BaseFileStore, ComponentEnum.FILE_STORE)
|
|
|
|
Resolution follows a 3-source fallback identical to the old ``_resolve``:
|
|
``kwargs`` -> ``context`` -> ``app_context`` component registry.
|
|
The resolved value is cached on the instance for its lifetime
|
|
(steps are rebuilt per job call via ``_build_steps``).
|
|
"""
|
|
|
|
__slots__ = ("base_cls", "comp_enum", "attr", "optional", "key", "_cache_attr")
|
|
|
|
def __init__(
|
|
self,
|
|
base_cls: type,
|
|
comp_enum: ComponentEnum,
|
|
attr: str | None = None,
|
|
*,
|
|
optional: bool = False,
|
|
) -> None:
|
|
self.base_cls = base_cls
|
|
self.comp_enum = comp_enum
|
|
self.attr = attr
|
|
self.optional = optional
|
|
self.key: str = ""
|
|
self._cache_attr: str = ""
|
|
|
|
def __set_name__(self, owner: type, name: str) -> None:
|
|
self.key = name
|
|
self._cache_attr = f"_ref_{name}"
|
|
|
|
def __get__(self, obj: "BaseStep | None", objtype: type | None = None):
|
|
if obj is None:
|
|
return self
|
|
cached = obj.__dict__.get(self._cache_attr, _UNSET)
|
|
if cached is not _UNSET:
|
|
return cached
|
|
value = self._resolve(obj)
|
|
obj.__dict__[self._cache_attr] = value
|
|
return value
|
|
|
|
def __set__(self, obj: "BaseStep", value) -> None:
|
|
obj.__dict__[self._cache_attr] = value
|
|
|
|
def __delete__(self, obj: "BaseStep") -> None:
|
|
obj.__dict__.pop(self._cache_attr, None)
|
|
|
|
def _resolve(self, obj: "BaseStep"):
|
|
for source in (obj.kwargs, obj.context or {}):
|
|
value = source.get(self.key)
|
|
if isinstance(value, self.base_cls):
|
|
return value
|
|
|
|
name = obj.kwargs.get(self.key, "default")
|
|
if obj.app_context is None:
|
|
if self.optional:
|
|
return None
|
|
raise RuntimeError(f"app_context is not set when resolving '{self.key}'")
|
|
comp = obj.app_context.components[self.comp_enum].get(name)
|
|
if comp is None:
|
|
if self.optional:
|
|
return None
|
|
raise KeyError(f"Component '{name}' not found in {self.comp_enum.value}")
|
|
return getattr(comp, self.attr) if self.attr else comp
|
|
|
|
|
|
class BaseStep(ComponentMixin, ABC):
|
|
"""Composable unit of an LLM workflow."""
|
|
|
|
component_type = ComponentEnum.STEP
|
|
|
|
def __new__(cls, *args, **kwargs):
|
|
# Snapshot init args so copy() can rebuild an equivalent instance later.
|
|
instance = object.__new__(cls)
|
|
instance._init_args = copy.copy(args)
|
|
instance._init_kwargs = copy.copy(kwargs)
|
|
return instance
|
|
|
|
def __init__(
|
|
self,
|
|
name: str | None = None,
|
|
backend: str = "",
|
|
app_context: "ApplicationContext | None" = None,
|
|
language: str = "",
|
|
prompt_dict: dict[str, str] | None = None,
|
|
input_mapping: dict[str, str] | None = None,
|
|
output_mapping: dict[str, str] | None = None,
|
|
**kwargs,
|
|
):
|
|
super().__init__(name=name, backend=backend, app_context=app_context, **kwargs)
|
|
self.language: str = language
|
|
self.input_mapping = input_mapping
|
|
self.output_mapping = output_mapping
|
|
self.context: RuntimeContext | None = None
|
|
|
|
# Load class-level prompts first, then overlay caller-provided overrides.
|
|
# Walk MRO in reverse so most-derived class wins; subclasses without their
|
|
# own YAML inherit prompts from their parent (e.g. CronDreamer inherits
|
|
# dreamer.yaml from Dreamer).
|
|
self.prompt = PromptHandler(language=self.language)
|
|
for cls in reversed(self.__class__.__mro__):
|
|
self.prompt.load_prompt_by_class(cls)
|
|
self.prompt.load_prompt_dict(prompt_dict)
|
|
|
|
# ----- Component references (resolved lazily on first access) ----------
|
|
|
|
as_llm: ChatModelBase = Ref(ChatModelBase, ComponentEnum.AS_LLM, "model")
|
|
as_llm_formatter: FormatterBase = Ref(FormatterBase, ComponentEnum.AS_LLM_FORMATTER, "formatter")
|
|
as_token_counter: TokenCounterBase = Ref(TokenCounterBase, ComponentEnum.AS_TOKEN_COUNTER, "token_counter")
|
|
file_store: BaseFileStore = Ref(BaseFileStore, ComponentEnum.FILE_STORE)
|
|
embedding: BaseEmbeddingModel = Ref(BaseEmbeddingModel, ComponentEnum.EMBEDDING_MODEL)
|
|
|
|
@abstractmethod
|
|
async def execute(self):
|
|
"""Run the step's logic against ``self.context``."""
|
|
|
|
async def __call__(self, context: RuntimeContext | None = None, **kwargs):
|
|
# Clear cached Ref values so context-supplied overrides take effect.
|
|
for key in [k for k in self.__dict__ if k.startswith("_ref_")]:
|
|
del self.__dict__[key]
|
|
self.context = RuntimeContext.from_context(context, **kwargs)
|
|
assert self.context is not None
|
|
if self.input_mapping:
|
|
self.context.apply_mapping(self.input_mapping)
|
|
result = await self.execute()
|
|
if self.output_mapping:
|
|
self.context.apply_mapping(self.output_mapping)
|
|
return result
|
|
|
|
async def parse_file(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]:
|
|
"""Parse ``path`` with the parser whose ``supported_extensions`` claims its suffix.
|
|
|
|
First registered match wins (config insertion order). Falls back to the
|
|
``default`` parser (stat-only) when no parser claims the suffix — that's
|
|
how attachments / binaries / unknown types still produce a FileNode.
|
|
"""
|
|
if self.app_context is None:
|
|
raise RuntimeError("app_context is not set when resolving file parser")
|
|
file_parser_dict: dict[str, BaseFileParser] = self.app_context.components[ComponentEnum.FILE_PARSER]
|
|
|
|
suffix = Path(path).suffix.lstrip(".").lower()
|
|
|
|
parser: BaseFileParser | None = None
|
|
if suffix:
|
|
for candidate in file_parser_dict.values():
|
|
if suffix in {ext.lower().lstrip(".") for ext in candidate.supported_extensions}:
|
|
parser = candidate
|
|
break
|
|
|
|
if parser is None:
|
|
parser = file_parser_dict.get("default")
|
|
|
|
if parser is None:
|
|
raise RuntimeError(
|
|
f"No file parser supports {path} (suffix={suffix!r}) and no 'default' parser is configured",
|
|
)
|
|
|
|
return await parser.parse(path)
|
|
|
|
def prompt_format(self, prompt_name: str, **kwargs) -> str:
|
|
"""Format a named prompt template with the given kwargs."""
|
|
return self.prompt.prompt_format(prompt_name=prompt_name, **kwargs)
|
|
|
|
def get_prompt(self, prompt_name: str) -> str:
|
|
"""Return a named prompt template as-is."""
|
|
return self.prompt.get_prompt(prompt_name=prompt_name)
|
|
|
|
def copy(self, **kwargs) -> "BaseStep":
|
|
"""Construct a new instance from the original init args, applying overrides."""
|
|
return self.__class__(*self._init_args, **{**self._init_kwargs, **kwargs})
|
|
|
|
def get_job(self, name: str) -> "BaseJob | None":
|
|
"""Return a job by name."""
|
|
if self.app_context is None:
|
|
raise RuntimeError("Cannot get job without an app context")
|
|
return self.app_context.jobs.get(name)
|
|
|
|
async def run_job(self, name: str, **kwargs) -> Response:
|
|
"""Execute a job by name and kwargs, return the final response."""
|
|
job: "BaseJob | None" = self.get_job(name)
|
|
if job is None:
|
|
raise RuntimeError(f"Job {name} not found")
|
|
return await job(**kwargs)
|
|
|
|
def add_as_tool(self, toolkit: Toolkit, job_name: str, **kwargs) -> None:
|
|
"""Add the step as a tool to the toolkit."""
|
|
job: "BaseJob | None" = self.get_job(job_name)
|
|
if job is None:
|
|
raise RuntimeError(f"Job {job_name} not found")
|
|
|
|
async def run_job(**_kwargs) -> ToolResponse:
|
|
response = await job(**{**_kwargs, **kwargs})
|
|
return ToolResponse(content=[TextBlock(type="text", text=response.answer)])
|
|
|
|
toolkit.register_tool_function(
|
|
tool_func=run_job,
|
|
func_name=job_name,
|
|
func_description=job.description,
|
|
json_schema={
|
|
"type": "function",
|
|
"function": {
|
|
"name": job_name,
|
|
"description": job.description,
|
|
"parameters": job.parameters,
|
|
},
|
|
},
|
|
)
|