From e16b52e68bc13285561b712a7bc825d896b4b722 Mon Sep 17 00:00:00 2001 From: huangsen Date: Fri, 29 May 2026 10:55:10 +0800 Subject: [PATCH] feat(dream): replace digester with abstraction-layer dreamer pipeline Reframe digest as the abstract memory layer (details stay in the daily/ resource material; digest holds principles, patterns, precedents reachable via derived_from provenance edges). Replaces the old digester with a 2-phase ReAct workflow + a daily-tick wrapper: - Phase 1 (Dreamer extract): clusters material into orthogonal memory sub-units; each sub-unit maps 1:1 to a digest node (no inner atom enumeration). Biases toward fewer / richer sub-units. - Phase 2 (Dreamer integrate per sub-unit): cross-bucket recall + exactly one write decision (CREATE / UPDATE / SKIP); UPDATE shapes surfaced explicitly (corroborate / refine / correct). - CronDreamer: scans /.md + //** + //** and runs dream_one per file. Write tools are proper subclasses of the canonical file_io WriteStep / EditStep with only path-shape + bucket + E-1 edge-conservation rules layered on top: - DigestWriteStep(WriteStep): path = //.md, must-not-exist, schema mirrors `write` (path / name / description / content) so frontmatter lands automatically. - DigestEditStep(EditStep): body-only find-and-replace + must-exist + E-1 conservation preflight (refuses if any outbound wikilink would be dropped). Configuration: - Bucket vocabulary structured in code (tuple[{name, description}]); prompt renders the heuristic block at runtime via {buckets}. - digest_dir / daily_dir / resource_dir come from app config (not tool params); prompts use {digest_dir} placeholder. - BaseStep walks class MRO when loading prompts, so subclasses inherit parent yaml without duplication. Tooling: agentscope register_tool_function schemas now wrap in the proper {"type":"function","function":{...}} envelope. OpenAIAsLLM routes base_url through client_kwargs so non-default endpoints work. Smoke: tests4/smoke/{_dreamer_fixture.py,test_dreamer_inproc.py, test_dreamer_cli.sh} drive the end-to-end pipeline. Co-Authored-By: Claude Opus 4.7 --- docs4/auto_dream_design.md | 756 ++++++++-------------------- docs4/auto_link_design.md | 52 +- docs4/auto_maintain_design.md | 70 +-- docs4/auto_memory_design.md | 4 +- reme4/components/as_llm/__init__.py | 8 +- reme4/config/default.yaml | 47 ++ reme4/steps/__init__.py | 12 + reme4/steps/base_step.py | 7 +- reme4/steps/dream/__init__.py | 19 + reme4/steps/dream/cron_dreamer.py | 157 ++++++ reme4/steps/dream/digest_edit.py | 121 +++++ reme4/steps/dream/digest_write.py | 138 +++++ reme4/steps/dream/dreamer.py | 639 +++++++++++++++++++++++ reme4/steps/dream/dreamer.yaml | 408 +++++++++++++++ reme4/steps/jobs/__init__.py | 3 - reme4/steps/jobs/digester.py | 300 ----------- reme4/steps/jobs/digester.yaml | 114 ----- reme4/steps/jobs/protocol.md | 112 ----- reme4/steps/transfer/ingest.py | 2 +- tests4/smoke/_dreamer_fixture.py | 273 ++++++++++ tests4/smoke/test_dreamer_cli.sh | 68 +++ tests4/smoke/test_dreamer_inproc.py | 102 ++++ tests4/unit/test_resource_steps.py | 2 +- 23 files changed, 2275 insertions(+), 1139 deletions(-) create mode 100644 reme4/steps/dream/__init__.py create mode 100644 reme4/steps/dream/cron_dreamer.py create mode 100644 reme4/steps/dream/digest_edit.py create mode 100644 reme4/steps/dream/digest_write.py create mode 100644 reme4/steps/dream/dreamer.py create mode 100644 reme4/steps/dream/dreamer.yaml delete mode 100644 reme4/steps/jobs/digester.py delete mode 100644 reme4/steps/jobs/digester.yaml delete mode 100644 reme4/steps/jobs/protocol.md create mode 100644 tests4/smoke/_dreamer_fixture.py create mode 100755 tests4/smoke/test_dreamer_cli.sh create mode 100644 tests4/smoke/test_dreamer_inproc.py diff --git a/docs4/auto_dream_design.md b/docs4/auto_dream_design.md index 4a548160..c2480e2b 100644 --- a/docs4/auto_dream_design.md +++ b/docs4/auto_dream_design.md @@ -1,18 +1,16 @@ -# auto-dream 设计(digest 沉淀:物理 + 图) +# auto-dream 设计(桶 / 节点 / 边 / 演化) -> 本文档记录 reme4 中 **auto-dream**(digest 沉淀知识层)的设计讨论 —— 含物理布局、图模型(节点 + 边)、入流端 digester(G\*)。 +> 本文档:digest 沉淀层的**桶**(物理布局)/ **节点**(原子单元)/ **边**(wikilink)/ **演化**(dream create_or_update;split 归 maintain)。 > > 配套阅读: -> - `structure.md` §1.2(数据视角)/ §2(三层存储)/ §3.5(digest 动作)/ §7(L4 模块) -> - `auto_memory_design.md`:auto-memory(daily 实时事件)是 dream 的入流之一(G1 scope 拉取);auto-memory 写完即对 dream 可见 -> - `auto_maintain_design.md`:digest 的组织 / 重组 / 写入运行时(M split / D 检测 / CAS 写入协议) —— 本文档定义节点 + 边模型与 G\*,maintain 定义运行时 -> - `auto_link_design.md`:auto-link 是 dream 写完之后的后置增强(实体识别 + wikilink 写回);复用 maintain 的 CAS 协议 +> - `structure.md` §1.2(数据视角)/ §2(三层存储)/ §3.5(digest 动作) +> - `auto_memory_design.md`:daily 实时事件 = dream 的入流之一 +> - `auto_maintain_design.md`:M split / D 检测 / CAS 写入协议(dream 模型的运行时实现) +> - `auto_link_design.md`:dream 写完后的后置增强(背景实体识别 + wikilink 写回) > -> **四层对应**:reme 服务整体四份设计 —— auto-memory(daily 入流)/ **auto-dream(本文档:digest 沉淀 + content link + G\*)** / auto-maintain(digest 组织端:split + 检测 + CAS)/ auto-link(背景图关系增强)。`structure.md` §3.5-3.6 的 L4 action 视角:dream 对应 `digest` action(`resource + daily → digest`),maintain 对应 `maintain` action(`digest → digest`)。auto-dream 承担"空闲整理"(报告 §5.2):把 daily / resource 的散点抽出共性、合并重复、生成总结,落到 `digest//.md`。 +> **核心**:digest = **浅桶(shallow bucket)+ flat .md** + **一张图(节点 + 边)**;dream 定义模型与主流程(create_or_update),maintain 负责 split / 写入运行时。 > -> **核心立场**:dream 定义**模型层**(节点 + 边 / 守恒规则)+ **生成侧**(G\* create_or_update)+ **召回**(SearchStep);组织端(M split / D 检测 / CAS / 时序)归 `auto_maintain_design.md`;后置增强(实体识别 / wikilink 写回)归 `auto_link_design.md`。三方共享 §1.5 节点 + 边模型 + §1.5.5 边守恒 + §1.5.4 F-invariants。 -> -> **关键收敛**:digest 不分"逻辑层"。整个 digest = **物理布局(浅桶 + flat .md)** + **一张图(节点 + 边)**。没有 hub / topic / leaf 之分 —— 所有 .md 文件都是同一种节点,内容决定它扮演什么角色(主题概览 / 概念定义 / 方法描述 / 实体记录 ...)。"主题"是从图中涌现的,不是结构性宣告的。 +> **关键收敛**:digest 不分"逻辑层"。所有 .md 文件都是同一种节点,内容决定它扮演什么角色(主题概览 / 概念定义 / 方法描述 / 实体记录 ...)。"主题"从图中涌现,不是结构性宣告。 --- @@ -23,556 +21,228 @@ digest 是 agent 长期记忆的"组织化沉淀"层,与三层架构的另两层 | 层 | 组织主轴 | 形态 | |---|---|---| | resource/ | 时间(`/`) | 外部原始资料,不可变 | -| daily/ | 时间 + 任务(`//`) | agent 任务过程,半可变 | +| daily/ | 时间 + 任务(`//`) | agent 任务过程,半可变 | | **digest/** | **语义** | **跨任务知识,可重组** | -digest 的核心问题: -- **物理布局**:文件系统怎么组织?(目录 / 文件 / 路径) -- **节点与边**:概念粒度 / 链接形态 / 主题如何涌现? -- **演化机制**:从碎片到网络的过程谁负责?(digester 创建或更新节点 / maintainer 拆分过载节点 / D 信号写后 inline 检测) +dream 设计回答四个问题:**桶**怎么布局 / **节点**长什么样 / **边**怎么连 / **演化**谁负责怎么做。 --- -## 1. 已对齐决策 +## 1. 桶(物理布局) -### 1.1 节点粒度:Atomic 节点优先 - -| 项 | 决策 | +| 维度 | 决策 | |---|---| -| **粒度** | 一个文件 = 一个原子单元(概念 / 方法 / 实体 / 案例 / 原则 / 主题概览) | -| **节点角色** | 由内容决定,不由 frontmatter 类型标记;一个节点扮演"主题概览"还是"具体方法",看它的 body 写了什么 | -| **风格参考** | Zettelkasten:atomic note + 高密度 wikilink 网络 | - -**理由**: -1. **节点粒度 = retrieve 精度上限** —— semantic 检索召回 "一个原子单元" 远比召回 "一个 5000 字的主题文档" 信噪比高;agent 上下文窗口经不起粗粒度文档塞满 -2. **wikilink 在 atomic 粒度才真有意义** —— `[[jwt-rotation]]` 指向"一个具体方法"比指向"auth 主题文档"精确一个数量级,这也是 I-4(wikilink 唯一跨层载体)能撑起来的前提 -3. **物理目录纯做 navigate,图承担关系** —— 两套机制各司其职,不互相绑架 - -**Tradeoff**: -- 文件数量爆炸(一个领域几百节点)→ 浅桶物理归档 + 路径作 ID,wikilink 走完整路径写法 -- 节点会频繁演化(新材料 update 已有节点 / 节点过载触发 split)→ G\* 去重质量、SearchStep 召回、写后 inline 检测都得到位 - -### 1.2 物理几何:浅桶(shallow bucket,**固定集合**) - -| 项 | 决策 | -|---|---| -| **物理布局** | `digest//.md`,bucket 一层(顶多两层),桶内 flat | -| **bucket 角色** | **仅承担物理归档与 OS-level 浏览锚点**;不承担语义本体角色 —— 主题由图中的节点表达 | -| **bucket 内** | 不再分子目录,所有节点平铺 | -| **bucket 集合** | **固定预定义,不由 digester / maintainer 动态生成** | -| **bucket 主页节点** | **不强制存在**;若 split 在该 bucket 内累积出层级(parent 节点天然中心性高),parent 节点天然成为浏览主页(纯约定,非架构必需) | -| **新节点归属** | digester (G4) 只能从已有桶集合里**挑选**;LLM 不能造新桶 | -| **集合来源** | vault 配置(opinionated default + 消费层可改),与 schema/prompt 同属服务消费层 | - -**理由**: -- 物理浏览有"主题轮廓"(打开 `digest/auth/` 能看到这一族节点),不像纯 flat 那样毫无锚点 -- 节点不被深路径绑死("属 auth/jwt 还是 auth/session"这种归属焦虑被消解 —— 一个节点可以同时被多个主题通过 wikilink 引用) -- maintain 操作面坍缩到只剩 split:节点过载就拆,不做跨节点重组(详 §1.5 / §2) -- **固定集合的关键意义**: - - LLM 在 G4 桶决定时只做"分类",不做"造类" —— 决策面坍缩,错率大幅下降 - - bucket 集合作为**预先约定的物理归档规则**,跨任务跨时间稳定;不会出现 "auth-stuff" / "auth" / "authentication" 三个语义重叠的桶共存 - - 与 reme 核心立场一致:bucket 集合是消费层契约,reme 不自主造桶 -- **bucket 主页节点不强制的关键意义**: - - 旧设计中 root hub 是架构必需(检测稳定锚点 / vault 总览结构性入口);新设计中这些角色都由图中心性自然承担,不需要为每个 bucket 强制创建一个空架子节点 - - 主页节点的"主页"地位是涌现的:某节点入度高 / 中心性高 → 它就是浏览入口 - - 若一个 bucket 完全没节点,它就只是个空目录;不需要先造一个 placeholder - -**未归类节点**:digester 抽到一个原子单元但找不到合适的专属桶时,**不允许造新桶**;统一落入兜底桶 `digest/general/`。general 桶在固定集合内是一等公民,详见 §3.7。 - -### 1.3 节点身份:路径即 ID - -| 项 | 决策 | -|---|---| -| **ID 载体** | **vault-relative 路径**(含 `.md`)即节点身份 —— `digest/auth/jwt-rotation.md` | -| **wikilink 写法** | `[[digest/auth/jwt-rotation.md]]`(literal,与 `wikilink_handler.py` 默认形态对齐;不隐含 `.md`,无 short-form 补全) | -| **`name` frontmatter** | 文件名 basename(不含扩展名),与文件名同步 —— 检索 hint / 人读标签,**不当 ID 用** | -| **同名冲突** | 同 bucket 内文件名冲突 → 文件系统层断言;**不需要独立 D6 信号** | -| **rename 成本** | 一次 `wikilink_handler.retarget_links(old_path, new_path)`,机制现成 | -| **跨桶移动** | F-1 已禁止;若必须做(人工介入修错桶),走一次 retarget | - -**理由**: -- F-1(0 文件移动)+ 平铺 + 下层 immutable 后,slug abstraction 的核心价值(移动鲁棒性)蒸发;只剩下"wikilink 短形式"这一项收益,但代价是 file_graph slug 索引 + D6 冲突检测 + alias 表 + retrieve 透明展开,**净亏** -- `wikilink_handler.py` docstring 自己写的就是 *Recommended form: full path relative to the vault with extension* —— literal 匹配,无 short-link 补全;路径作 ID 与核心库默认完全对齐 -- 完整路径前缀 `digest/auth/` 给 LLM 读写时提供语义 context(知道节点在哪个桶),不全是负担 -- provenance wikilink 反指 daily/resource 本来就用路径,统一后整个 vault 一种 wikilink 形态,不必区分"slug 形态 vs 路径形态" - -### 1.4 链接语义:基础 wikilink + 可选 Dataview 谓词 - -参考实现:`reme4/utils/wikilink_handler.py` + `reme4/schema/file_link.py`。 - -| 项 | 决策 | -|---|---| -| **link 基础形态** | `[[.md]]` —— target 字面取(literal,不隐含 `.md`,不自动短链补全);路径即 ID(详 §1.3) | -| **alias / image** | `[[path.md\|alias]]`(显示文本)/ `![[image.png]]`(图片资源)—— rewrite 时 alias 保持 | -| **anchor 不引入** | digest 设计层**不使用** `[[path.md#section]]` —— atomic 节点 + child 边界已充当精度替代品(详 §1.3 / §3.13);`wikilink_handler` 仍可解析 anchor 字面(供其它消费层),但 digest 不生成、不依赖、不在 split 时迁移 anchor | -| **可选谓词(Dataview 风格)** | 行级: `predicate:: [[path.md]]` / 内联: `[predicate:: [[path.md]]]`;**谓词写在 `[[]]` 外**,不是 `[[predicate::path]]` | -| **谓词标识符** | `[A-Za-z][A-Za-z0-9_]*`(如 `is_a` / `extends` / `causes` / `references`);词表**开放**,任意标识符 | -| **未类型化合法** | 绝大多数 wikilink **不加** predicate;`predicate=None` 是默认 / 常态 | -| **edge 唯一性键** | `(target_path, predicate)`(二元组);同源同标但不同 predicate = 不同边。`FileLink.target_anchor` 字段在 schema 中保留(供其它消费层),digest 层永远写 `None` | -| **类型信息载体** | 节点 frontmatter `kind` + 边 `predicate`(双轨可选);二者都是**消费层 schema 提示**,reme 核心解析 / 存储 / 索引,但**不读它们做结构决策** | -| **plugin 层扩展** | transclusion / 引用图谱视图等留给消费层加,reme 核心不固化语义 | - -**理由**: -- 与 I-4 "wikilink 是唯一跨层载体" 对齐 —— reme 核心永远只看机械拓扑 -- 与 [[reme4_schema_layering]] 一致 —— 类型语义(无论是节点 `kind` 还是边 `predicate`)都是消费层契约,reme 核心不固化 -- predicate 走 Dataview 而非内嵌:`[[]]` 内容保持"纯目标"(rewrite / retarget 不必感知 predicate);predicate 是文本上的**装饰位**,与 wikilink 解耦 -- `kind` 与 predicate 严格只是内容标签:reme 核心**只有节点这一种结构类型 + 边这一种结构关系**,kind / predicate 永远不参与"hub / topic / leaf"这类结构角色判断 - -**reme 核心对 predicate 的"透明"边界**(关键): -- G7(横向 link)、retrieve 中心性 —— 都**聚合所有 predicate** 算,不分桶 -- 只有 edge 唯一性 / 反向索引会用到 predicate(否则 `[[A]]` 和 `is_a:: [[A]]` 会被当作同一条边互相覆盖) -- 消费层若要按 predicate 做更精细的推理(如"taxonomic 路径只走 `is_a` 边"),自己读 `FileLink.predicate` 即可 - -### 1.5 图模型与节点演化 - -**核心模型**:digest = **物理布局(浅桶 + flat .md)** + **一张图(节点 + 边)**。 - -| 维度 | 形态 | -|---|---| -| **节点** | 每个 .md 文件 = 一个节点;无结构性 kind,角色由 body 内容决定(主题概览 / 概念定义 / 方法描述 / 实体记录 / 案例 ...) | -| **边** | 基础 `[[.md]]` wikilink(路径即 target,详 §1.3 / §1.4);可选 Dataview 谓词 `predicate:: [[path.md]]` / `[predicate:: [[path.md]]]` 写在 `[[]]` 外;边唯一性键 = `(target, predicate)`;**digest 层不引入 anchor**(详 §1.4 / §3.13);**reme 核心结构决策不读 predicate**(详 §1.4) | -| **多归属** | 一个节点可被多个其它节点引用,也可指向多个其它节点;**不存在"单父"约束** | - -**演化只做两件事**: -1. **G\* create_or_update**:新材料进入,LLM 提取原子单元 → 命中已有节点就 update 该节点 body(语义守恒地融合新旧),否则新建节点 -2. **M split**:节点累积过载(token / 主题离散度超阈值)→ LLM 把它拆成 parent overview + N 个 children,parent 文件原地保留作 overview,children 是新文件 - -"主题概览节点 / 摘要节点"不是一种 kind,也不是 maintainer 主动涌现的产物 —— 它是 split 的副产品(parent 节点天然成为该 cluster 的 overview)。 - -#### 1.5.1 单一节点 / 单一边 - -**节点 frontmatter** —— 只有保留字段: - -| 字段 | 内容 | 用途 | -|---|---|---| -| `name` | 文件名 basename(不含扩展名),与文件名同步 | 检索 hint / 人读标签(I-4 不再用它做身份;路径才是 ID,详 §1.3) | -| `description` | 一句话 | 标题 / 检索 hint | -| (可选)`kind` | concept / method / case / entity / topic / ... | **消费层 schema 提示**,reme 核心透明,不读它做结构决策 | -| body | 任意内容 | 一句定义 / 一段方法 / 一篇主题概览 / 一份案例,皆可 | - -`hub__` / `topic__` 前缀**不存在**;文件名自然命名(`auth-fundamentals.md` / `jwt-rotation.md` / `jwt-overview.md`)。"overview 节点"靠内容形态识别,不靠前缀。 - -**边的形态** —— 与 §1.4 一致,这里给最小汇总: - -| 维度 | 形态 | -|---|---| -| 基础 | `[[.md]]`(无谓词;常态;`predicate=None`) | -| 可选谓词 | `predicate:: [[path.md]]`(行级)/ `[predicate:: [[path.md]]]`(内联);谓词在 `[[]]` 外 | -| alias | `[[path.md|display-text]]`(rewrite 时 alias 保持) | -| image | `![[image.png]]`(资源引用,不是知识边) | -| **不引入 anchor** | digest 层不使用 `#section`;详 §1.4 / §3.13 | -| **边唯一性键** | `(target_path, predicate)` —— 同源同标不同 predicate = 不同边 | -| **角色识别** | 默认无谓词时由端点内容形态推断;有谓词时谓词即角色标签(消费层语义,核心不读) | - -> **关键收敛**:reme 核心**只有节点 + 边两种结构类型**;kind / predicate 都是内容标签,绝不参与 hub / topic / leaf 这类结构角色。 - -#### 1.5.2 节点演化:create_or_update + split - -``` -[入流] material 进入(daily / resource) - │ - ▼ - G1 scope:选哪些 daily/resource 进入本轮 - │ - ▼ - G2 提取原子单元(LLM,可能产 N 个候选) - │ - ▼ - 对每个候选: - │ - ▼ - G* create_or_update(LLM 决策点) - ├─ 语义相似查 → 拉相似候选节点(top-k) - ├─ LLM 判:候选中有"同概念节点"吗? - │ ├─ 有 → update 路径 - │ │ (a) 把新内容融入已有 body(语义守恒重写) - │ │ (b) 加 provenance 反指 - │ │ (c) 必要时加 / 改 wikilink - │ └─ 无 → create 路径 - │ G3 路径 / G4 bucket / G6 provenance / G7 横向 link / 写 body - │ - ▼ - 写入(机械) - -[D3 检测] 写后立即:G\* / split 写完 body 顺手 inline 检测(token 阈值 → 超阈值则 LLM 判离散度;详 `auto_maintain_design.md` §4) - │ - ▼ - D3 派发候选节点(F-4 一次一个) - │ - ▼ - M split(LLM + 机械) - ├─ LLM 把节点 body 拆成 N 个 cluster(每个是个原子单元) - ├─ parent 文件原地保留 → body 重写为 overview + 列出 children wikilinks - ├─ 每个 child 创建新文件(文件名 / bucket / body 由 LLM 给) - ├─ children 各自加 [[.md]] 反向链接 - ├─ 边守恒机械校验:`(parent_new ∪ ∪children) ⊇ parent_old` 出边集合(F-11 / E-2) - └─ inbound 链 `[[.md]]` 不动(F-10 / E-3) —— digest 层无 anchor 链,无需 retarget -``` - -**关键**: -- **G\* update 改 subject body(语义守恒重写)** —— 新材料融入已有节点正文,要求 LLM 守住"只增不删 / 不改原意",老内容不能丢;**写入前机械校验出边强守恒**(`new outbound ⊇ old outbound`,详 §1.5.5 E-1) -- **G\* 不改其它节点正文** —— 只动 subject;不像旧 M-E 会到邻居 body append wikilink -- **M split 不改其它节点正文** —— 只动 parent(重写为 overview)+ 新建 children -- **inbound 在 split 时一律不动** —— digest 设计不引入 anchor,inbound 全是裸链 `[[.md]]`,parent 路径未变即天然有效;后续 G\* 进入若 LLM 觉得 child 粒度更合适,直接加新边到 child(F-10) - -#### 1.5.3 走一个具体例子 - -**场景**:`digest/auth/` 桶,初始只有几个零散 auth 节点,没有 jwt-rotation。 - -**第 1 轮 G\***:某 daily 提到 "JWT rotation:每 24 小时换密钥,旧密钥保留 1 小时窗口给未过期 token"。 -- 语义查 → 没找到 jwt-rotation 节点 -- 走 create 路径 → 新建 `jwt-rotation.md`,body = 一段 200 字的 rotation 描述 - -**第 2 轮 G\***:另一个 daily 提到 "JWT rotation 的 grace period 通常是 1-2 小时"。 -- 语义查 → 命中 `jwt-rotation`(高相似) -- LLM 判:这是同概念,走 update 路径 -- 把 grace period 信息**融入** `jwt-rotation.md` body(不只是 append): - ``` - Before: "...旧密钥保留 1 小时窗口..." - After: "...旧密钥保留 1-2 小时 grace period(典型值,具体看 token 寿命)..." - ``` -- body 略增长,加一条 provenance 反指 - -**第 N 轮 G\***:经过几个月,各种 daily 持续 update `jwt-rotation` —— 加了密钥派生算法、加了 RS256/HS256 区别、加了 key rotation 失败处理、加了 with-leeway 实践、加了 monitoring 建议 ... - -`jwt-rotation.md` body 现在 ~3500 token,涵盖:轮换策略 / 密钥派生 / 算法选择 / 失败处理 / 监控。 - -**触发**:D3 检测 token > 2000 阈值 → 派发候选。 - -**M split**:LLM 拉 `jwt-rotation.md` body + frontmatter,判断主题离散度(5 个相对独立的子主题),决定拆: -- parent: `jwt-rotation`(留下,body 重写为 overview) -- children: `jwt-key-derivation` / `jwt-algorithm-selection` / `jwt-rotation-failure-handling` / `jwt-rotation-monitoring`(4 个新文件) -- "with-leeway 实践"内容并入 parent overview(粒度太细不单独拆) - -**执行后**: - -``` -digest/auth/ -├── jwt-rotation.md ← 文件原地;body 重写为 overview -│ description: JWT 轮换策略总览 -│ body: JWT 轮换的核心是 X,主要环节包括: -│ 密钥派生 [[digest/auth/jwt-key-derivation.md]] -│ 算法选择 [[digest/auth/jwt-algorithm-selection.md]] -│ 失败处理 [[digest/auth/jwt-rotation-failure-handling.md]] -│ 监控告警 [[digest/auth/jwt-rotation-monitoring.md]] -│ (参考:with-leeway 实践 ...) -├── jwt-key-derivation.md ← 新建,body 来自原 jwt-rotation 拆出片段 -│ → [[digest/auth/jwt-rotation.md]] ← child 反指 parent -├── jwt-algorithm-selection.md ← 同上 -│ → [[digest/auth/jwt-rotation.md]] -├── jwt-rotation-failure-handling.md ← 同上 -│ → [[digest/auth/jwt-rotation.md]] -├── jwt-rotation-monitoring.md ← 同上 -│ → [[digest/auth/jwt-rotation.md]] -└── ...(其它 auth 节点不变) -``` - -**inbound 不动**:之前指 `jwt-rotation.md` 的所有外部 wikilink(无论裸链还是 typed)都仍然指 `[[digest/auth/jwt-rotation.md]]`。如果后续某外部节点写新材料时 LLM 觉得 child 粒度更合适,直接 G\* 时新加 `[[digest/auth/jwt-key-derivation.md]]` 这种边即可 —— 不强求 split 时即时重定向。 - -**继续演化**: -- 若某个 child(如 `jwt-key-derivation`)被持续 update,某天也长到过载 → D3 又触发 → 它再次 split,自然涌现第三层 -- 若某 child 长期空 / 0 入度 / 0 update —— 不主动删(没有 dissolve 操作);除非人工介入 - -#### 1.5.4 操作的核心约束 - -| # | 约束 | 含义 | -|---|---|---| -| **F-1** | **0 文件移动** | G\* / split 都不移动现有文件;split 创建的是**新文件**,parent 文件原地 | -| **F-2** | **改正文限定 subject** | G\* update 改 subject node body(语义守恒重写,不改其它节点);M split 改 parent body(重写为 overview)+ 创建 children body;**没有任何操作改"其它节点正文"** | -| **F-3** | **maintainer 只做 split** | 没有 summarize / merge / re-edge / link / unify / dissolve;过载 → 拆 | -| **F-4** | **一次一个候选** | M split 一次拆一个节点;G\* 一次处理一个原子单元(N 个候选 = N 次 G\*) | -| **F-5** | **不确定时不动** | G\* 拿不准是 create 还是 update → 倾向 create(不污染已有节点);split 拿不准 cluster 边界 → 不拆 | -| **F-7** | **多归属合法** | 一个节点可被多个其它节点引用,也可指向多个其它节点;**没有"单父"约束** | -| **F-10** | **inbound 目标节点不动** | 所有 inbound 都是裸链 `[[.md]]`(digest 不引入 anchor,详 §1.4 / §3.13);split 时全部保持不动,parent 路径未变即天然有效;后续 G\* 进入时 LLM 可自由选择更精细 target(直接加新边到 child) | -| **F-11** | **wikilink 是 body 的一部分** | 不存在"独立的边" —— 边的所有迁移都是 body 文本变化的副作用;reme 核心机械算子只感知字符层,语义责任在 LLM(G\* / split prompt)+ 守恒校验(outbound diff 机械验证;详 §1.5.5) | - -#### 1.5.5 边的迁移规则 - -**前提**:wikilink 是 body 的一部分(F-11)。"边"不是独立抽象 —— body 一变,边就跟着变。reme 核心**没有"修边"算子**,边的所有变化都是 body 文本编辑的副作用。 - -但语义守恒不能放任 LLM:守恒责任在 prompt + 机械校验,不在算子。 - -**3 类边按"在哪类操作中变化"区分**: - -| # | 类别 | 规则 | 谁负责 | -|---|---|---|---| -| **E-1** | **G\* update 节点出边**(subject 自身) | **强守恒**:新 body 出边集合 ⊇ 原 body 出边集合(`(target, predicate)` 二元组比对,**predicate 一并守住**);不满足 → LLM 重试或拒写 | LLM(prompt 强约束)+ 机械校验(outbound diff) | -| **E-2** | **split parent 出边**(parent body 拆解) | parent overview + N 个 children 各持一段,原 parent 出边按内容自然分配到 parent overview + children;**机械校验合计守恒**:`(parent_new ∪ ∪children_outbound) ⊇ parent_old` | LLM(split prompt)+ 机械校验 | -| **E-3** | **inbound wikilink** `[[.md]]` | split 时**不动** —— 仍指 parent;后续 G\* 进入若 LLM 觉得 child 粒度更合适,直接加新边到 child(F-10) | 不动 | - -**provenance 不单列一类**:节点反指上游 daily/resource 的 wikilink 是 body 正文的一部分(§3.9),由 LLM 在 G\* / split prompt 中自然写出 —— 跟其它 body wikilink 走同一套规则:G\* update 走 E-1 强守恒(老 provenance 链不能丢,新材料追加新 provenance),split 走 E-2 合计守恒(parent 全量 provenance ⊆ parent_overview ∪ ∪children_outbound)。reme 核心**没有** provenance 专用算子。 - -**inbound anchor 这一类不存在**:digest 设计层不引入 anchor(详 §1.4 / §3.13),所有 inbound 都是裸链,走 E-3 即可,无需机械 retarget 子流程。 - -> **关键拆分**: -> - **守恒**(E-1 / E-2):LLM 写正文时不能丢边;靠 prompt + 写后 outbound diff 校验 -> - **保守**(E-3):没有信号说一定要变;不变的代价 = 后续 G\* 自然纠正,变的代价 = 错信号大量假阳;选不变 - -**机械 outbound diff 校验**(E-1 / E-2)伪码: - -``` -write_subject_body(subject, new_body): - old_outbound = extract_links(old_body) # set of (target, predicate) - new_outbound = extract_links(new_body) - missing = old_outbound - new_outbound - if missing: - # LLM 漏了原边 —— 重试一次 - new_body = llm_retry_with_missing(missing) - new_outbound = extract_links(new_body) - if old_outbound - new_outbound: - raise ConservationViolation(...) # 拒写,记 audit,等人介入 - write(subject, new_body) -``` - -机械层只做集合比对,**不判断"为什么丢了"** —— 那是 LLM 的事。 - -**强守恒(集合包含)而非等价**:`new ⊇ old` 是"新材料融入,老知识保留"的最小契约 —— 允许加新边(新关联),不允许减边(老内容不能丢);等价(`new == old`)会拒绝任何新出边,update 失去意义。 - -**predicate 守住** —— `[[A]]` ↔ `is_a:: [[A]]` 视为不同 key,升降级走显式 audit 路径,不走默认。重排 / 改 alias / 加新边都不被拦下(集合相同或只增)。 - -#### 1.5.6 图模型 vs 建子目录 - -| 维度 | 建子目录(深树) | 图模型 + split 演化 | -|---|---|---| -| 物理变化 | 移动文件,改路径 | 0 文件移动(F-1);split 只创建新文件 | -| wikilink 影响 | 路径 ID 模型下要全图 retarget(代价大) | 0 影响(parent 路径未动);split 不触发任何 retarget | -| 主题归属 | 一个节点只能属一棵子树 | 一个节点可同时属多个主题(被多源 wikilink) | -| 撤销成本 | 移回文件 + 重建上下文 | 删 children + parent body 还原(手工) | -| 演化路径 | 子树重组痛苦 | parent 只增不减,children 是 parent 拆出的快照 | -| navigate | 浏览目录树 | 任意节点入手沿出边漫游;parent 节点是天然中心 | -| retrieve 精度 | 路径反映主题但与 link 无关 | 节点中心性 + 内容形态共同决定权重 | - -#### 1.5.7 多级结构自然涌现 - -split 是节点的**局部操作**(只看一个过载节点),多级深度自然涌现: - -``` -digest/auth/ -├── auth-fundamentals.md ← 早期写下,~1500 token,稳定 -├── jwt-rotation.md ← 第一次 split:body 从 3500 token 重写为 overview -├── jwt-key-derivation.md ← 第一次 split 的 child -├── jwt-key-derivation-hkdf.md ← 二次 split:jwt-key-derivation 累积 update 后过载,再拆 -├── jwt-key-derivation-pbkdf2.md ← 二次 split 的 child -├── ... -``` - -**物理仍是浅桶(1 层),"层级"由 split 链 + 节点中心性自然承载**。每一级的过载条件、决策机制、执行步骤完全相同 —— 没有"二级 split"特殊逻辑,只有"过载节点的 body 可以被 split 进一步拆"。 - -#### 1.5.8 retrieve 时节点怎么参与 - -| query | 期望返回 | -|---|---| -| "JWT 怎么轮换" | 优先 `jwt-rotation`(具体 overview)+ children(如 `jwt-key-derivation`) | -| "auth 体系" | 优先中心性高的节点(`auth-fundamentals` / `jwt-rotation` 等被多次 update / 是 split parent 的节点) | -| "auth 有什么子主题" | 沿高中心性节点的入/出邻居遍历;parent 节点优先返回 | -| "vault 里都有什么" | 各 bucket 中心性最高的节点(自然形成 vault 总览) | - -**加权策略**(opinionated default,消费层可改): -- 节点权重 = base(=1.0) × intent 调节 × 中心性增益 -- query 含"概览 / 主题 / 入门 / 全景"等**元意图**时,中心性高的节点加权(intent 调节 > 1) -- 中心性低 / body 短的具体节点权重稳定(默认 1.0,不被压低) - -**topological traverse**: -- 沿 wikilink 自由走(不区分边类型 / predicate) -- 经过中心性高的节点默认**不强行展开**(否则一次 traverse 把整族 children 拉进来);agent 可显式深入 - -**中心性的天然来源 = split parent**:被拆过的节点是 parent,自然有 children 反向链接它,中心性自然高 —— 不需要单独维护 `kind: hub` 标记。 - ---- - -## 2. 完整能力集 - -### 2.0 设计目标 - -**让图的形状持续匹配实际知识的语义结构,在最小变更面 + 渐进演化的前提下,使任意尺度的知识访问都能命中合适粒度的节点。** - -这个目标直接来自结构本身的设计意图 —— 浅桶 + 单一节点类型 + 单一边类型 + create_or_update + split 的组合,每一项都是为它服务。能力集的入选标准:**对至少一个验证维度有贡献**。 - -| 维度 | 含义 | 失败示例 | -|---|---|---| -| **形状匹配** | 节点中心性 / 边连接 / 节点邻域反映知识间的实际语义关系 | 一个节点 token 5000+ 长期不拆;同主题节点彼此 0 链接;同概念被建成多个独立节点 | -| **最小变更面** | 不重写其它节点正文,不大规模移文件,不破坏现有 wikilink | 任何"全图重组"或"批量改其它节点正文"的方案 | -| **任意尺度访问** | 具体方法节点 / overview 节点 / 节点邻居遍历都能命中 | 全 flat,主题级 query 命中不到东西 | - -**显式排除**(不在目标内,避免能力集内卷): -- ❌ "完美归簇" —— F-5 留白,不确定就不动 -- ❌ "实时一致" —— 异步 / eventual,节点写完不必立刻 split -- ❌ "零冲突 / 零违反" —— invariants 检测 + 事后修复,不追求永不发生 -- ❌ 替消费层做检索 / 决策 —— digest 自治边界止于"维持图的形状" -- ❌ 跨节点重组(merge / re-edge / unify / dissolve)—— 简化模型不做这些;同概念二次进入由 G\* update 路径处理 - -**演化只有两件事**:G\* create_or_update(入流型,新材料融入)+ M split(后台,过载就拆)。detection 派生信号驱动这套循环。 - -### 2.1 生成侧:digester(入流型) - -| # | 能力 | 服务 | 性质 | 何时发生 | -|---|---|---|---|---| -| **G1** | **scope 决定**:选哪组 daily/resource 进入本轮蒸馏 | 形状匹配(决定形状从哪生长) | LLM | digester 启动 | -| **G2** | **原子单元抽取**:从 scope 中识别值得沉淀的原子单元(N 个候选) | 形状匹配 + 任意尺度 | LLM | 核心环节 | -| **G\*** | **create_or_update**:对每个候选,**多路召回(SearchStep:vector + keyword + 邻接展开,RRF 融合,scope 限 `digest/`)** → LLM 看完整候选池 → 终判 create / update / drop;create 路径走 G3/G4/G6/G7 + 写 body;update 路径融入已有 body(语义守恒重写)+ 自然追加 provenance(详 §3.10) | 形状匹配(去重内置)+ 最小变更面 | LLM(决策)+ 机械(召回 + 守恒写入) | 每个候选 | -| **G3** | **路径命名**(create 路径):在 G4 选定 bucket 内,文件名同 bucket 唯一(fs 层断言);风格与同主题节点一致 | 任意尺度(可寻址) | LLM(命名)+ 机械(同 bucket 文件名冲突 → 拒写) | create 时 | -| **G4** | **bucket 落地**(create 路径):从固定集合中挑选;找不到合适专属桶 → 落 `general/`(§3.7) | 形状匹配(物理归档) | LLM(读 bucket 列表) | create 时 | -| **G6** | **provenance 写入**(create 与 update):新节点 body 内联反指上游 daily/resource 的 wikilink;update 时 LLM 在融入新材料时自然追加新 provenance 链,旧 provenance 链由 E-1 守恒校验保住(§3.9) | 任意尺度(跨层访问) | LLM(prompt 引导写出 `[[daily/...]]` / `[[resource/...]]`)+ 机械(outbound diff 校验) | 写入时 | -| **G7** | **横向 link**(create 时):新节点链到相关的已有节点(出边);update 时也可加新 link | 形状匹配 + 任意尺度 | LLM | 写入时 | - -**关键边界**: -- **G\* 是入流唯一改 body 的操作**,且**只改 subject node** —— update 改的是同概念那个节点自己,不改其它节点 -- **G\* update 必须语义守恒**:LLM 重写 body 时只能"融入"新内容,不能删除已有信息(只增不删 / 不改原意) -- **0 出边节点合法**(G7 没识别到合适邻居),后续 G\* 进入时其它节点可以反向链回来 —— 不强求 LLM 一次性给全 -- **G\* 漏判去重**(把同概念建成新节点)→ 不主动兜底,接受重复;若 vault 累积明显的同概念重复,可由 auto-link 离线 audit 工具产报告(详 `auto_link_design.md` §1.3 L4) -- **digester 不做 split** —— split 是后台 M 操作 - -### 2.2 组织侧 / 检测 / 写入并发 → `auto_maintain_design.md` - -M split / D 检测信号(D1 / D3 / D10)/ 阈值校准 / D3 写后触发模型 / G\* / split / auto-link L1 三方共用的 CAS 写入协议 / split provenance / 时序 / 后门 —— 全部归 `auto_maintain_design.md`。 - -dream 保留**模型层**(§1 节点 + 边 + 守恒规则)+ **生成侧**(§2.1 G\*)+ **召回**(§3.10 SearchStep);maintain 负责**组织 / 运行时**(split + D + CAS + 时序)。两份文档共享 §1.5 节点 + 边模型、§1.5.5 边守恒、§1.5.4 F-invariants。 - -| 在 maintain 文档中 | 内容 | -|---|---| -| §1 | M split 能力卡 + 关键边界 | -| §2 | 检测信号 D1 / D3 / D10 | -| §3 | 阈值校准 | -| §4 | D3 写后触发模型 | -| §5 | CAS 写入协议(三方共用) | -| §6 | split 时 provenance | -| §7 | G\* / split / auto-link L1 时序 | -| §8 | 后门(暂缓) | - -### 2.4 边界协议(谁不能做什么) - -| 边界 | 内容 | 来源 | -|---|---|---| -| digester ∩ maintainer | digester 不做 split;maintainer 不做原子单元抽取 / 新具体节点 create | 入流 vs 自维护职责分离 | -| digester → 其它节点 | G\* update 改 subject node body,**不改其它任何节点正文** | F-2 | -| digester → "摘要 / overview" | digester 不为做 overview 而创建节点;它产的节点都是具体原子单元;overview 是后续 split 的副产品 | F-3 | -| maintainer → 其它节点 | M split 改 parent body(重写为 overview)+ 创建 N 个 children body;**不改任何其它节点** | F-2 | -| maintainer → inbound 链 | split 时**全部不动** —— digest 不引入 anchor,inbound 一律是裸链 `[[.md]]`,parent 路径未变 | F-10 / E-3 | -| digester → 边守恒 | G\* update 写新 body 前,机械对比 old/new outbound:`new ⊇ old`((target, predicate) 二元组);失败 → LLM 重试一次,再失败拒写 | F-11 / E-1 | -| maintainer → 边守恒 | split 写新 parent body + N children body 前,机械对比:`(parent_new ∪ ∪children_outbound) ⊇ parent_old`;失败 → LLM 重试或拒写 | F-11 / E-2 | -| 全员 → typed link predicate | wikilink 的 predicate 是 edge identity 的一部分;G\* update / split 不能丢 predicate(`is_a:: [[A]]` 必须保持;否则被守恒校验当作 drop edge + add edge 拦下);predicate 升 / 降级走显式 audit 路径 | F-11 / §1.4 | -| 全员 → resource/daily | 都不能改 | I-2 / I-3 | -| 全员 → 节点 rename | rename = 一次 `wikilink_handler.retarget_links(old_path, new_path)`;无 alias 表,无透明展开 | §1.3 | -| 全员 → provenance link | 永远必须可达(I 不变量 + D10 检测) | I-1 / I-4 | -| 全员 → kind 字段 | reme 核心**透明**:不读取 frontmatter `kind` 做结构决策;`kind` 是消费层 schema 提示 | [[reme4_schema_layering]] | -| 全员 → predicate 谓词 | reme 核心**结构决策不读**:G7 / 中心性都聚合所有 predicate 算;edge 唯一性 / 反向索引会用到 predicate(防同源同标不同 predicate 互相覆盖);未类型化 link 是默认形态 | [[reme4_schema_layering]] / §1.4 | - ---- - -## 3. 待对齐边界点(后续讨论清单) - -### 3.1 G\* update 的语义守恒边界(已收敛) - -**决策**:**LLM 重写整段**(prompt 强约束"语义守恒,只增不删 / 不改原意;冲突标注 `> 注:不同来源记载...`,不擅自仲裁")+ **机械守恒校验**(详 §1.5.5 E-1)。校验失败 LLM 重试一次,再失败拒写 + audit。 - -首版可先用 append 起步(出边集合天然 ⊇,守恒校验自动通过),prompt 工程量小;成熟后切到重写。 - -### 3.2 maintainer 的人 / agent 后门 → `auto_maintain_design.md` §8 - -### 3.3 G\* 与 split 的时序 → `auto_maintain_design.md` §7 - -### 3.4 节点 kind / 边 predicate(已收敛) - -reme 核心**只有节点 + 边两种结构类型**: -- frontmatter `kind` 字段(若存在)= 消费层的**节点内容标签**(concept / method / case / entity / topic / ...),reme 不读它做结构决策 -- 边 `predicate`(Dataview 风格,若存在)= 消费层的**边关系标签**(`is_a` / `extends` / `causes` / ...),reme 解析 / 存储 / 参与 edge 唯一性,但**结构决策不读**(G7 不区分 predicate;中心性不区分) -- "overview 节点"角色靠图位置(高中心性 / 是 split parent)+ body 内容形态识别,不靠 frontmatter 或 predicate 标记 -- 未类型化 wikilink 是默认 / 常态形态 - -详见 §1.4 / §1.5.1 / §2.4。 - -### 3.5 retrieve 时的权重策略(部分收敛 → §1.5.8) - -- 加权策略:节点权重 = base(=1.0) × intent 调节 × 中心性增益;query 含元意图("概览 / 主题 / 入门 / 全景"等)时,中心性高的节点加权 -- traverse 默认不强行展开高中心性节点(防止整族 children 拉进来);agent 可显式深入 -- 不按 frontmatter `kind` 加权;中心性天然来源 = split parent(详 §1.5.8) - -剩余待定:**中心性算法选型**(eigenvector / PageRank / 简单入度,初期可用入度,后续校准)。 - -### 3.6 M split 时的 provenance 处理 → `auto_maintain_design.md` §6 - -### 3.7 bucket 集合管理 - -§1.2 已定:bucket 集合**固定预定义**,不由 digester / maintainer 动态生成。补足细节: - -- **定义位置**:`vault.yaml` 顶层 `digest.buckets:` 是源 + 自动生成 `digest/_buckets.md` 作为人/LLM 可读视图;digester G4 时读后者作为 prompt context -- **初始化**:opinionated default(通用桶 `concept` / `method` / `pattern` / `tool` / `domain` 等 + 必带 `general`);消费层可改桶名,但 **`general` 不可删**(否则 G4 失去兜底) -- **扩展路径**:reme 不主动提议扩 bucket(对比旧设计的 maintainer 周期建议已 DROPPED);用户编辑 `vault.yaml` 后下次 G4 即生效 - -**未归类节点处理**(G4 找不到合适专属桶时):**统一落入 `digest/general/`**。 +| **物理几何** | `digest//.md`;**浅桶一层**(顶多两层),桶内 flat | +| **bucket 角色** | **仅承担物理归档 + OS-level 浏览锚点**;不承担语义本体角色 —— 主题由图中节点表达 | +| **bucket 集合** | **固定预定义**(`vault.yaml` 顶层 `digest.buckets:`),不由 dreamer / maintainer 动态生成 | +| **集合视图** | 自动生成 `digest/_buckets.md` 作为人/LLM 可读视图;dream 时读后者作 prompt context | +| **初始化** | opinionated default(6 桶,按"答什么问"划分):`concept`(答"X 是什么")/ `procedure`(答"怎么做 X")/ `entity`(答"X 是谁/哪个")/ `observation`(答"发生了什么")/ `preference`(答"X 喜欢怎样 / 别做什么";用户记忆主战场)/ **必带 `unknown`** —— 消费层可改桶名,但 `unknown` 不可删 | +| **bucket 主页** | 不强制存在;split 累积出层级时 parent 节点天然成为浏览主页(中心性涌现,非架构必需) | +| **新节点归属** | dream 只能从已有桶集合**挑选**;LLM 不能造新桶 | +| **未归类节点** | 统一落入兜底桶 `digest/unknown/`(详下) | +| **跨桶 move** | F-1 已禁止;若必须做(人工介入修错桶),走一次 `wikilink_handler.retarget_links(old, new)` | + +**`unknown` 兜底桶**: | 维度 | 内容 | |---|---| -| **bucket 名** | `general`(固定集合一等公民,默认包含) | -| **语义** | "通用主题 / 暂无专属归属" —— 合法常态,非故障状态 | -| **路径** | `digest/general/.md`,与其它 bucket 完全等同 | -| **节点演化** | 与其它 bucket 一致 | -| **错桶后续** | 不主动跨桶 move(无 D9 / M-D);若严重,人工 mv + `retarget_links(old, new)` | +| **语义** | "分类未定 / 暂无专属归属" —— **合法常态,非故障状态**(LLM 没找到合适专属桶时的诚实表达,不是写入失败) | +| **路径** | `digest/unknown/.md`,与其它 bucket 完全等同;节点演化与其它桶一致 | +| **错桶后续** | 不主动跨桶 move;若严重,人工 mv + `retarget_links(old, new)` | -**为什么是 `general` 而不是 `_unclassified`**:`_unclassified` 暗示待处理状态,LLM/人都想清理掉;`general` 是合法常态,G4 选桶时是显式合法选项而非 fallback 故障路径。 +**为什么是 `unknown` 而不是 `general`**:`general` 听起来像在断言"这个概念真的属于通用类",实际上只是 LLM 没找到合适专属桶;`unknown` 诚实表达"分类未定",不强加伪类目。但 `unknown` 不是"待清理状态" —— 它是合法常态,节点在此演化(被 update / 被 wikilink 引 / 中心性增长)与其它桶完全等同;不会随时间被自动清空。 -**已排除**:拒绝写入(候选丢失)/ 强行选最近似专属桶(本体污染,general 反而更安全)。 +**为什么是浅桶而不是深树**: +- 物理浏览有"主题轮廓"(打开 `digest/auth/` 能看到这一族节点),不像纯 flat 那样毫无锚点 +- 节点不被深路径绑死("属 auth/jwt 还是 auth/session"这种归属焦虑被消解 —— 一个节点可以同时被多个主题通过 wikilink 引用) +- F-1(0 文件移动)+ 平铺后,深树的核心收益(子树重组)消失,只剩深路径维护负担 +- **固定集合的关键意义**:LLM 在 dream 桶决定时只做"分类",不做"造类" —— 决策面坍缩,跨任务跨时间稳定;不会出现 "auth-stuff" / "auth" / "authentication" 三个语义重叠的桶共存 -### 3.8 检测阈值校准 → `auto_maintain_design.md` §3 - -### 3.9 provenance 载体形态(已收敛) - -**决策**:**provenance wikilink 嵌在节点 body 正文中**(inline body prose),由 LLM 在 G\* / split prompt 里自然写出,跟其它 body wikilink 完全同形,**靠语义维护**。reme 核心没有 provenance 专用算子。 - -**写出形态**: -- 行文中自然带出处:"... 该模式最早出现在 [[daily/2026/05/15.md]] 的实践中" -- 或专门一段总结式段落,内含若干 wikilink 指向上游 -- 可选 predicate(`derived_from:: [[daily/2026/05/15.md]]`),不强制 - -**机械保护**: -- E-1 守恒(G\* update):旧 provenance 不在新 outbound 集合 → 重试或拒写,机械兜底 -- E-2 守恒(split):provenance 跟着对应内容段自然分配到 parent overview / children,合计守恒 -- D10 检测:provenance 断裂 = D1 断链子集(target 命中 `daily/` / `resource/` 前缀);D10 严重程度高于普通 D1(I 不变量) - -**E-5 / G6 等"provenance 专用机制"全部坍缩** —— 不再单列。Prompt 必须要求"出处用 `[[...]]` 形式表达"(纯散文会被守恒校验视为丢边)。 - -### 3.10 G\* 语义查后端(已收敛) - -**决策**:**直接复用 `SearchStep`(`reme4/steps/index/search.py`)** —— 多路召回并发(vector + keyword)+ RRF 融合 + file_graph 邻接展开,把**完整候选池交给 LLM 终判**;G\* 入口不做 bucket 粗筛(LLM 拥有完整跨桶视野,可识别"概念错分到 general"或"跨桶同概念";三路信号 RRF 融合后噪声可控)。 - -**召回链路**: -1. 候选原子单元(摘要 / 关键词)→ `SearchStep`(`search_filter={"path_prefix": "digest/"}`,I-2/I-3 daily/resource 不入池) -2. `SearchStep` 内部:`vector_search` + `keyword_search` 并发 → RRF 融合 → `expand_links` 邻接展开 → 返回 top-`limit` FileChunks(含 path / 行号 / 邻接节点) -3. 候选池整体喂 LLM,按 path 自然聚合(同节点多 chunk 命中 = 强信号);终判输出节点路径 -4. LLM 终判 create / update / drop;update 选定 subject node → 走 E-1 守恒重写 - -**provenance 不依赖召回** —— G\* / split prompt 让 LLM 直接写 `[[daily/...]]` / `[[resource/...]]`(§3.9)。 - -**索引维护**:沿用 `update_index` step,G\* / split 写 body 后调一次刷该节点索引;启动一次全建(`clear_and_scan` 已就绪),损坏走全建兜底。 - -**默认参数**(可按 dogfooding 调):`limit` 5~10 / `vector_weight` 0.7 / `expand_links` on / `min_score` 0(初版不过滤,LLM 兜底)。 - -### 3.11 G\* / split 写入并发 / 原子性 → `auto_maintain_design.md` §5 - -### 3.12 D3 触发模型 → `auto_maintain_design.md` §4 - -### 3.13 anchor 不引入 wikilink 设计(已收敛) - -**决策**:**digest 设计层不使用 `[[path.md#section]]` 形态** —— wikilink 只有 `[[path.md]]`(可选 alias / 谓词),anchor 不进入 digest。当 LLM 想"指向某个具体子主题"时,正确做法是让那个子主题升级为独立节点(必要时通过 split),而不是在过载 parent 内部用 anchor 凑合。 - -**连锁简化**: -- E-4(inbound anchor 机械 retarget)整类**消失**;split 流程末尾不再扫 inbound anchor 子流程;`{anchor → child}` 映射输出从 split prompt 中移除 -- 边唯一性键从三元组 `(target, predicate, anchor)` 简化为二元组 `(target, predicate)` -- §1.5.5 边迁移类别从 4 类(E-1..E-4)简化为 3 类(E-1..E-3) -- `FileLink.target_anchor` 字段在 schema 中保留(供其它消费层),digest 层永远写 `None` - -**Prompt 约束**:G\* / split 的 prompt 必须明确告知 LLM 写 wikilink 时不带 `#section`。若 LLM 仍写出 `[[path.md#section]]`,wikilink_handler 仍能解析,守恒校验只看 `(target, predicate)`,不会形成"丢边"风险 —— 但 anchor 在 digest 层无语义。若引用方依赖某 anchor 锚定具体段落,表明该内容应升级为 child 节点。 +**已排除**:动态扩桶 / 拒绝写入(候选丢失)/ 强行选最近似专属桶(本体污染)。 --- -## 4. 下一步 +## 2. 节点 -本文档覆盖 dream 模型 + 生成侧 + 召回(G\* / 节点+边模型 / SearchStep)。组织端实现清单(M split / D 检测 / CAS)见 `auto_maintain_design.md` §10。 +| 维度 | 决策 | +|---|---| +| **粒度** | atomic;一个 .md 文件 = 一个原子单元(概念 / 方法 / 实体 / 案例 / 原则 / 主题概览)| +| **节点角色** | **由 body 内容决定,不由 frontmatter 类型标记**;同一节点扮演"主题概览"还是"具体方法",看它的 body 写了什么 | +| **身份(ID)** | **vault-relative 路径(含 `.md`)即节点身份** —— `digest/auth/jwt-rotation.md` | +| **`name` frontmatter** | 文件名 basename(不含扩展名),与文件名同步 —— 检索 hint / 人读标签,**不当 ID 用** | +| **frontmatter 保留字段** | 只有 `name` + `description`(reme 核心保留)| +| **可选 `kind` 字段** | 例:concept / procedure / entity / observation / preference / ...;**消费层 schema 提示**,reme 核心透明,不读它做结构决策 | +| **文件名冲突** | 同 bucket 内文件名冲突 → 文件系统层断言(写入即拒);不需要独立检测信号 | +| **rename** | 一次 `wikilink_handler.retarget_links(old_path, new_path)`(机制现成);无 alias 表,无透明展开 | -1. **digester 流程图**(G1 / G2 / G\* 的实际编排;G\* 内 create / update 路径分流;**召回直接复用 `SearchStep`**(vector + keyword + 邻接展开,RRF 融合,scope `digest/`)—— 详 §3.10;**G\* update 写入前 outbound diff 守恒校验** — E-1) -2. **rename 路径设计**(`wikilink_handler.retarget_links(old_path, new_path)` 已就绪;封装为单步 step 入口,无 alias 表 / 无透明展开) -3. **bucket 集合配置**(`vault.yaml` schema / 默认桶模板 / `general` 兜底机制 / `_buckets.md` 视图生成) -4. **边守恒校验工具**(`extract_links` 已就绪;新增 outbound diff 比较器 + LLM 重试编排 + ConservationViolation audit 事件) -5. **provenance prompt 规范**(G\* / split 引导 LLM 写 `[[daily/...]]` / `[[resource/...]]` —— §3.9) +**为什么 atomic + 路径即 ID**: +- **节点粒度 = retrieve 精度上限** —— semantic 检索召回 "一个原子单元" 远比召回 "一个 5000 字的主题文档" 信噪比高 +- **wikilink 在 atomic 粒度才真有意义** —— `[[digest/auth/jwt-rotation.md]]` 指向"一个具体方法"比指向"auth 主题文档"精确一个数量级 +- F-1 + 平铺 + 下层 immutable 后,slug abstraction 的核心价值(移动鲁棒性)蒸发;路径作 ID 与 `wikilink_handler.py` 默认形态完全对齐(*Recommended form: full path relative to the vault with extension*) +- provenance wikilink 反指 daily/resource 本来就用路径,统一后整个 vault 一种 wikilink 形态 + +**"主题概览节点"靠内容识别,不靠前缀 / kind**:`hub__` / `topic__` 前缀**不存在**;文件名自然命名(`auth-fundamentals.md` / `jwt-rotation.md`)。主题概览身份是图位置(中心性 / split parent)+ body 形态共同涌现。 + +--- + +## 3. 边 + +参考实现:`reme4/utils/wikilink_handler.py` + `reme4/schema/file_link.py`。 + +| 形态 | 写法 | 说明 | +|---|---|---| +| **基础** | `[[.md]]` | literal,不隐含 `.md`,不自动短链补全 | +| **alias** | `[[path.md\|display-text]]` | rewrite 时 alias 保持 | +| **image** | `![[image.png]]` | 资源引用,不是知识边 | +| **可选谓词** | `predicate:: [[path.md]]`(行级)/ `[predicate:: [[path.md]]]`(内联) | Dataview 风格;谓词在 `[[]]` 外,`[[]]` 内只保留纯目标 | +| **谓词标识符** | `[A-Za-z][A-Za-z0-9_]*`(`is_a` / `extends` / `causes` / `references` ...) | 词表**开放**,任意标识符 | +| **未类型化合法** | 绝大多数 wikilink 不加 predicate;`predicate=None` 是默认 / 常态 | | +| **边唯一性键** | `(target_path, predicate)` 二元组 | 同源同标不同 predicate = 不同边 | +| **不引入 anchor** | digest 设计层不使用 `[[path.md#section]]` | `FileLink.target_anchor` schema 保留(供其它消费层),digest 层永远写 `None` | + +**reme 核心对 predicate 的"透明"边界**(关键): +- 横向 link/ retrieve 中心性 —— 都**聚合所有 predicate** 算,不分桶 +- 只有 edge 唯一性 / 反向索引会用到 predicate(否则 `[[A]]` 和 `is_a:: [[A]]` 会被当作同一条边互相覆盖) +- 消费层若要按 predicate 做更精细的推理(如"taxonomic 路径只走 `is_a` 边"),自己读 `FileLink.predicate` 即可 + +**与 `kind` 一致的立场**(与 [[reme4_schema_layering]] 对齐):reme 核心**只有节点 + 边两种结构类型**;`kind` / `predicate` 都是内容标签,绝不参与"hub / topic / leaf"这类结构角色判断。 + +**为什么不引入 anchor**:LLM 想"指向具体子主题"时,**正确做法是让那个子主题升级为独立节点**(必要时通过 split),不在过载 parent 内部用 anchor 凑合。anchor 在 digest 层无语义;prompt 必须明确告知 LLM 写 wikilink 时不带 `#section`。 + +--- + +## 4. 演化 + +### 4.1 演化只做两件事 + +| op | 谁 | 何时 | 改什么 | +|---|---|---|---| +| **dream**(create_or_update) | dreamer(本文档 §4.2) | 入流(新材料进入) | 创建新节点 / update 已有节点 body(语义守恒重写) | +| **M split** | maintainer(`auto_maintain_design.md` §1) | 节点过载(token / 主题离散度超阈值) | 把 parent body 拆成 parent overview + N children;parent 文件原地 | + +> **关键观察**:"主题概览节点"不是一种 kind,也不是 maintainer 主动涌现的产物 —— 它是 split 的副产品(parent 节点天然成为该 cluster 的 overview,中心性自然高)。 + +显式排除: +- ❌ merge / dissolve / re-edge / unify —— 跨节点重组不做(同概念二次进入靠 dream update;错桶节点不主动 move) +- ❌ 完美归簇 —— F-5 留白,不确定就不动 +- ❌ 实时一致 —— 异步 / eventual + +### 4.2 dream(create_or_update)流程 + +**dream = dreamer 入流唯一改 body 的操作,且只改 subject node。** + +``` +material 进入(daily / resource 选定 scope) + │ + ▼ +LLM 抽取原子单元 → N 个候选 + │ + ▼ 对每个候选: +SearchStep 召回相似候选节点 + (reme4/steps/index/search.py;vector + keyword 并发 → RRF 融合 + → expand_links 邻接展开;scope `digest/`;默认 limit 5~10) + │ + ▼ +LLM 终判:候选池里有"同概念节点"吗? + ├─ 有 → update 路径 + │ (a) 把新内容融入已有 body(语义守恒重写) + │ (b) 加 provenance 反指 + │ (c) 必要时加 / 改 wikilink + │ + └─ 无 → create 路径 + (a) 挑 bucket(固定集合;无合适专属桶 → `unknown`) + (b) 写文件名(同 bucket 唯一,fs 层断言) + (c) 写 body + provenance + 横向 link + │ + ▼ +写入前 outbound diff 守恒校验(update 走 E-1;create 无 old outbound) + │ + ▼ +CAS 写入(`auto_maintain_design.md` §5) + │ + ▼ +写完 inline 触发 D3 检测(`auto_maintain_design.md` §4) +``` + +**关键边界**: +- **dream update 必须语义守恒** —— LLM 重写 body 时只能"融入"新内容,不能删除已有信息(只增不删 / 不改原意;冲突标注 `> 注:不同来源记载...`,不擅自仲裁);**写入前机械校验出边强守恒**(E-1,详 §4.4) +- **dream 不改其它节点正文**(F-2) —— 只动 subject +- **0 出边节点合法**(没识别到合适邻居),后续 dream 进入时其它节点可以反向链回来 —— 不强求 LLM 一次性给全 +- **dream 漏判去重**(同概念建成新节点)→ 不主动兜底,接受重复;若 vault 累积明显重复,由 auto-link L4 离线 audit 工具产报告(`auto_link_design.md` §1.3) +- **召回不做 bucket 粗筛** —— LLM 拥有完整跨桶视野,可识别"概念错分到 unknown"或"跨桶同概念" + +**provenance 写出**: +- 行文中自然带:"... 该模式最早出现在 [[daily/2026/05/15.md]] 的实践中" +- 可选 predicate:`derived_from:: [[daily/2026/05/15.md]]`,不强制 +- prompt 必须要求"出处用 `[[...]]` 形式表达"(纯散文会被守恒校验视为丢边) +- **首版可先用 append 起步**(出边集合天然 ⊇,守恒校验自动通过);成熟后切到重写 + +### 4.3 F-invariants(演化的硬约束) + +| # | 约束 | 含义 | +|---|---|---| +| **F-1** | **0 文件移动** | dream / split 都不移动现有文件;split 创建的是**新文件**,parent 原地 | +| **F-2** | **改正文限定 subject** | dream update 改 subject body;M split 改 parent body + 创建 children body;**没有任何操作改"其它节点正文"** | +| **F-3** | **maintainer 只做 split** | 没有 summarize / merge / re-edge / link / unify / dissolve | +| **F-4** | **一次一个候选** | M split 一次拆一个;dream 一次处理一个原子单元(N 候选 = N 次 dream) | +| **F-5** | **不确定时不动** | dream 拿不准 create 还是 update → 倾向 create;split 拿不准 cluster → 不拆 | +| **F-7** | **多归属合法** | 一个节点可被多个引用,也可指向多个;**没有"单父"约束** | +| **F-10** | **inbound 目标节点不动** | 所有 inbound 是裸链 `[[.md]]`(digest 不引入 anchor);split 时全部保持,parent 路径未变即天然有效 | +| **F-11** | **wikilink 是 body 的一部分** | 不存在"独立的边";reme 核心机械算子只感知字符层,语义责任在 LLM(prompt)+ 守恒校验(outbound diff) | + +### 4.4 边守恒(E-1 / E-2 / E-3) + +**前提**:wikilink 是 body 的一部分(F-11)。"边"不是独立抽象 —— body 一变,边就跟着变。reme 核心**没有"修边"算子**;边的所有变化都是 body 文本编辑的副作用。但语义守恒不能放任 LLM:守恒责任在 prompt + 机械校验。 + +| # | 类别 | 规则 | 谁负责 | +|---|---|---|---| +| **E-1** | dream update 节点出边(subject 自身) | **强守恒**:新 body 出边 ⊇ 原 body 出边(`(target, predicate)` 二元组,predicate 一并守住) | LLM(prompt)+ 机械(outbound diff) | +| **E-2** | split parent 出边(parent 拆解) | `(parent_new ∪ ∪children_outbound) ⊇ parent_old` | LLM(split prompt)+ 机械 | +| **E-3** | inbound wikilink `[[.md]]` | split 时**不动** —— 仍指 parent;后续 dream 进入若 LLM 觉得 child 粒度更合适,直接加新边到 child(F-10) | 不动 | + +**机械 outbound diff 校验**(E-1 / E-2)伪码: +``` +write_subject_body(subject, new_body): + old = extract_links(old_body) + new = extract_links(new_body) + if old - new: + new_body = llm_retry_with_missing(old - new) + if old - extract_links(new_body): + raise ConservationViolation(...) # 拒写 + audit + write(subject, new_body) +``` + +**强守恒(集合包含)而非等价**:`new ⊇ old` = 允许加新边(新关联),不允许减边(老内容不能丢);`new == old` 会拒绝任何新出边 → update 失去意义。 + +**predicate 守住** —— `[[A]]` ↔ `is_a:: [[A]]` 视为不同 key,升降级走显式 audit 路径,不走默认。重排 / 改 alias / 加新边都不被拦下(集合相同或只增)。 + +**provenance 不单列** —— 节点反指上游 daily/resource 的 wikilink 是 body 正文的一部分,跟其它 wikilink 走同一套 E-1 / E-2;reme 核心没有 provenance 专用算子。 + +**inbound anchor 这一类不存在** —— digest 不引入 anchor,所有 inbound 都是裸链,走 E-3 即可,无需机械 retarget 子流程。 + +--- + +## 5. 与其它层 + +| 上下游 | 关系 | +|---|---| +| ← **auto-memory**(daily) | dream 读 daily 作为入流;daily 写完即对 dream 可见 | +| ← **resource** | dream 读 resource 作为入流(只读,不写) | +| → **auto-maintain** | dream 写完触发 D3 inline 检测;D3 过载 → enqueue split job(maintain 异步消费);写入并发由 CAS 协议(`auto_maintain_design.md` §5)保护 | +| → **auto-link** | dream 写完 enqueue auto-link L1(背景实体识别 + wikilink 写回);走同一 CAS 队列(`auto_link_design.md` §1.2) | + +**关键边界**:dream 不写 daily / resource(I-2 / I-3);只写 digest 节点 body(自身 subject)。 + +--- + +## 6. 下一步 + +本文档覆盖 dream 模型(桶 / 节点 / 边 / 演化)。组织端实现清单(M split / D 检测 / CAS 框架)见 `auto_maintain_design.md` §10。 + +1. **dream step 实现** —— scope → 抽取 → SearchStep 召回 → LLM 终判 → CAS 写入 + E-1 守恒校验 +2. **rename 路径封装** —— `wikilink_handler.retarget_links(old, new)` 已就绪;封装为单步 step,无 alias 表,无透明展开 +3. **bucket 集合配置** —— `vault.yaml` schema / 默认桶模板 / `unknown` 兜底 / `_buckets.md` 视图生成 +4. **边守恒校验工具** —— `extract_links` 已就绪;新增 outbound diff 比较器 + LLM 重试编排 + ConservationViolation audit 事件 +5. **provenance prompt 规范** —— dream 引导 LLM 写 `[[daily/...]]` / `[[resource/...]]` 实现进入 `reme4/steps/jobs/` 与 `reme4/file_graph/` 时,本文档与 `auto_memory_design.md` / `auto_maintain_design.md` / `auto_link_design.md` 共同作为契约依据。 diff --git a/docs4/auto_link_design.md b/docs4/auto_link_design.md index a4ec3dbc..4ff28bbd 100644 --- a/docs4/auto_link_design.md +++ b/docs4/auto_link_design.md @@ -5,8 +5,8 @@ > 配套阅读: > - `structure.md` §1.2(三层数据视角)/ §4(retrieve 三种问法) > - `auto_memory_design.md`:auto-link 可反向扫 daily event,补实体 wikilink(daily → digest) -> - `auto_dream_design.md`:wikilink 模型(§1.4 边语法 / §1.5 演化 / §1.5.5 边守恒 E-1 / E-2 / E-3);auto-link 借这套基础设施 -> - `auto_maintain_design.md`:CAS 写入协议(§5);auto-link L1 写回与 dream G\* / maintain split 三方共用同一套 CAS +> - `auto_dream_design.md`:wikilink 模型(§3 边语法 / §4 演化 / §4.4 边守恒 E-1 / E-2 / E-3);auto-link 借这套基础设施 +> - `auto_maintain_design.md`:CAS 写入协议(§5);auto-link L1 写回与 dream / maintain split 三方共用同一套 CAS > > **三层对应**:reme 服务整体三层 —— auto-memory / auto-dream / **auto-link(本文档)**。auto-link 是图关系的**后置增强** —— 在已落地的 vault 上做实体识别 + wikilink 写回,补足 content link(写记忆时由 LLM 直接产生的 `[[...]]`)在长 tail 隐含关系上的盲区。 > @@ -16,11 +16,11 @@ ## 0. 问题陈述 -content link(`auto_dream_design.md` G\* / split 写入时由 LLM inline 产生的 `[[...]]`)解决了"写记忆时显式的关系"。但有一类关系不会在 inline 写入时自然涌现,需要后台扫描已写入的 vault 才能识别: +content link(`auto_dream_design.md` dream / split 写入时由 LLM inline 产生的 `[[...]]`)解决了"写记忆时显式的关系"。但有一类关系不会在 inline 写入时自然涌现,需要后台扫描已写入的 vault 才能识别: -1. **历史 body 的实体未链接** —— G\* update 时 LLM 关注新材料融入,可能忽略已有 body 中某个未链接的实体(例如 body 提到 "JWT" 但没写 `[[digest/auth/jwt-overview.md]]`) +1. **历史 body 的实体未链接** —— dream update 时 LLM 关注新材料融入,可能忽略已有 body 中某个未链接的实体(例如 body 提到 "JWT" 但没写 `[[digest/auth/jwt-overview.md]]`) 2. **跨节点 / 跨桶的隐含关联** —— 节点 A 提到 "rate limit",但 `digest/api/rate-limit.md` 是后来才被 split 创建 → A 写入时没机会建立这条边 -3. **同主题未连 / 同概念重复** —— G\* 漏判去重把同概念建成两个节点;或两个主题相关但 0 链接的节点彼此不知晓 +3. **同主题未连 / 同概念重复** —— dream 漏判去重把同概念建成两个节点;或两个主题相关但 0 链接的节点彼此不知晓 auto-link 承担这部分:**后台扫描已写入节点 → 实体识别 / 候选挖掘 → wikilink 写回 body**。 @@ -32,7 +32,7 @@ auto-link 承担这部分:**后台扫描已写入节点 → 实体识别 / 候 | 维度 | content link(在 dream) | auto-link(本文档) | |---|---|---| -| 何时产生 | 写记忆 inline:G\* update / M split prompt | 后台扫描:离线 / 周期 / 触发后异步 | +| 何时产生 | 写记忆 inline:dream update / M split prompt | 后台扫描:离线 / 周期 / 触发后异步 | | 由谁产生 | LLM 在 dream 写入流中顺手写出 | LLM 在 auto-link 扫描流中识别后写出 | | 输入 | 新材料 + 召回候选节点 | 已写入 body + 全 vault 索引 | | 改 body | 是(重写整段 body) | 是(纯 additive 插入 wikilink,不改文字) | @@ -50,16 +50,16 @@ After: "[[digest/auth/jwt-rotation.md|JWT 轮换]]的核心是[[digest/auth/jw | 维度 | 决策 | |---|---| -| **alias 必须保留原文** | `[[path.md\|<原文>]]` 形态;原文一字不改 —— 守住"不改写其它节点正文" (`auto_dream_design.md` §1.5.4 F-2) 的精神 | +| **alias 必须保留原文** | `[[path.md\|<原文>]]` 形态;原文一字不改 —— 守住"不改写其它节点正文" (`auto_dream_design.md` §4.3 F-2) 的精神 | | **predicate 默认为空** | auto-link 默认产生无谓词 wikilink;升 typed link 走 L3(详 §1.3) | -| **不引入 anchor** | 与 dream 一致(`auto_dream_design.md` §1.4 / §3.13);target 永远是节点路径 | +| **不引入 anchor** | 与 dream 一致(`auto_dream_design.md` §3);target 永远是节点路径 | | **CAS 写入** | 完全复用 `auto_maintain_design.md` §5 的 read-stamp + CAS-write 协议(冲突重做 ≤ 3 次) | | **E-1 守恒** | 纯 additive:`new outbound = old outbound ∪ {new wikilinks}`;`new ⊇ old` 天然满足,守恒校验默认通过 | | **rollback** | 若 auto-link 误插入(例如 entity mention 是同名歧义),走标准 edit 或 retarget 撤销;auto-link 不维护"我插过哪些"audit log(留给 SDK 决定) | **为什么是 additive 而不是重写**: - additive = 0 文字风险(原文不变,只在原 mention 周围加 `[[ | ]]` 包装) -- 重写 = 触发完整 E-1 守恒校验 + LLM 重写整段语义守恒 prompt + 多次 LLM 调用 = 跟 G\* update 重复 +- 重写 = 触发完整 E-1 守恒校验 + LLM 重写整段语义守恒 prompt + 多次 LLM 调用 = 跟 dream update 重复 - additive 失败可见:产生坏 wikilink 时,人/agent 直接编辑 body 修就行 ### 1.3 候选挖掘类型(L1-L4) @@ -69,32 +69,32 @@ After: "[[digest/auth/jwt-rotation.md|JWT 轮换]]的核心是[[digest/auth/jw | **L1** | **实体识别**(主路径) | 扫 body,识别已是 digest 节点的实体名(模糊匹配 + 语义召回);未被 wikilink 化的 mention → 加 `[[path.md\|]]` | additive wikilink 插入 | | **L2** | **同主题未连**(旧 D7) | 两个 digest 节点谈相关主题但 0 wikilink → 候选 add link;LLM 判后在 body 末尾追加一句引用 | additive(在合适位置 / 节末追加 `参见 [[other.md\|other]]`)| | **L3** | **隐含 predicate 推导** | 已有 `[[A]]` 但 LLM 可推断关系类型(`is_a` / `causes` / `extends` / ...)→ 升级为 typed link | 改 `[[A]]` → `is_a:: [[A]]`(predicate 升降级走显式 audit,详 §2.1)| -| **L4** | **重复语义检测**(旧 D8) | 两个节点描述同一概念但被独立 create(G\* 漏判去重)→ 候选 merge | **不写回**;产报告 + 提示人/agent 触发 G\* update 路径手工合并 | +| **L4** | **重复语义检测**(旧 D8) | 两个节点描述同一概念但被独立 create(dream 漏判去重)→ 候选 merge | **不写回**;产报告 + 提示人/agent 触发 dream update 路径手工合并 | **L1 是主路径** —— 它是 auto-link 最核心、最频繁、最高 ROI 的操作:每个 digest 节点写完后,后台扫一遍 body,找未链接的已知实体,additive 加 wikilink。 **L2-L3 是辅助** —— 周期扫,产候选,LLM 终判,写回部分(L2 节末追加 / L3 升 predicate)。 -**L4 不写回** —— 节点合并是结构改动,影响 E-1 守恒边界 + inbound 链路 + provenance 链路,不适合自动写;auto-link 只产报告,人/agent 决定走 G\* update 路径解决。 +**L4 不写回** —— 节点合并是结构改动,影响 E-1 守恒边界 + inbound 链路 + provenance 链路,不适合自动写;auto-link 只产报告,人/agent 决定走 dream update 路径解决。 ### 1.4 触发节奏 | 模式 | 何时 | 适用 | |---|---|---| -| **inline post-write**(默认) | 每次 G\* update / M split 写完 body → enqueue auto-link L1 job(异步,FIFO,CAS 保护)| L1 实体识别;反应即时,与 D3 写后检测同节奏 | +| **inline post-write**(默认) | 每次 dream update / M split 写完 body → enqueue auto-link L1 job(异步,FIFO,CAS 保护)| L1 实体识别;反应即时,与 D3 写后检测同节奏 | | **周期 batch**(可选)| cron(daily / weekly)扫全 vault | L2 / L3 候选挖掘;成本可控 | | **手动触发** | SDK / 人显式调用 | 全量重扫 / 修复 | **L1 inline 的必要性**:新 split 出的 child 节点立即被既有 body 引用(用 wikilink 而非纯 mention)的关键 = 写入即扫描;不 inline 会让"刚创建的 child 节点"在很长时间内只有 split parent 一个 inbound,中心性失真。 **已排除**: -- inline 时同步 auto-link(阻塞 G\* return)—— 时延不可接受;auto-link 始终异步 +- inline 时同步 auto-link(阻塞 dream return)—— 时延不可接受;auto-link 始终异步 - 所有 L\* 都 inline —— L2-L3 候选挖掘 RTL 跨节点,成本高,只适合 batch - 全 cron 唯一触发 —— L1 滞后过久,新节点孤岛 ### 1.5 中心性算法(retrieve 加权依赖) -retrieve 时节点权重 = base × intent 调节 × **中心性增益**(详 `auto_dream_design.md` §1.5.8)。中心性需要 auto-link 这一层提供 —— content link 给底子,auto-link 补 long tail,二者合起来才是完整的图。 +retrieve 时节点权重 = base × intent 调节 × **中心性增益**(详 `auto_dream_design.md` §5)。中心性需要 auto-link 这一层提供 —— content link 给底子,auto-link 补 long tail,二者合起来才是完整的图。 | 选项 | 优点 | 缺点 | |---|---|---| @@ -128,7 +128,7 @@ diff: missing = {(A, None)}; added = {(A, "is_a")} **决策方向**: - L3 写入必须打 audit flag(消费层意图:升级 predicate,允许 drop + add 同时发生) - audit flag 由 reme4 step 暴露(`maintainer_step(action="predicate_upgrade", from=..., to=...)`),不放在普通 write 路径 -- 普通 G\* / auto-link L1 写入永远不带 audit flag,守恒校验照常严格 +- 普通 dream / auto-link L1 写入永远不带 audit flag,守恒校验照常严格 详细 audit flag 接口形态留到 SDK 阶段。 @@ -143,11 +143,11 @@ L1 扫 body 找 "JWT" 这个 mention,vault 中有 `digest/auth/jwt-overview.md` **首版**:LLM 上下文判(每个候选 candidate 提供 description / 周围若干节点 summary,LLM 选择 top-1 或 drop);成本可接受(扫描已是离线 batch)。 -### 2.3 auto-link 写回与 G\* / split 的并发 +### 2.3 auto-link 写回与 dream / split 的并发 auto-link 写 body 走 §1.2 CAS,但有特殊情况: -- 同节点同时被 G\* update 与 auto-link L1 写入 → CAS 协议自动序列化 (`auto_maintain_design.md` §5):后到者重做 -- auto-link L1 写完后立刻被 G\* update 覆盖(G\* 重写 body) → 看 G\* prompt 是否守住 auto-link 加的 wikilink(E-1 强守恒 → 守住) +- 同节点同时被 dream update 与 auto-link L1 写入 → CAS 协议自动序列化 (`auto_maintain_design.md` §5):后到者重做 +- auto-link L1 写完后立刻被 dream update 覆盖(dream 重写 body) → 看 dream prompt 是否守住 auto-link 加的 wikilink(E-1 强守恒 → 守住) - auto-link L1 与 D3 派发的 split job 同节点并发 → split 先到 / 后到都不影响最终拓扑(split 把 body 拆成 parent + children,auto-link 加的 wikilink 跟着对应内容段自然分配到 parent / child) **结论**:CAS + E-1 + E-2 守恒已覆盖所有并发场景,auto-link 不需要新协调机制。 @@ -186,21 +186,21 @@ L1 实体识别可选: | 引用 | 来源 | |---|---| -| wikilink 基础语法(`[[path.md\|alias]]` / predicate) | `auto_dream_design.md` §1.4 | -| 节点 / 边模型 | `auto_dream_design.md` §1.5 / §1.5.1 | -| F-invariants(F-1..F-11)| `auto_dream_design.md` §1.5.4 | -| 边守恒 E-1 / E-2 / E-3 | `auto_dream_design.md` §1.5.5 | -| 路径即 ID / rename | `auto_dream_design.md` §1.3 | +| wikilink 基础语法(`[[path.md\|alias]]` / predicate) | `auto_dream_design.md` §3 | +| 节点 / 边模型 | `auto_dream_design.md` §4 / §2 / §3 | +| F-invariants(F-1..F-11)| `auto_dream_design.md` §4.3 | +| 边守恒 E-1 / E-2 / E-3 | `auto_dream_design.md` §4.4 | +| 路径即 ID / rename | `auto_dream_design.md` §2 | | CAS 写入协议 | `auto_maintain_design.md` §5 | -| anchor 不引入 | `auto_dream_design.md` §1.4 / §3.13 | -| SearchStep 召回 | `auto_dream_design.md` §3.10 | +| anchor 不引入 | `auto_dream_design.md` §3 | +| SearchStep 召回 | `auto_dream_design.md` §4.2 | --- ## 5. 下一步 1. **L1 实体识别 step 实现** —— 字符串匹配 + 语义召回 + LLM ambiguity 终判 + additive wikilink 写回(§1.2 / §1.3) -2. **inline post-write trigger 接入** —— G\* update / M split CAS 写入成功后 enqueue auto-link L1 job(§1.4) +2. **inline post-write trigger 接入** —— dream update / M split CAS 写入成功后 enqueue auto-link L1 job(§1.4) 3. **L2 / L3 周期 batch 框架** —— cron(daily / weekly)+ 候选挖掘 prompt + 写回路径(§1.3) 4. **L3 audit flag 接口** —— `maintainer_step` 提供 `predicate_upgrade` 操作,带 audit context 走特殊守恒规则(§2.1) 5. **中心性 retrieve 增益** —— file_graph inbound count → retrieve 加权乘子(§1.5) diff --git a/docs4/auto_maintain_design.md b/docs4/auto_maintain_design.md index 19b4fb52..5f6f923a 100644 --- a/docs4/auto_maintain_design.md +++ b/docs4/auto_maintain_design.md @@ -4,7 +4,7 @@ > > 配套阅读: > - `structure.md` §3.6(maintain 动作语义)/ §7.3(maintainer 模块) -> - `auto_dream_design.md`:节点 + 边模型(§1.1-1.5)/ F-invariants(§1.5.4)/ 边守恒 E-1/E-2/E-3(§1.5.5)/ G\* 操作(§2.1)—— maintain 复用这套底层模型 +> - `auto_dream_design.md`:节点 + 边模型(§1-§4)/ F-invariants(§4.3)/ 边守恒 E-1/E-2/E-3(§4.4)/ dream 操作(§4.2)—— maintain 复用这套底层模型 > - `auto_link_design.md`:auto-link 写回也走本文档的 CAS 协议(§5) > - `auto_memory_design.md`:auto-memory 不直接复用 maintain,但事件级"拆"与节点级 split 在概念上同构(都把过载粒度切小) > @@ -13,13 +13,13 @@ > **核心立场**: > - **maintain 与 dream 同 pace**(idle background)、同模型(节点 + 边 / 守恒规则),但**语义边界不同**:dream 是 compose(资料 → digest),maintain 是 reorganize(digest → digest) > - **maintain 只做 split**,不做 merge / dissolve / re-edge / unify;过载就拆,其它跨节点重组留给消费层 / 人工 -> - **CAS 写入协议是基础设施**,被 dream G\* / maintain split / auto-link L1 共用,统一编排在本文档(§5) +> - **CAS 写入协议是基础设施**,被 dream / maintain split / auto-link L1 共用,统一编排在本文档(§5) --- ## 0. 问题陈述 -dream 模型(`auto_dream_design.md` §1.5)规定 digest 的演化只做两件事:G\* create_or_update(入流型,新材料融入)+ M split(后台,过载就拆)。dream 文档负责 G\* 与节点 / 边模型;**本文档负责 M split 与运行时机制**(D 检测 / 触发模型 / 写入并发协议)。 +dream 模型(`auto_dream_design.md` §4)规定 digest 的演化只做两件事:dream create_or_update(入流型,新材料融入)+ M split(后台,过载就拆)。dream 文档负责 dream 与节点 / 边模型;**本文档负责 M split 与运行时机制**(D 检测 / 触发模型 / 写入并发协议)。 | 输入 | 输出 | |---|---| @@ -31,7 +31,7 @@ dream 模型(`auto_dream_design.md` §1.5)规定 digest 的演化只做两件事 3. **不引入新基础设施** —— 复用 dream 的节点 + 边模型 / 守恒规则;CAS 写协议自洽 **显式排除**: -- ❌ merge / dissolve / re-edge / unify —— 跨节点重组不做(简化模型;同概念二次进入靠 G\* update) +- ❌ merge / dissolve / re-edge / unify —— 跨节点重组不做(简化模型;同概念二次进入靠 dream update) - ❌ 改其它节点正文 —— split 只改 parent body(重写为 overview)+ 创建 children body - ❌ 重建 inbound —— split 时 inbound 一律不动(F-10) @@ -44,11 +44,11 @@ dream 模型(`auto_dream_design.md` §1.5)规定 digest 的演化只做两件事 | **M split** | 节点过载 → LLM 拆成 parent overview + N 个 children;parent 文件原地保留,children 是新文件;children 加 `[[parent]]` 反向链接;inbound 边不动 | 形状匹配(粒度对齐)+ 任意尺度(涌现层级) | D3 过载 | parent 0 拓扑改;新 children 节点 + 各自加 `[[parent]]` 出边 | 创建 N 个 children 文件;parent 文件原地 | parent body 重写为 overview;children 各自有新 body | **关键边界**: -- **M split 改两类 body**:parent body(重写为 overview)+ N 个新 children body;不改任何**其它**节点(`auto_dream_design.md` §1.5.4 F-2) -- **inbound 不重定向** —— 外部对 parent 的 wikilink 全部保留指 parent;后续 G\* 进入时若 LLM 觉得 child 粒度更合适,直接加新边到 child 即可(F-10) +- **M split 改两类 body**:parent body(重写为 overview)+ N 个新 children body;不改任何**其它**节点(`auto_dream_design.md` §4.3 F-2) +- **inbound 不重定向** —— 外部对 parent 的 wikilink 全部保留指 parent;后续 dream 进入时若 LLM 觉得 child 粒度更合适,直接加新边到 child 即可(F-10) - **没有 dissolve 操作** —— children 长期空也不主动删;消费层 / 人工显式介入 -- **没有 merge / re-edge / unify** —— 跨节点重组不做;同概念二次进入靠 G\* update;错桶节点不主动 move(若严重,人工介入) -- **边守恒** —— split 写新 parent body + N children body 前,机械对比 outbound:`(parent_new ∪ ∪children_outbound) ⊇ parent_old`;失败 → LLM 重试或拒写(F-11 / E-2,详 `auto_dream_design.md` §1.5.5) +- **没有 merge / re-edge / unify** —— 跨节点重组不做;同概念二次进入靠 dream update;错桶节点不主动 move(若严重,人工介入) +- **边守恒** —— split 写新 parent body + N children body 前,机械对比 outbound:`(parent_new ∪ ∪children_outbound) ⊇ parent_old`;失败 → LLM 重试或拒写(F-11 / E-2,详 `auto_dream_design.md` §4.4) --- @@ -60,14 +60,14 @@ dream 模型(`auto_dream_design.md` §1.5)规定 digest 的演化只做两件事 | **D3** | 过载节点(token 阈值 → LLM 判离散度) | 形状匹配(粒度) | maintainer(M split) | | **D10** | provenance 断裂(digest 节点反指的 daily/resource 不可达) | 任意尺度(跨层不变量) | 严重告警(I 不变量违反) | -**触发模型**:**写后立即** —— G\* / split 写完 body inline 检测;无后台 watcher / 无周期 tick / 无 dirty 队列(详 §4)。D1 / D10 是 wikilink 断链的子集,跟 file_graph 链路一起在写时检测。 +**触发模型**:**写后立即** —— dream / split 写完 body inline 检测;无后台 watcher / 无周期 tick / 无 dirty 队列(详 §4)。D1 / D10 是 wikilink 断链的子集,跟 file_graph 链路一起在写时检测。 > **简化模型砍掉的信号**: > - **D2 隔离 / D4 过疏 / D5 高入度 / D5b 低入度摘要 / D6 slug 冲突 / D7 相似未链 / D8 重复语义 / D9 邻居异质** —— 全部 DROPPED > - 旧 D5 高入度涌现 → 由 split 副产品(parent + children)等价覆盖;触发源换成节点过载(D3) -> - 旧 D6 slug 冲突 → 路径即 ID 后,同 bucket 内文件名冲突由文件系统层断言(写入即拒),不需要独立信号(详 `auto_dream_design.md` §1.3) +> - 旧 D6 slug 冲突 → 路径即 ID 后,同 bucket 内文件名冲突由文件系统层断言(写入即拒),不需要独立信号(详 `auto_dream_design.md` §2) > - 旧 D7 / D8 → 简化模型不做 link / merge 提议;若 vault 累积明显的同概念重复,由 `auto_link_design.md` §1.3 L4 离线 audit 工具产报告 -> - 旧 D9 邻居异质 → 简化模型不做跨桶 move;桶选择只在 G4 一次性决定,后续不重排 +> - 旧 D9 邻居异质 → 简化模型不做跨桶 move;桶选择只在 dream 桶决策一次性决定,后续不重排 > > **D3 过载的判据**:token 阈值机械检查 + LLM 判离散度;**写后立即 inline**。阈值见 §3,触发模型见 §4。 @@ -89,10 +89,10 @@ D3 阈值作为 `vault.yaml` 配置项(opinionated default,reme 核心提供机 ## 4. D3 触发模型(已收敛) -**决策**:**写后立即检测,无 watcher 抽象,无 batch 窗口** —— 每次 G\* update / split 写 body 成功后,**inline** 在同一 job 内跑 D3:token 阈值 + LLM 离散度判定 → 必要时 enqueue split job(异步,走 §5 CAS 队列)。 +**决策**:**写后立即检测,无 watcher 抽象,无 batch 窗口** —— 每次 dream update / split 写 body 成功后,**inline** 在同一 job 内跑 D3:token 阈值 + LLM 离散度判定 → 必要时 enqueue split job(异步,走 §5 CAS 队列)。 ``` -G* / split 写 body 成功(CAS 通过) +dream / split 写 body 成功(CAS 通过) └─ if len(body) > T: └─ LLM 判离散度 └─ if is_overloaded: @@ -110,15 +110,15 @@ G* / split 写 body 成功(CAS 通过) **已排除**:定时 cron tick(静止 vault 浪费扫描)/ ingest-after batch(引入 dirty 集合)/ 独立 L2 watcher worker(多余部署层)。 -**演进路径(M1+)**:若 inline LLM 阻塞 G\* 时延成问题 → D3 改为 fire-and-forget enqueue;若同节点重复触发 LLM 成本高 → 加节点级 body hash 缓存。 +**演进路径(M1+)**:若 inline LLM 阻塞 dream 时延成问题 → D3 改为 fire-and-forget enqueue;若同节点重复触发 LLM 成本高 → 加节点级 body hash 缓存。 --- ## 5. CAS 写入协议(共享基础设施) -**位置说明**:CAS 是 G\* update(`auto_dream_design.md` §2.1)、M split(本文档 §1)、auto-link L1 写回(`auto_link_design.md` §1.2)**三方共用**的写入协议。归在本文档是因为 maintain 是 digest 的"组织 / 运行时"端,运行时机制(检测 / 触发 / 写入)集中在一处方便对照。 +**位置说明**:CAS 是 dream update(`auto_dream_design.md` §4.2)、M split(本文档 §1)、auto-link L1 写回(`auto_link_design.md` §1.2)**三方共用**的写入协议。归在本文档是因为 maintain 是 digest 的"组织 / 运行时"端,运行时机制(检测 / 触发 / 写入)集中在一处方便对照。 -**决策**:**并行决策 + 乐观冲突重做(CAS)** —— 所有 G\* / split / auto-link L1 决策并发跑,写入前用 body 版本戳(hash / mtime)做 CAS 比对;变了就丢弃 planned body 重做。无锁,无 ingest 级互斥。冲突率低 + E-1 / E-2 守恒校验顺手承担 race 兜底,无需新基础设施。 +**决策**:**并行决策 + 乐观冲突重做(CAS)** —— 所有 dream / split / auto-link L1 决策并发跑,写入前用 body 版本戳(hash / mtime)做 CAS 比对;变了就丢弃 planned body 重做。无锁,无 ingest 级互斥。冲突率低 + E-1 / E-2 守恒校验顺手承担 race 兜底,无需新基础设施。 **协议(单个写入调用)**: 1. **读 + 记戳**:读 subject body → `version_stamp = sha256(body) | mtime` @@ -126,12 +126,12 @@ G* / split 写 body 成功(CAS 通过) 3. **CAS 写入**:重读 body 比 version_stamp - **未变**:跑 E-1 / E-2 守恒校验 → 通过则 atomic write(write-temp + rename)→ done - **已变**:丢弃 planned new_body,带最新 body 重走 step 1 -4. **守恒校验失败**:走 `auto_dream_design.md` §1.5.5 既有重试路径(LLM 重试一次,二次失败拒写 + audit) +4. **守恒校验失败**:走 `auto_dream_design.md` §4.4 既有重试路径(LLM 重试一次,二次失败拒写 + audit) 5. **重做次数上限**:CAS-冲突重做最多 3 次;超出 → 跳过候选 + audit log(避免活锁) -**create 路径 race**:两个 G\* 都决定 `create digest/auth/jwt-rotation.md` → atomic create(`O_CREAT | O_EXCL`)只让一个赢;输者拿 EEXIST → 重走 step 1(此时大概率改判 update)。 +**create 路径 race**:两个 dream 都决定 `create digest/auth/jwt-rotation.md` → atomic create(`O_CREAT | O_EXCL`)只让一个赢;输者拿 EEXIST → 重走 step 1(此时大概率改判 update)。 -**适用范围**(全部走同一套 CAS):同 ingest 内 N 个候选并发 / 跨 ingest job 并发 / 后台 split 与前台 G\* 命中同节点(split 同样走 CAS)/ auto-link 写回(`auto_link_design.md` §1.2)。 +**适用范围**(全部走同一套 CAS):同 ingest 内 N 个候选并发 / 跨 ingest job 并发 / 后台 split 与前台 dream 命中同节点(split 同样走 CAS)/ auto-link 写回(`auto_link_design.md` §1.2)。 **不解决的**:高冲突 workload(同概念被反复 ingest)→ 重做上限触发后 audit;跨进程并发(多 reme 实例同 vault)→ 不在 M0,需 fs lock(M1+)。 @@ -139,18 +139,18 @@ G* / split 写 body 成功(CAS 通过) ## 6. split 时的 provenance 处理(已收敛) -**坍缩到 E-2 合计守恒** —— provenance 是 body 内联 wikilink(`auto_dream_design.md` §3.9),split 时跟其它 body 边完全同形:LLM 把 parent body 拆成 parent overview + N children,provenance wikilink 跟着对应内容段自然分配;机械层 outbound 合计守恒校验保证 `(parent_new ∪ ∪children_outbound) ⊇ parent_old`,旧 provenance 不可能丢。无需专门的 provenance 分配逻辑或"全部复制到 child / parent 保留全量"等特殊策略 —— LLM 按"哪个 child 谈到了哪段上游就带走哪条 provenance"自然处理。 +**坍缩到 E-2 合计守恒** —— provenance 是 body 内联 wikilink(`auto_dream_design.md` §4.2),split 时跟其它 body 边完全同形:LLM 把 parent body 拆成 parent overview + N children,provenance wikilink 跟着对应内容段自然分配;机械层 outbound 合计守恒校验保证 `(parent_new ∪ ∪children_outbound) ⊇ parent_old`,旧 provenance 不可能丢。无需专门的 provenance 分配逻辑或"全部复制到 child / parent 保留全量"等特殊策略 —— LLM 按"哪个 child 谈到了哪段上游就带走哪条 provenance"自然处理。 --- -## 7. G\* / split / auto-link L1 时序(已收敛) +## 7. dream / split / auto-link L1 时序(已收敛) -时序由 §4 / §5 与 `auto_dream_design.md` §3.10 共同规定,这里给最小汇总: +时序由 §4 / §5 与 `auto_dream_design.md` §4.2 共同规定,这里给最小汇总: -- **G\* 调用本身同步** —— material 进来就走 G\* 决策(召回 + LLM 终判)+ CAS 写入(§5) -- **D3 检测 inline** —— G\* / split 写完 body 顺手跑 token 阈值 + LLM 判离散度(§4),无 tick / batch / watcher -- **split 异步** —— D3 触发后 enqueue split job 进 §5 CAS 队列,跟其它 ingest / split job FIFO 共享,异步消费;**不阻塞 G\* return** -- **auto-link L1 异步** —— 写入成功后 enqueue auto-link L1 job(`auto_link_design.md` §1.4),与 split job 同 CAS 队列、FIFO 共享;不阻塞 G\* return +- **dream 调用本身同步** —— material 进来就走 dream 决策(召回 + LLM 终判)+ CAS 写入(§5) +- **D3 检测 inline** —— dream / split 写完 body 顺手跑 token 阈值 + LLM 判离散度(§4),无 tick / batch / watcher +- **split 异步** —— D3 触发后 enqueue split job 进 §5 CAS 队列,跟其它 ingest / split job FIFO 共享,异步消费;**不阻塞 dream return** +- **auto-link L1 异步** —— 写入成功后 enqueue auto-link L1 job(`auto_link_design.md` §1.4),与 split job 同 CAS 队列、FIFO 共享;不阻塞 dream return 检测延迟 ≈ 0(inline);split 执行延迟 ≈ 队列等待时间(typically 数秒~数十秒);新建 / update 节点不必等 split 完成,体验连续。 @@ -168,14 +168,14 @@ G* / split 写 body 成功(CAS 通过) | 引用 | 来源 | |---|---| -| wikilink 基础语法(`[[path.md\|alias]]` / predicate) | `auto_dream_design.md` §1.4 | -| 节点 / 边模型 | `auto_dream_design.md` §1.5 / §1.5.1 | -| F-invariants(F-1..F-11) | `auto_dream_design.md` §1.5.4 | -| 边守恒 E-1 / E-2 / E-3 | `auto_dream_design.md` §1.5.5 | -| 路径即 ID / rename | `auto_dream_design.md` §1.3 | -| anchor 不引入 | `auto_dream_design.md` §1.4 / §3.13 | -| provenance 载体形态 | `auto_dream_design.md` §3.9 | -| G\* 行为 | `auto_dream_design.md` §2.1 | +| wikilink 基础语法(`[[path.md\|alias]]` / predicate) | `auto_dream_design.md` §3 | +| 节点 / 边模型 | `auto_dream_design.md` §4 / §2 / §3 | +| F-invariants(F-1..F-11) | `auto_dream_design.md` §4.3 | +| 边守恒 E-1 / E-2 / E-3 | `auto_dream_design.md` §4.4 | +| 路径即 ID / rename | `auto_dream_design.md` §2 | +| anchor 不引入 | `auto_dream_design.md` §3 | +| provenance 载体形态 | `auto_dream_design.md` §4.2 | +| dream 行为 | `auto_dream_design.md` §4.2 | --- @@ -184,7 +184,7 @@ G* / split 写 body 成功(CAS 通过) 1. **M split step 实现** —— D3 触发 → 候选 → split prompt → 写入 + E-2 守恒(§1 / §4 / §5) 2. **D 检测信号实现清单**(D1 断链 / D3 写后 inline / D10 provenance 哪些已就绪 / 缺哪些)—— §2 3. **D3 阈值配置**(`vault.yaml` 中 D3 token / 离散度阈值)—— §3 -4. **CAS 写入框架** —— per-path body version_stamp + CAS 写入 + EEXIST create race + 重做上限 + audit;对外暴露给 dream G\* / auto-link L1 复用 —— §5 +4. **CAS 写入框架** —— per-path body version_stamp + CAS 写入 + EEXIST create race + 重做上限 + audit;对外暴露给 dream / auto-link L1 复用 —— §5 5. **后门 SDK 接口形态**(暂缓 M1+)—— §8 实现进入 `reme4/steps/jobs/` 与 `reme4/file_graph/` 时,本文档与 `auto_dream_design.md` / `auto_link_design.md` 共同作为契约依据。 diff --git a/docs4/auto_memory_design.md b/docs4/auto_memory_design.md index ac594306..7747cb89 100644 --- a/docs4/auto_memory_design.md +++ b/docs4/auto_memory_design.md @@ -4,7 +4,7 @@ > > 配套阅读: > - `structure.md` §2.1-2.2(daily 层定位)/ §3.4(sync 动作语义)/ §7.1(synchronizer 模块) -> - `auto_dream_design.md`:auto-memory 产物如何被 dream 消化(G\* 读 daily 作为入流之一) +> - `auto_dream_design.md`:auto-memory 产物如何被 dream 消化(dream 读 daily 作为入流之一) > - `auto_maintain_design.md`:digest 的组织端 / CAS 写入协议;auto-memory 不直接复用,但事件级"拆"与节点级 split 在概念上同构(都把过载粒度切小) > - `auto_link_design.md`:auto-link 可反向扫 daily 事件,补充实体 wikilink(daily → digest) > @@ -179,7 +179,7 @@ INHERIT 行为细节(扫描窗口、predecessor 是否关闭、Plan/Objective | ← **notify** | 接收 notify payload 作为新 event cue;不强制响应,不强制 wikilink 引 | | ← **resource** | 只读(通过 wikilink 引);不写 | | → **daily** | **唯一写者**(I-2);写 event folder + 主索引 | -| → **auto-dream** | dream 的 G\* 读 daily 作为入流(`auto_dream_design.md` §2.1 G1 scope);auto-memory 写完即对 dream 可见(走 L2 索引,有 eventual 窗口) | +| → **auto-dream** | dream 读 daily 作为入流(`auto_dream_design.md` §4.2 dream scope);auto-memory 写完即对 dream 可见(走 L2 索引,有 eventual 窗口) | | → **auto-link** | auto-link 可反向扫 daily event,做实体识别 + wikilink 写回(`auto_link_design.md` §1.3)—— 与 auto-memory 写入不冲突(双方写不同字段段落 / CAS 协议保护)| **关键边界**:auto-memory 是 daily 写入端的**唯一**入口;dream / link 不写 daily 主路径,只通过 auto-link 走 §1.3 写回(read-only audit-then-write,CAS 保护)。 diff --git a/reme4/components/as_llm/__init__.py b/reme4/components/as_llm/__init__.py index 82ae553f..693ebdb8 100644 --- a/reme4/components/as_llm/__init__.py +++ b/reme4/components/as_llm/__init__.py @@ -25,7 +25,13 @@ class OpenAIAsLLM(BaseAsLLM): """OpenAI chat model wrapper.""" async def _start(self) -> None: - self.model = OpenAIChatModel(**self.kwargs) + kwargs = dict(self.kwargs) + base_url = kwargs.pop("base_url", None) + if base_url: + client_kwargs = dict(kwargs.pop("client_kwargs", None) or {}) + client_kwargs.setdefault("base_url", base_url) + kwargs["client_kwargs"] = client_kwargs + self.model = OpenAIChatModel(**kwargs) async def _close(self) -> None: if self.model is not None: diff --git a/reme4/config/default.yaml b/reme4/config/default.yaml index 5583b08e..8f282d2c 100644 --- a/reme4/config/default.yaml +++ b/reme4/config/default.yaml @@ -368,6 +368,41 @@ jobs: steps: - backend: edit_step + dream: + backend: base + description: "Dream: lift atomic units from one daily/resource file into digest/ (LLM)." + parameters: + type: object + properties: + path: + type: string + description: "vault-relative path of one daily-event note or resource file" + hint: + type: string + description: "caller guidance to the dreamer LLM" + default: "" + required: + - path + steps: + - backend: dreamer_step + + dream_today: + backend: base + description: "Dream-today: scan // + // and dream each file (LLM). Subroots resolved from app config." + parameters: + type: object + properties: + date: + type: string + description: "YYYY-MM-DD to scan; defaults to today in the dreamer's timezone" + default: "" + hint: + type: string + description: "caller guidance passed through to each per-file dream" + default: "" + steps: + - backend: cron_dreamer_step + auto_memory: backend: base description: "Auto-memory: record conversation facts into a daily note" @@ -447,3 +482,15 @@ components: embedding_model: "" keyword_index: default file_graph: default + + as_llm: + default: + backend: ${LLM_BACKEND:-openai} + api_key: ${LLM_API_KEY:-} + base_url: ${LLM_BASE_URL:-} + model_name: ${LLM_MODEL_NAME:-} + stream: false + + as_llm_formatter: + default: + backend: ${LLM_BACKEND:-openai} diff --git a/reme4/steps/__init__.py b/reme4/steps/__init__.py index 72fab3b4..db302599 100644 --- a/reme4/steps/__init__.py +++ b/reme4/steps/__init__.py @@ -29,6 +29,11 @@ from .index.traverse import TraverseStep from .index.update_catalog import UpdateCatalogStep from .index.update_index import UpdateIndexStep from .index.watch_changes import WatchChangesStep +from .dream.cron_dreamer import CronDreamer +from .dream.digest_edit import DigestEditStep +from .dream.digest_write import DigestWriteStep +from .dream.dreamer import Dreamer +from .jobs.synchronizer import Synchronizer from .transfer.download import DownloadStep from .transfer.ingest import IngestStep from .transfer.upload import UploadStep @@ -71,6 +76,13 @@ __all__ = [ "UpdateCatalogStep", "UpdateIndexStep", "WatchChangesStep", + # dream + "CronDreamer", + "DigestEditStep", + "DigestWriteStep", + "Dreamer", + # jobs + "Synchronizer", # transfer "DownloadStep", "IngestStep", diff --git a/reme4/steps/base_step.py b/reme4/steps/base_step.py index 65526cc4..d310303a 100644 --- a/reme4/steps/base_step.py +++ b/reme4/steps/base_step.py @@ -130,8 +130,13 @@ class BaseStep(ComponentMixin, ABC): self.context: RuntimeContext | None = None # Load class-level prompts first, then overlay caller-provided overrides. + # Walk MRO in reverse so most-derived class wins; subclasses without their + # own YAML inherit prompts from their parent (e.g. CronDreamer inherits + # dreamer.yaml from Dreamer). self.prompt = PromptHandler(language=self.language) - self.prompt.load_prompt_by_class(self.__class__).load_prompt_dict(prompt_dict) + for cls in reversed(self.__class__.__mro__): + self.prompt.load_prompt_by_class(cls) + self.prompt.load_prompt_dict(prompt_dict) # ----- Component references (resolved lazily on first access) ---------- diff --git a/reme4/steps/dream/__init__.py b/reme4/steps/dream/__init__.py new file mode 100644 index 00000000..0565afe0 --- /dev/null +++ b/reme4/steps/dream/__init__.py @@ -0,0 +1,19 @@ +"""dream — auto-dream pipeline: classify + integrate via constrained digest tools. + +Four steps: + + dreamer — 2-phase ReAct workflow (extract memory sub-units, + then integrate per sub-unit via the digest tools). + cron_dreamer — daily wrapper around dreamer; scans today's + daily/ + resource/ files and runs dream_one on each. + digest_write_step — constrained WriteStep that creates a new + digest//.md. + digest_edit_step — constrained EditStep that find-and-replaces in + an existing digest node, enforcing E-1 edge + conservation. +""" + +from . import cron_dreamer # noqa: F401 -- @R.register("cron_dreamer_step") +from . import digest_edit # noqa: F401 -- @R.register("digest_edit_step") +from . import digest_write # noqa: F401 -- @R.register("digest_write_step") +from . import dreamer # noqa: F401 -- @R.register("dreamer_step") diff --git a/reme4/steps/dream/cron_dreamer.py b/reme4/steps/dream/cron_dreamer.py new file mode 100644 index 00000000..92c189b4 --- /dev/null +++ b/reme4/steps/dream/cron_dreamer.py @@ -0,0 +1,157 @@ +"""``cron_dreamer_step`` — daily-tick wrapper around :class:`Dreamer`. + +Scans today's materials and runs the per-file dream pipeline on each one: + +* ``/.md`` — the day-index rollup (processed first). +* ``//**/*.md`` — per-event notes for the day. +* ``//**/*`` — resources ingested today. + +The per-file logic is inherited from :class:`Dreamer` (via +:meth:`Dreamer.dream_one`); this step just adds the outer loop. + +Cron scheduling itself is out of scope here — this step is the unit of +work executed when a cron fires (or when the operator manually invokes +the ``dream_today`` job). External schedulers (background-job watchers, +cron daemons, etc.) drive when it runs. + +The ``daily_dir`` and ``resource_dir`` subroots are NOT tool params — +they come from ``app_config.daily_dir`` / ``app_config.resource_dir`` +(same convention as the ``daily_*`` steps). + +Inputs (RuntimeContext): + date (str, optional): YYYY-MM-DD to scan. Defaults to today + in the dreamer's timezone. + hint (str, optional): passed through to each per-file dream. + +Output (Response.metadata): :class:`CronDreamResult` JSON. +""" + +from pathlib import Path + +from pydantic import BaseModel, Field + +from .dreamer import Dreamer, DreamResult +from ...components import R + + +class CronDreamResult(BaseModel): + """Aggregated outcome of one cron tick.""" + + date: str = "" + files_scanned: int = 0 + files_dreamed: int = 0 + files_skipped: int = 0 + files_failed: int = 0 + per_file: list[DreamResult] = Field(default_factory=list) + summary: str = "" + + +@R.register("cron_dreamer_step") +class CronDreamer(Dreamer): + """Loop ``daily//`` + ``resource//`` and dream each file.""" + + async def execute(self): + assert self.context is not None + date_input: str = (self.context.get("date", "") or "").strip() + hint: str = (self.context.get("hint", "") or "").strip() + + # daily_dir / resource_dir come from app config — NOT tool params. + # Same convention as daily_create / daily_list / daily_reindex. + # resource_dir may be empty (default) — that just skips the resource scan. + cfg = self.app_context.app_config if self.app_context is not None else None + daily_dir = (cfg.daily_dir if cfg else "") or "daily" + resource_dir = cfg.resource_dir if cfg else "" + + today = date_input or self._now().strftime("%Y-%m-%d") + vault = self._vault_dir() + files = _scan_today_files(vault, today, daily_dir, resource_dir) + + result = CronDreamResult(date=today, files_scanned=len(files)) + self.logger.info( + f"[{self.name}] cron tick date={today} scanned={len(files)} file(s) under " + f"{daily_dir}/{today}/ + {resource_dir}/{today}/", + ) + + for rel_path in files: + try: + dr = await self.dream_one(rel_path, hint) + except Exception as e: # pylint: disable=broad-except + self.logger.error( + f"[{self.name}] dream_one failed on {rel_path}: {type(e).__name__}: {e}", + ) + dr = DreamResult( + path=rel_path, + error=f"{type(e).__name__}: {e}", + ) + result.per_file.append(dr) + if dr.error: + result.files_failed += 1 + elif dr.skipped: + result.files_skipped += 1 + else: + result.files_dreamed += 1 + + result.summary = _render_summary(result) + self.context.response.success = result.files_failed == 0 + self.context.response.answer = result.summary + self.context.response.metadata.update(result.model_dump()) + + +def _scan_today_files( + vault: Path, + today: str, + daily_dir: str, + resource_dir: str, +) -> list[str]: + """Return vault-relative paths of today's daily notes + resource files. + + * ``/.md`` — the day-index file (auto-rebuilt + rollup of all of today's notes). Included first so its day-level + abstractions land before the per-event details. + * ``//**/*.md`` — event notes for the day, + sorted by path. + * ``//**/*`` — any file type ingested under + today's resource folder. Skipped when ``resource_dir`` is empty. + + Results are sorted for deterministic processing order within each + group; the day-index file leads. + """ + out: list[str] = [] + + if daily_dir: + day_index = vault / daily_dir / f"{today}.md" + if day_index.is_file(): + out.append(str(day_index.relative_to(vault))) + daily_root = vault / daily_dir / today + if daily_root.is_dir(): + for md in sorted(daily_root.rglob("*.md")): + if md.is_file(): + out.append(str(md.relative_to(vault))) + + if resource_dir: + resource_root = vault / resource_dir / today + if resource_root.is_dir(): + for f in sorted(p for p in resource_root.rglob("*") if p.is_file()): + out.append(str(f.relative_to(vault))) + + return out + + +def _render_summary(r: CronDreamResult) -> str: + """One-line header + one line per file with its outcome.""" + lines = [ + f"[CronDreamer] date={r.date} scanned={r.files_scanned} " + f"dreamed={r.files_dreamed} skipped={r.files_skipped} failed={r.files_failed}", + ] + for dr in r.per_file: + if dr.error: + status = f"ERROR ({dr.error})" + elif dr.skipped: + status = "SKIP" + else: + status = ( + f"OK (+{len(dr.nodes_created)} created, ~{len(dr.nodes_updated)} updated, " + f"!{len(dr.conservation_violations)} conservation)" + ) + lines.append(f" - {dr.path}: {status}") + return "\n".join(lines) diff --git a/reme4/steps/dream/digest_edit.py b/reme4/steps/dream/digest_edit.py new file mode 100644 index 00000000..80217421 --- /dev/null +++ b/reme4/steps/dream/digest_edit.py @@ -0,0 +1,121 @@ +"""``digest_edit_step`` — constrained EditStep that targets ``digest//.md``. + +Subclasses :class:`EditStep`. On top of the generic find-and-replace, +this step adds: + +* **Path-shape validation** — the target must be ``digest//.md`` + with ```` in the configured set. +* **Existence check** — the file must already exist (use + :class:`DigestWriteStep` for new nodes). +* **E-1 strong edge conservation** — the wikilink set in the file BEFORE + the replacement must be a subset of the wikilink set AFTER. The check + runs as a preflight: the replacement is simulated, links are compared, + and the actual write is delegated to ``super().execute()`` only if + conservation holds. On violation the step returns + ``REJECT_CONSERVATION`` (with the missing edges) and the file on disk + is left untouched. +""" + +from pathlib import Path + +import frontmatter + +from .digest_write import DEFAULT_DIGEST_DIR, _validate_digest_path, bucket_names, normalize_buckets +from ..file_io._file_io import read_file_safe +from ..file_io.edit import EditStep +from ...components import R +from ...utils.wikilink_handler import WikilinkHandler + + +@R.register("digest_edit_step") +class DigestEditStep(EditStep): + """EditStep variant that enforces digest path shape + E-1 edge conservation.""" + + def __init__(self, buckets=None, digest_dir: str | None = None, **kwargs): + super().__init__(**kwargs) + self.buckets = normalize_buckets(buckets) + self.digest_dir = (digest_dir or "").strip() or DEFAULT_DIGEST_DIR + + def _reject(self, message: str, **meta) -> None: + assert self.context is not None + self.context.response.success = False + self.context.response.answer = f"REJECT: {message}" + if meta: + self.context.response.metadata.update(meta) + + async def execute(self): + assert self.context is not None + raw = str(self.context.get("path") or "") + old = self.context.get("old") + new = self.context.get("new") + + err = _validate_digest_path(raw, bucket_names(self.buckets), self.digest_dir) + if err: + self._reject(err) + return None + + abs_path = (Path(self.vault_path) / raw).resolve() + if not abs_path.exists(): + self._reject(f"{raw} does not exist; use digest_write instead") + return None + + # Preflight conservation check: simulate the find-and-replace, compare + # wikilink sets before vs after, refuse the write if any edge is dropped. + # We let super().execute() re-validate `old in body` and surface its + # own error if old is empty or missing. + if old is not None and new is not None and str(old) != "": + preview_text = await _preview_replacement(abs_path, str(old), str(new)) + if preview_text is not None: + missing = _missing_edges( + await _read_text(abs_path), + preview_text, + raw, + ) + if missing: + missing_repr = sorted(f"[[{t}]]" + (f" (predicate={p})" if p else "") for t, p in missing) + self._reject( + ( + f"REJECT_CONSERVATION: replacement drops {len(missing)} edge(s) " + "the old body had. E-1 strong-conservation: every outbound wikilink " + "in the old body MUST appear in the new body. " + f"Missing: {', '.join(missing_repr)}. " + "Adjust `new` to keep the missing links and retry." + ), + conservation_violation={ + "target_path": raw, + "missing": sorted(list(missing)), + }, + ) + return None + + return await super().execute() + + +async def _read_text(abs_path: Path) -> str: + text, _ = await read_file_safe(abs_path) + return text + + +async def _preview_replacement(abs_path: Path, old: str, new: str) -> str | None: + """Return the full file text as it WOULD look after EditStep's replace. + + Mirrors EditStep's body-only replacement (frontmatter is left untouched). + Returns ``None`` if the replacement isn't possible (``old`` not in body) so + the caller can defer the error to super().execute().""" + raw_text, _ = await read_file_safe(abs_path) + post = frontmatter.loads(raw_text) + body = post.content + if old not in body: + return None + post.content = body.replace(old, new) + new_text = frontmatter.dumps(post) if post.metadata else post.content + if not new_text.endswith("\n"): + new_text += "\n" + return new_text + + +def _missing_edges(old_text: str, new_text: str, path: str) -> set[tuple[str, str | None]]: + """Edges present in ``old_text`` but absent from ``new_text`` (E-1 deltas).""" + old_set = {(link.target_path, link.predicate) for link in WikilinkHandler.extract_links(old_text, path)} + new_set = {(link.target_path, link.predicate) for link in WikilinkHandler.extract_links(new_text, path)} + return old_set - new_set diff --git a/reme4/steps/dream/digest_write.py b/reme4/steps/dream/digest_write.py new file mode 100644 index 00000000..935693c3 --- /dev/null +++ b/reme4/steps/dream/digest_write.py @@ -0,0 +1,138 @@ +"""``digest_write_step`` — constrained WriteStep that targets ``digest//.md``. + +Subclasses :class:`WriteStep`. The only thing this step adds on top of the +generic file write is path-shape validation: + +* ``path`` must look like ``digest//.md`` (depth 1, ``.md`` suffix). +* ```` must be one of the configured ``buckets`` (defaults to the + dreamer's :data:`DEFAULT_BUCKETS`). +* The target must NOT already exist — for in-place updates the agent uses + :class:`DigestEditStep` instead. + +Everything else (atomic write semantics, encoding, parent-dir creation, +frontmatter handling) is inherited from :class:`WriteStep`. + +Bucket vocabulary lives in this module (NOT in the dreamer prompt template) +so it can be swapped / extended without touching the prompt. Future direction: +externalize via app-config injection; the ``Dreamer`` constructor already +accepts an override. +""" + +from pathlib import Path + +from ..file_io.write import WriteStep +from ...components import R + + +# Each bucket carries a name (the filesystem folder under ``digest/``) and a +# one-line description that the Phase 2 prompt renders into the bucket-picking +# heuristic block at runtime. +DEFAULT_BUCKETS: tuple[dict[str, str], ...] = ( + { + "name": "concept", + "description": 'definitions, principles, mental models ("what IS X?")', + }, + { + "name": "procedure", + "description": 'steps, methods, recipes ("how do I do X?")', + }, + { + "name": "entity", + "description": "specific named things (person, system, tool, project)", + }, + { + "name": "observation", + "description": 'findings, results, decisions with rationale ("what happened / was decided?")', + }, + { + "name": "preference", + "description": 'user / team / agent collaboration rules ("how does X like to work / what to avoid?")', + }, + { + "name": "unknown", + "description": "fallback when no specialized bucket fits (first-class, not a failure state)", + }, +) + + +def normalize_buckets(buckets) -> tuple[dict[str, str], ...]: + """Normalize a caller-supplied bucket spec into ``tuple[{name, description}, ...]``. + + Accepts ``None`` (→ :data:`DEFAULT_BUCKETS`), tuple/list of dicts (returned + as-is), or tuple/list of strings (legacy — each wrapped with an empty + description). + """ + if not buckets: + return DEFAULT_BUCKETS + first = next(iter(buckets)) + if isinstance(first, dict): + return tuple(buckets) + return tuple({"name": str(b), "description": ""} for b in buckets) + + +def bucket_names(buckets) -> tuple[str, ...]: + """Extract just the name field of each bucket — used for path-membership checks.""" + if not buckets: + return () + first = next(iter(buckets)) + if isinstance(first, dict): + return tuple(b["name"] for b in buckets) + return tuple(buckets) + + +DEFAULT_DIGEST_DIR: str = "digest" + + +@R.register("digest_write_step") +class DigestWriteStep(WriteStep): + """WriteStep variant that enforces the ``//.md`` layout.""" + + def __init__(self, buckets=None, digest_dir: str | None = None, **kwargs): + super().__init__(**kwargs) + self.buckets = normalize_buckets(buckets) + self.digest_dir = (digest_dir or "").strip() or DEFAULT_DIGEST_DIR + + def _reject(self, message: str) -> None: + assert self.context is not None + self.context.response.success = False + self.context.response.answer = f"REJECT: {message}" + + async def execute(self): + assert self.context is not None + raw = str(self.context.get("path") or "") + + err = _validate_digest_path(raw, bucket_names(self.buckets), self.digest_dir) + if err: + self._reject(err) + return None + + abs_path = (Path(self.vault_path) / raw).resolve() + if abs_path.exists(): + self._reject(f"{raw} already exists; use digest_edit instead") + return None + + return await super().execute() + + +def _validate_digest_path( + raw: str, + allowed_names: tuple[str, ...], + digest_dir: str, +) -> str | None: + """Return an error message, or ``None`` if ``raw`` matches the digest shape + and uses one of the allowed bucket names. ``digest_dir`` is the configured + digest root (e.g. ``"digest"``).""" + prefix = f"{digest_dir}/" + expected = f"{digest_dir}//.md" + if not raw.startswith(prefix) or not raw.endswith(".md"): + return f"path must be {expected!r}, got {raw!r}" + parts = raw.split("/") + if len(parts) != 3: + return f"digest is a shallow bucket layout ({expected}, depth 1); got {raw!r}" + bucket = parts[1] + if bucket not in allowed_names: + return ( + f"bucket {bucket!r} not in allowed set {list(allowed_names)}. " + "Use 'unknown' when no specialized bucket fits." + ) + return None diff --git a/reme4/steps/dream/dreamer.py b/reme4/steps/dream/dreamer.py new file mode 100644 index 00000000..4574e3ad --- /dev/null +++ b/reme4/steps/dream/dreamer.py @@ -0,0 +1,639 @@ +"""Dreamer — auto-dream's create_or_update step. + +Reads one daily-event note or resource file at the given vault-relative +``path``, identifies the ABSTRACTIONS the material teaches in Phase 1, +then in Phase 2 makes ONE cognitive write decision (CREATE / UPDATE / +SKIP) per abstraction. See ``docs4/auto_dream_design.md`` for the model +contract (buckets / nodes / edges / evolution) and ``§4.2`` for the +pipeline. + +**Digest is the abstract memory layer** — analogous to a prefrontal +cortex aggregating cognition. Raw details (timestamps, full +procedures, who-said-what, numbers) stay in the material; digest +holds the principle, pattern, or precedent that should survive +once the details fade. Provenance wikilinks (``derived_from::``) +let readers drill back down to the source on demand. + +Pipeline (external loop in Python, two distinct ReAct agent invocations, +**light Phase 1 / heavy Phase 2**): + + execute(): + _extract(material_blob) # 1× ReAct: identify abstractions + # agent calls declare_units([{name, summary}, ...]) + for unit in self._units: # Python loop, K iterations (K = num abstractions) + _integrate_unit(unit) # 1× ReAct per abstraction: agent sees full material + + # the sub-unit's name/summary, recalls, decides + # bucket, makes ONE write decision (CREATE / + # UPDATE / SKIP). Sub-unit ↔ digest node is 1:1. + +* **Phase 1 (extract / abstract)** uses a minimal toolkit + (``declare_units`` + ``read``). The agent identifies the + abstractions the material teaches — principles, patterns, + precedents worth carrying forward once specifics fade. Multiple + raw facts that illustrate the same abstraction collapse into + ONE sub-unit. Prompt biases toward fewer / coarser sub-units; + filing detail under a digest sub-unit is the wrong layer. + No event-level umbrella node is manufactured — the material + itself plays that role via ``derived_from`` provenance edges. + +* **Phase 2 (integrate per abstraction)** runs once per declared + sub-unit with a fresh ReAct session (clean context) and the full + read + write toolkit (``search``, ``traverse``, ``read``, + ``list``, ``stat``, ``frontmatter:read``, ``digest_write``, + ``digest_edit``). Three UPDATE shapes are surfaced explicitly + in the prompt: + + - **corroborate** (most common): the abstraction already + exists; the material is one more instance → append a + ``derived_from::`` provenance wikilink so confidence + accumulates; body unchanged in substance. + - **refine**: the material reveals nuance / scope / edge + cases the abstraction under-specified → tighten the + relevant span + add the new provenance link. + - **correct**: the material contradicts the abstraction → + tighten to the narrower form both old and new support, + or annotate the contradiction inline + add provenance. + + CREATE is reserved for genuinely new abstractions not yet in + the vault. SKIP should be uncommon — even an additional + instance of an existing abstraction usually warrants a + corroborate-style UPDATE. + +The trade-off vs heavy Phase 1: full material is sent to LLM K +times in Phase 2 (one per abstraction). The advantages: no +information loss in summary, focused reasoning per call, and +granularity tuned at a single prompt (Phase 1) rather than two. + +Mechanical guardrails at the write boundary: + +* ``digest_write`` (subclass of WriteStep) rejects paths outside + ``//.md`` (where ``digest_dir`` comes + from app config and ``bucket`` is in the fixed bucket set), and + refuses if the path already exists. +* ``digest_edit`` (subclass of EditStep) is a body-only find-and-replace + on an existing digest node, gated by E-1 strong-conservation: the + outbound link set BEFORE the replacement must be a subset of the link + set AFTER. If any edge would be dropped the tool returns + ``REJECT_CONSERVATION`` and the agent must adjust ``new`` to keep + the missing links before retrying. + +Invocation form (CLI / MCP): + reme dream path=daily/2026-05-28/auth-refactor/auth-refactor.md + reme dream path=resource/2026-05-28/spec.pdf hint="focus on auth" +""" + +import datetime +import zoneinfo +from pathlib import Path + +from agentscope.agent import ReActAgent +from agentscope.message import Msg, TextBlock +from agentscope.tool import Toolkit, ToolResponse +from pydantic import BaseModel, Field + +from .digest_edit import DigestEditStep +from .digest_write import DigestWriteStep, bucket_names, normalize_buckets +from ..base_step import BaseStep +from ...components import R + + +_EXTRACT_READ_TOOLS: tuple[str, ...] = ("read",) + +_INTEGRATE_READ_TOOLS: tuple[str, ...] = ( + "search", + "traverse", + "read", + "list", + "stat", + "frontmatter:read", +) + + +def _pack_material(file_store, path: str) -> str: + """Render one daily-event note or resource file into a prompt block.""" + try: + absolute = (Path(file_store.vault_path or ".") / path).resolve() + except Exception as e: + return f"### {path}\n(error resolving path: {type(e).__name__}: {e})\n" + + if not absolute.is_file(): + return f"### {path}\n(file not found)\n" + + try: + return f"### {path}\n{absolute.read_text(encoding='utf-8')}\n" + except Exception as e: + return f"### {path}\n(error reading: {type(e).__name__}: {e})\n" + + +class DreamResult(BaseModel): + """Outcome of one dreamer invocation. + + Per-tool audit lives in the toolkit layer (not exposed back to the + orchestrator). Structured outcome here is the input path the call + processed, the memory sub-units the agent declared in Phase 1, + what got created / updated in Phase 2, and any conservation + rejections that occurred along the way. + """ + + used_llm: bool = False + skipped: bool = False + path: str = "" + units: list[dict] = Field(default_factory=list) + nodes_created: list[str] = Field(default_factory=list) + nodes_updated: list[str] = Field(default_factory=list) + conservation_violations: list[dict] = Field(default_factory=list) + summary: str = "" + error: str = "" + + +@R.register("dreamer_step") +class Dreamer(BaseStep): + """auto-dream create_or_update step. + + Inputs (from RuntimeContext): + path (str, required): vault-relative path of one + daily-event note or resource file to dream over. Pass + empty string to no-op. + hint (str, optional): caller guidance to the LLM + (e.g. "focus on the auth-related decisions"). + buckets (list[str], optional): override the fixed bucket + set; default ``DEFAULT_BUCKETS``. + + Output (written to context.response.answer): + ``DreamResult`` JSON in ``metadata``; LLM summary in ``answer``. + + CLI / MCP form: + reme dream path=daily/2026-05-28/auth-refactor/auth-refactor.md + """ + + def __init__( + self, + toolkit: Toolkit | None = None, + console_enabled: bool = False, + timezone: str | None = None, + buckets: list[str] | tuple[str, ...] | None = None, + **kwargs, + ): + super().__init__(**kwargs) + self.toolkit = toolkit + self.console_enabled = console_enabled + self.timezone = timezone + self.buckets = normalize_buckets(buckets) + assert "unknown" in bucket_names(self.buckets), "bucket set must include 'unknown' as the unclassified fallback" + # Per-invocation outcome trackers, populated by tool callbacks. + self._units: list[dict] = [] + self._created: list[str] = [] + self._updated: list[str] = [] + self._violations: list[dict] = [] + + def _now(self) -> datetime.datetime: + if self.timezone: + try: + return datetime.datetime.now(zoneinfo.ZoneInfo(self.timezone)) + except Exception as e: + self.logger.error(f"Invalid timezone: {self.timezone}, error={e}") + return datetime.datetime.now() + + def _vault_dir(self) -> Path: + vr = getattr(self.file_store, "vault_path", None) + return Path(vr).resolve() if vr else Path.cwd().resolve() + + def _digest_dir(self) -> str: + """Resolve the configured digest subroot (defaults to ``"digest"``).""" + if self.app_context is not None: + val = getattr(self.app_context.app_config, "digest_dir", "") or "" + if val: + return val + return "digest" + + def _llm_available(self) -> bool: + try: + return self.as_llm is not None + except Exception: + return False + + def _make_declare_units_tool(self): + """Tool closure: agent commits the memory sub-units present in the material. + + Each unit is one orthogonal chunk of memory-worth information in this + material — e.g. for an analysis note: the subject, the method, the + decision, the finding, the open question. Free-form; not bound to the + digest bucket vocabulary (Phase 2 picks the bucket per atom at write + time). + """ + + async def declare_units(units: list[dict]) -> ToolResponse: + if not isinstance(units, list): + return ToolResponse( + content=[ + TextBlock( + type="text", + text=f"REJECT: units must be a list, got {type(units).__name__}", + ), + ], + ) + cleaned: list[dict] = [] + for i, u in enumerate(units): + if not isinstance(u, dict): + return ToolResponse( + content=[ + TextBlock( + type="text", + text=f"REJECT: units[{i}] must be an object", + ), + ], + ) + name = str(u.get("name", "")).strip() + summary = str(u.get("summary", "")).strip() + if not name or not summary: + return ToolResponse( + content=[ + TextBlock( + type="text", + text=f"REJECT: units[{i}] missing required 'name' or 'summary'", + ), + ], + ) + cleaned.append({"name": name, "summary": summary}) + # Last call wins; replaces any previous declaration in this session. + self._units = cleaned + return ToolResponse( + content=[ + TextBlock( + type="text", + text=( + f"OK: declared {len(cleaned)} memory sub-unit(s) " + f"({', '.join(u['name'] for u in cleaned)}). " + "Phase 1 closed. Downstream will process each sub-unit in a separate session." + ), + ), + ], + ) + + return declare_units + + def _make_digest_write_tool(self): + """Tool closure: wraps :class:`DigestWriteStep` and tracks creates.""" + + digest_dir = self._digest_dir() + + async def digest_write(path: str, name: str, description: str, content: str) -> ToolResponse: + step = DigestWriteStep( + file_store=self.file_store, + buckets=self.buckets, + digest_dir=digest_dir, + app_context=self.app_context, + ) + await step(path=path, name=name, description=description, content=content) + assert step.context is not None + resp = step.context.response + if not resp.success: + return ToolResponse(content=[TextBlock(type="text", text=resp.answer)]) + self._created.append(path) + return ToolResponse(content=[TextBlock(type="text", text=f"OK: created {path}")]) + + return digest_write + + def _make_digest_edit_tool(self): + """Tool closure: wraps :class:`DigestEditStep` and tracks updates / conservation violations.""" + + digest_dir = self._digest_dir() + + async def digest_edit(path: str, old: str, new: str) -> ToolResponse: + step = DigestEditStep( + file_store=self.file_store, + buckets=self.buckets, + digest_dir=digest_dir, + app_context=self.app_context, + ) + await step(path=path, old=old, new=new) + assert step.context is not None + resp = step.context.response + if not resp.success: + violation = (resp.metadata or {}).get("conservation_violation") + if violation: + self._violations.append(violation) + return ToolResponse(content=[TextBlock(type="text", text=resp.answer)]) + self._updated.append(path) + return ToolResponse(content=[TextBlock(type="text", text=f"OK: updated {path}")]) + + return digest_edit + + def _build_extract_toolkit(self) -> Toolkit: + """Minimal toolkit for the extract agent: declare_units + read-only.""" + toolkit = Toolkit() + for job_name in _EXTRACT_READ_TOOLS: + self.add_as_tool(toolkit, job_name) + declare_units_desc = ( + "Commit the list of MEMORY SUB-UNITS present in this material — the orthogonal " + "information chunks worth lifting into long-term memory. Each entry is one focused " + "sub-unit (e.g. for an analysis note: the subject, the method, a decision, a finding). " + "Free-form — sub-units are NOT bucket names, just an agent-internal clustering of the " + "material's key information. Call EXACTLY ONCE after reading. Downstream processes each " + "sub-unit in its own session and picks the bucket per atom at write time." + ) + toolkit.register_tool_function( + tool_func=self._make_declare_units_tool(), + func_name="declare_units", + func_description=declare_units_desc, + json_schema={ + "type": "function", + "function": { + "name": "declare_units", + "description": declare_units_desc, + "parameters": { + "type": "object", + "properties": { + "units": { + "type": "array", + "description": ( + "Memory sub-units identified in the material. Each is one " + "orthogonal information chunk; the same topic does not get " + "duplicated, but multiple distinct topics each get their own entry." + ), + "items": { + "type": "object", + "properties": { + "name": { + "type": "string", + "description": ( + "Short kebab-case identifier for the sub-unit " + "(e.g. 'jwt-rotation-decision', 'pr-size-pref'). " + "Agent-internal only — not the eventual digest slug." + ), + }, + "summary": { + "type": "string", + "description": ( + "1-2 sentences pointing the downstream agent at " + "the SPECIFIC part of the material this sub-unit " + "covers (e.g. 'the JWT rotation cadence decision " + "in the 决定 section, driven by SOC2')." + ), + }, + }, + "required": ["name", "summary"], + }, + }, + }, + "required": ["units"], + }, + }, + }, + ) + return toolkit + + def _build_integrate_toolkit(self) -> Toolkit: + """Full read + conservation-aware write toolkit for the integrate agent.""" + toolkit = self.toolkit or Toolkit() + for job_name in _INTEGRATE_READ_TOOLS: + self.add_as_tool(toolkit, job_name) + digest_dir = self._digest_dir() + path_shape = f"'{digest_dir}//.md'" + digest_write_desc = ( + "Create a NEW digest node — same shape as the canonical `write` job, plus " + f"path-shape validation: `path` must be {path_shape} where " + f"bucket is one of {list(bucket_names(self.buckets))} (use 'unknown' when " + "no specialized bucket fits — it is a first-class bucket, not a failure " + "state). `name` and `description` go into the YAML frontmatter; `content` " + "is the body. Fails if the path already exists; use `digest_edit` then." + ) + toolkit.register_tool_function( + tool_func=self._make_digest_write_tool(), + func_name="digest_write", + func_description=digest_write_desc, + json_schema={ + "type": "function", + "function": { + "name": "digest_write", + "description": digest_write_desc, + "parameters": { + "type": "object", + "properties": { + "path": { + "type": "string", + "description": f"vault-relative path; must match {path_shape}", + }, + "name": { + "type": "string", + "description": "frontmatter name (usually the slug)", + }, + "description": { + "type": "string", + "description": "frontmatter description — one-line summary of the abstraction", + }, + "content": { + "type": "string", + "description": "body (markdown; no frontmatter — name/description go in the fields above)", + }, + }, + "required": ["path", "name", "description", "content"], + }, + }, + }, + ) + digest_edit_desc = ( + "Find-and-replace inside an existing digest node's body — same shape as " + "the canonical `edit` job, plus path-shape validation (`path` must be " + f"{path_shape}, file must exist) and E-1 strong edge conservation: every " + "outbound wikilink present BEFORE the replacement must still be present " + "AFTER. If you drop any edge the tool returns REJECT_CONSERVATION and you " + "must adjust `new` to keep the missing links. Operates on body only; " + "frontmatter is untouched. Prefer narrow `old` spans." + ) + toolkit.register_tool_function( + tool_func=self._make_digest_edit_tool(), + func_name="digest_edit", + func_description=digest_edit_desc, + json_schema={ + "type": "function", + "function": { + "name": "digest_edit", + "description": digest_edit_desc, + "parameters": { + "type": "object", + "properties": { + "path": { + "type": "string", + "description": f"vault-relative path; must match {path_shape} and exist", + }, + "old": { + "type": "string", + "description": ( + "Substring to locate in the EXISTING body (frontmatter excluded). " + "Must match verbatim. Pick a span large enough to be unique." + ), + }, + "new": { + "type": "string", + "description": ( + "Replacement text. Should weave new material into the existing " + "wording without dropping any wikilinks the `old` span contained." + ), + }, + }, + "required": ["path", "old", "new"], + }, + }, + }, + ) + return toolkit + + async def _extract(self, material_blob: str, hint: str, vault_dir: Path) -> str: + """Phase 1: one ReAct invocation — read material + declare_units. Returns LLM summary.""" + toolkit = self._build_extract_toolkit() + agent = ReActAgent( + name="reme_dreamer_extract", + model=self.as_llm, + sys_prompt=self.prompt_format( + "extract_system_prompt", + vault_dir=str(vault_dir), + buckets=", ".join(bucket_names(self.buckets)), + ), + formatter=self.as_llm_formatter, + toolkit=toolkit, + ) + agent.set_console_output_enabled(self.console_enabled) + user_message = self.prompt_format( + "extract_user_message", + today=self._now().strftime("%Y-%m-%d"), + hint=hint or "(none)", + material_blob=material_blob, + ) + msg = await agent.reply(Msg(name="reme", role="user", content=user_message)) + return (msg.get_text_content() or "").strip() + + async def _integrate_unit(self, unit: dict, material_blob: str, hint: str, vault_dir: Path) -> str: + """One ReAct invocation per memory sub-unit. Returns LLM summary of writes.""" + toolkit = self._build_integrate_toolkit() + digest_dir = self._digest_dir() + buckets_block = "\n".join( + f" - `{digest_dir}/{b['name']}/`" + (f" — {b['description']}" if b.get("description") else "") + for b in self.buckets + ) + agent = ReActAgent( + name=f"reme_dreamer_integrate_{unit.get('name', 'unit')}", + model=self.as_llm, + sys_prompt=self.prompt_format( + "integrate_system_prompt", + vault_dir=str(vault_dir), + digest_dir=digest_dir, + buckets=buckets_block, + ), + formatter=self.as_llm_formatter, + toolkit=toolkit, + ) + agent.set_console_output_enabled(self.console_enabled) + user_message = self.prompt_format( + "integrate_user_message", + hint=hint or "(none)", + unit_name=unit.get("name", ""), + unit_summary=unit.get("summary", ""), + material_blob=material_blob, + ) + msg = await agent.reply(Msg(name="reme", role="user", content=user_message)) + return (msg.get_text_content() or "").strip() + + async def dream_one(self, path: str, hint: str = "") -> DreamResult: + """Run the full extract + integrate pipeline on one vault-relative + material path. Returns a structured :class:`DreamResult`. Safe to + call repeatedly on the same instance — per-invocation trackers are + reset at the start of each call. Used both by :meth:`execute` + (single file from context) and by :class:`CronDreamer` (loop over + today's materials). + """ + path = (path or "").strip() + hint = (hint or "").strip() + + if not path: + return DreamResult(used_llm=False, skipped=True) + + if not self._llm_available(): + return DreamResult( + used_llm=False, + skipped=True, + path=path, + error="no as_llm configured; dreaming requires an LLM", + ) + + material_blob = _pack_material(self.file_store, path) + + # Reset per-invocation trackers. + self._units.clear() + self._created.clear() + self._updated.clear() + self._violations.clear() + + vault_dir = self._vault_dir() + + # Phase 1 — extract (light). Agent calls declare_units to commit the + # memory sub-units worth lifting. + self.logger.info(f"[{self.name}] extract phase: path={path!r}") + extract_summary = await self._extract(material_blob, hint, vault_dir) + + if not self._units: + return DreamResult( + used_llm=True, + path=path, + summary=extract_summary or "SKIP: no memory sub-units declared", + skipped=True, + ) + + self.logger.info( + f"[{self.name}] integrate phase: {len(self._units)} sub-unit(s): " + f"{', '.join(u['name'] for u in self._units)}", + ) + + # Phase 2 — integrate, one fresh ReAct per sub-unit. Python-level + # loop, not agent loop. Each session decides bucket per atom written. + per_unit_replies: list[str] = [] + for i, unit in enumerate(self._units, start=1): + name = unit.get("name", "?") + try: + reply = await self._integrate_unit(unit, material_blob, hint, vault_dir) + except Exception as e: + self.logger.error( + f"[{self.name}] integrate {i}/{len(self._units)} (unit={name}) " f"failed: {type(e).__name__}: {e}", + ) + per_unit_replies.append(f"[{name}] FAILED: {type(e).__name__}: {e}") + continue + per_unit_replies.append(f"[{name}]\n{reply}") + + summary = ( + f"Declared {len(self._units)} sub-unit(s) " + f"({', '.join(u['name'] for u in self._units)}); " + f"created {len(self._created)}, updated {len(self._updated)}, " + f"conservation violations {len(self._violations)}.\n" + "\n\n".join(per_unit_replies) + ) + + return DreamResult( + used_llm=True, + path=path, + units=list(self._units), + nodes_created=list(self._created), + nodes_updated=list(self._updated), + conservation_violations=list(self._violations), + summary=summary, + skipped=False, + ) + + async def execute(self): + assert self.context is not None + path: str = (self.context.get("path", "") or "").strip() + hint: str = (self.context.get("hint", "") or "").strip() + + result = await self.dream_one(path, hint) + + if not path: + self.context.response.success = True + self.context.response.answer = "Skipped: no path supplied" + elif result.error: + self.context.response.success = False + self.context.response.answer = f"Error: {result.error}" + elif result.skipped: + self.context.response.success = True + self.context.response.answer = result.summary or "SKIP" + else: + self.context.response.success = True + self.context.response.answer = result.summary + self.context.response.metadata.update(result.model_dump()) diff --git a/reme4/steps/dream/dreamer.yaml b/reme4/steps/dream/dreamer.yaml new file mode 100644 index 00000000..fc251521 --- /dev/null +++ b/reme4/steps/dream/dreamer.yaml @@ -0,0 +1,408 @@ +extract_system_prompt: | + You are the **dreamer** — auto-dream's create_or_update step, + in its EXTRACT phase. Your ONLY job here is to read the material + and identify the ABSTRACTIONS it teaches — the principles, + patterns, decisions-as-precedent, cognitive takeaways — that + belong in long-term memory. You commit them by calling + `declare_units` exactly once. You do NOT do recall, integrate, + or write. A separate downstream invocation processes each unit + with the full material in context. + + vault_dir: {vault_dir} + + ## What digest memory is for + + Digest is the **abstract memory layer** — analogous to the + prefrontal cortex aggregating cognition. The raw details of + what happened (numbers, narratives, who said what, full + procedure text) STAY IN THE MATERIAL. Digest holds the + generalized lesson the reader should recall next time — + the part that survives once the specific event fades. + + When you cluster, you are NOT cataloguing the material's + contents — you are answering: *"What abstractions does + this material teach that I'd want a future agent / human + to have at-hand when facing a similar situation?"* + + ## What is a memory sub-unit? + + One sub-unit = one abstraction the material teaches. **One + sub-unit maps to AT MOST one digest node** — Phase 2 will + make exactly one write decision per sub-unit (CREATE / + UPDATE / SKIP). + + Multiple raw facts in the material that all illustrate the + same abstraction collapse to ONE sub-unit. The Redis-kid + versioning mechanism, the SOC2 CC6.1 rationale, and the new + 24h cadence are three FACTS, but they teach one abstraction: + "JWT rotation cadence is driven by short-credential + compliance, not by procedural convenience". That's one + sub-unit. The mechanism / numbers / RFC citation are + details — they stay in the daily note, the digest reaches + them through `derived_from::` provenance edges. + + Sub-units are NOT bucket names, NOT kinds, NOT the eventual + digest slug — they are an agent-internal handle for the + abstraction you've identified. Phase 2 picks the bucket / + slug / write decision per sub-unit. + + Typical abstractions, by material shape: + + * Analysis / decision notes: the underlying principle the + decision rests on; a pattern the analysis surfaces; + a constraint that will recur in similar problems. + * Discussion notes: a preference / convention that should + shape future work; a stable concept the discussion + crystallizes; an open question worth carrying forward. + * Resource content: a foundational concept; a procedure + that generalizes beyond this resource. + + ### Bias: fewer, richer sub-units over many narrow ones + + This is the abstract layer — heavy lifting toward few + high-leverage sub-units, not toward exhaustive coverage. + Heuristic for splitting two pieces into two sub-units vs + one: + + * Same abstraction shown by different facts? → ONE sub-unit. + * Genuinely different abstractions that a future reader + would invoke in DIFFERENT situations? → TWO sub-units. + * Will they evolve independently as more materials arrive? + → TWO sub-units. + + When in doubt, KEEP TOGETHER (or SKIP one of them entirely). + + Examples: + + * "JWT rotation cadence changed to 24h" + "Redis kid + versioning mechanism" + "SOC2 CC6.1 cited" → ONE + sub-unit (the abstraction: *short-credential compliance + drives auth infra cadence*). Mechanism + numbers are + details — they stay in the daily. + * "preference: small PRs" + "preference: no trailing + summary in replies" → TWO sub-units. Different + situations of invocation (code review vs response + style), independent evolution. + + ### What NOT to declare + + - A passing mention with no new abstraction (e.g. an OAuth + recap that just restates a known concept) → don't declare. + The material remains searchable via daily-note indexing; + detail-level recall doesn't need a digest entry. + - A fact whose only audience is the material itself + (a one-off timestamp, a single meeting attendance) → + don't declare. Not an abstraction. + + ### No event-level umbrella needed + + The material itself (the daily note or resource file) IS the + event-level aggregator. Every sub-unit you declare here will + carry a `derived_from:: [[]]` provenance + wikilink, so the material becomes the fan-out point linking + to all its derived digest nodes. Do NOT manufacture an + extra "X-event-summary" sub-unit just to aggregate the + others — the provenance graph already provides that view. + + ## What to do + + 1. **Read the material** — its body is packed in the user + message below. If it references `[[resource//]]` + and that asset is critical to understanding what + abstractions are present, you MAY open it via `read`; + otherwise skip external reads (this is the light phase). + + 2. **Identify the abstractions** the material teaches. + For each candidate, ask: *if I forgot all the details + of this material in 6 months, what one-line lesson + would I still want to recall?* That lesson is a + sub-unit candidate. + + 3. **Call `declare_units` ONCE** with the surviving list: + - `name` — short kebab-case handle for the abstraction + (e.g. `auth-cadence-compliance-driven`, + `small-pr-pref`). Agent-internal only; Phase 2 + picks the actual digest slug + bucket. + - `summary` — 1-2 sentences describing the abstraction + AND pointing at where in the material it's illustrated + (e.g. "abstraction: short-credential compliance + drives auth infra rotation cadence; illustrated by + the 30→24h decision in 决定 backed by the SOC2 CC6.1 + criticism in 观察"). Be concrete about WHERE the + supporting evidence lives, so Phase 2 can cite it as + provenance without re-reading. + + After `declare_units` returns OK, reply with one short line + listing the sub-unit names. + + If the material teaches no new abstraction worth long-term + memory (e.g. routine status updates, pure logs), do NOT call + `declare_units`; reply starting with `SKIP`. + + ## Boundaries + + - You CANNOT write to digest in this phase (no + digest_write / digest_edit tools here). + - You CANNOT do recall in this phase (no search/traverse here). + - You declare ABSTRACTIONS (sub-units), not detail copies. + Phase 2 handles recall + the single write decision per + sub-unit. + - The list you declare is the final scope for this dream call. + +extract_user_message: | + today: {today} + hint: {hint} + + # Material to cluster + + {material_blob} + + Identify the ABSTRACTIONS this material teaches (lessons / + principles / patterns worth recalling after the details fade). + Collapse multiple supporting facts into one sub-unit when they + illustrate the same abstraction. Call `declare_units([...])` + exactly once with the surviving list. Reply with one short + line listing the sub-unit names (or `SKIP` if the material + teaches no new abstraction). + + +integrate_system_prompt: | + You are the **dreamer** — auto-dream's create_or_update step, + in its INTEGRATE phase. This invocation processes ONE MEMORY + SUB-UNIT against the full material. You see the entire material + in the user message; Phase 1 told you which abstraction to + focus on and pointed you at the supporting evidence. Your job: + recall existing digest nodes (cross-bucket), decide CREATE / + UPDATE / SKIP for this sub-unit, and write. + + **Sub-unit maps 1:1 to a digest node.** Exactly ONE write + decision per session. + + ## Digest is the abstract memory layer + + Digest is **not** a faithful copy of the material — it is the + cognitive aggregation (think prefrontal cortex). The details + stay in the daily / resource file; digest holds the principle, + pattern, or precedent the agent should recall later. So: + + - **Body should be SHORT and abstract** (≈ 50-200 words for + most nodes; longer only when the concept genuinely needs it). + If your draft starts copying paragraphs from the material, + you're filing detail in the wrong layer. + - **Provenance edges carry the details.** Whenever this + abstraction is illustrated by a specific material, add a + `derived_from:: [[daily/...]]` or `[[resource/...]]` + wikilink — readers drill down through the edge, not through + re-stated facts in the body. + - **Wikilinks between digest nodes** carry the conceptual + graph: `relates_to::`, `depends_on::`, `is_a::`, etc. + + ## What to do + + ### a. Recall (search + read + optional traverse) + + - **Search** — call `search` with the sub-unit's likely slug + + its summary. The step returns top-K matched chunks PLUS a + one-hop link expansion (immediate wikilink neighbors of each + hit). Hits come from any path under the vault; you care + primarily about ones under `{digest_dir}/`. + + - **Read full bodies** — do NOT decide UPDATE on chunk snippets + alone. A snippet shows ~a paragraph of context, not the full + node. For any hit (or expanded neighbor) that looks like the + same abstraction, follow up with `read path=` to + read the complete body before deciding. + + - **Walk further if needed** — for 2+ hop exploration, use + `traverse path= depth=2 direction=both`, then + `read` the interesting paths. + + Recall is intentionally cross-bucket — the same abstraction + may already be filed under any bucket; surface it regardless + of where it lives. UPDATE may target a node in any bucket. + + ### b. Decide bucket + write — exactly one of: + + - **`digest_write(path, name, description, content)`** — for CREATE. + Same shape as the canonical `write` job; the digest variant + only adds path-shape validation. Use ONLY when no existing + digest node captures this abstraction. + - `path` must be `{digest_dir}//.md` where `bucket` + is one of the FIXED bucket vocabulary below (pick the + one a human would browse for this abstraction; use + `unknown` only as a last resort). + - `name` is the frontmatter name (usually the slug). + - `description` is the one-line summary of the abstraction + (lands in YAML frontmatter; downstream search relies on it). + - `content` is the body — short (≈ 50-200 words), abstract, + principle-oriented — NOT a transcript of the material. + Do NOT prepend `---` frontmatter into `content`; the step + composes the frontmatter from `name` + `description` + automatically. Include at least one + `derived_from:: [[]]` provenance wikilink + in the body so the abstraction can be traced back to its + source. + Fails if path exists; if so, this is actually an UPDATE — + re-do recall and switch to `digest_edit`. + + - **`digest_edit(path, old, new)`** — for UPDATE. + This is the cognitive engagement step. The existing digest + captures an earlier version of the abstraction; the new + material **corroborates, corrects, or refines** it. Three + typical shapes: + + 1. **Corroborate** (most common). The material is one + more instance of an abstraction already captured. + Body usually unchanged in substance — append a new + `derived_from:: [[]]` provenance + wikilink so the supporting evidence accumulates. + Optionally strengthen wording ("consistently + observed across N sources" / replace "appears to" with + "does"). One small `digest_edit` call is enough. + 2. **Refine** (frequent). The material reveals nuance, + scope, or edge cases the existing abstraction + under-specified. Edit the relevant span to be more + precise; add the new dimension; still add the new + `derived_from::` link. The body grows in precision, + not in detail. + 3. **Correct** (rarer). The material contradicts the + existing abstraction or shows it was overstated. + Either tighten the abstraction to the narrower form + that both old and new evidence support, or annotate + inline (`> note: contradicted by [[new-material]] — + `) without arbitrating; future passes can + reconcile. Still add the provenance link. + + Body-only find-and-replace (frontmatter is untouched). + Pick a `old` span big enough to be unique in the body. + Prefer narrow spans over rewriting the whole body. + Composition rule for `new`: only-add, not-delete — never + drop facts the old span contained. You MAY issue more + than one `digest_edit` against the SAME target if + multiple sections need updating; never write to a + different target as a side-effect. + + `digest_edit` ENFORCES edge conservation (E-1): every + outbound wikilink present BEFORE the replacement must still + be present AFTER. On `REJECT_CONSERVATION` the missing + links are listed — adjust `new` to keep them (or narrow + `old` so the link stays outside the replaced span), then + retry. + + - **SKIP** — use when: + * Phase 1 declared this sub-unit but on closer reading + the material teaches nothing new (the existing + abstraction's body already covers this instance AND + already has provenance to a comparable source), OR + * the sub-unit is too thin to lift as an abstraction — + a one-off datapoint that doesn't generalize. + + SKIP should be uncommon. If the abstraction exists and the + material adds even ONE new datapoint, prefer a Corroborate- + style UPDATE (provenance append) over SKIP — that's how + the abstraction's confidence accumulates. + + Write only the target you committed to for this sub-unit. + Never edit other nodes' bodies sideways — inbound relations are + queried later at search time, never written into target bodies. + + ## Bucket vocabulary + + Pick the bucket per sub-unit when you write. The vocabulary + is fixed and injected here (each line is one allowed bucket + with its picking heuristic — `{digest_dir}//` is what + a human will browse): + + {buckets} + + If the sub-unit straddles two buckets, pick the one matching + its CENTER OF GRAVITY — what a reader is most likely to search + for. Don't split into two writes. + + User-memory ground rule (applies when both `preference` and + `entity` are in the vocabulary above): anything about how the + user / team likes to work, what they explicitly said NOT to + do, what conventions they follow → `preference`. The user + themselves, when named as an individual, is `entity`; their + preferences live separately in `preference`. + + ## Wikilink form + + Always full vault-relative path with `.md`: + + - `[[{digest_dir}//.md]]` + - `[[daily///.md]]` + - `[[resource//]]` + + Short or extension-less forms do not resolve. + + Optional Dataview-style typed predicates (the predicate sits + outside the brackets): + + - line-level: `is_a:: [[{digest_dir}/concept/jwt.md]]` + - inline: `relies on [depends_on:: [[{digest_dir}/procedure/key-rotation.md]]]` + - typed provenance: `derived_from:: [[daily/2026/05/15/auth-refactor.md]]` + + Predicate vocabulary is open (any `[A-Za-z][A-Za-z0-9_]*`); + reuse existing predicates when reasonable. Most wikilinks are + bare (no predicate) — use a predicate only when the relation + has clear semantic weight. + + ## Provenance + + The body must weave at least one provenance wikilink — + `[[daily/...]]` or `[[resource/...]]` — so the graph stays + connected upstream. Do NOT write provenance as bare prose + ("from yesterday's notes"); the conservation check only sees + wikilinks, so prose provenance effectively vanishes on the + next update. + + ## Frontmatter + + Reserved fields (both optional): + + - `name` — basename without extension + - `description` — one-line summary + + Optional `kind` (downstream filtering hint; e.g. `concept` / + `procedure` / `entity` / `observation` / `preference` / ...) + — reme core does not read it for any structural decision. Do + NOT write a `status` field — there is no distill-pass marker + in this design. + + ## Reply + + Reply with ONE LINE summarizing your decision for this sub-unit, + including the UPDATE shape when applicable: + + - `CREATE {digest_dir}//.md` — for create + - `UPDATE {digest_dir}//.md (corroborate)` — provenance append + maybe wording strengthening + - `UPDATE {digest_dir}//.md (refine)` — abstraction made more precise / extended in scope + - `UPDATE {digest_dir}//.md (correct)` — abstraction tightened or contradiction annotated + - `SKIP: ` — for skip + + If `digest_edit` returned REJECT_CONSERVATION and you + recovered, append `(recovered from REJECT_CONSERVATION)` to + the UPDATE line. + +integrate_user_message: | + hint: {hint} + + # Your assigned memory sub-unit for this call + + name: {unit_name} + summary: {unit_summary} + + # Full material + + {material_blob} + + Process sub-unit `{unit_name}` (the summary above tells you + what abstraction this is and where its evidence lives in the + material). Do recall (search → read → optional traverse), + then make EXACTLY ONE write decision: CREATE one new node, + UPDATE one existing node (corroborate / refine / correct), or + SKIP. Pick the bucket. Keep the body short and abstract — + details stay in the material, reachable via `derived_from::` + provenance links. Reply with the one-line decision per the + format in the system prompt. diff --git a/reme4/steps/jobs/__init__.py b/reme4/steps/jobs/__init__.py index ff1927ec..e69de29b 100644 --- a/reme4/steps/jobs/__init__.py +++ b/reme4/steps/jobs/__init__.py @@ -1,3 +0,0 @@ -"""Jobs steps — composite ReAct-agent-driven workflows.""" - -from . import digester # noqa: F401 -- @R.register("digester") diff --git a/reme4/steps/jobs/digester.py b/reme4/steps/jobs/digester.py deleted file mode 100644 index f40b7f74..00000000 --- a/reme4/steps/jobs/digester.py +++ /dev/null @@ -1,300 +0,0 @@ -"""Smart Digester — knowledge distillation from daily notes to digest/. - -The Digester is the **cold-write** counterpart to AutoMemory (hot-write). -It reads completed work in ``daily//.md`` note files, -identifies entities / concepts / claims / methods worth preserving -long-term, and sinks them into ``digest/`` as canonical-entry nodes so -the main agent can retrieve them later via search and graph traversal. - -``digest/`` is the cold-tier root. Per ``protocol.md``, scope folders -under it may nest arbitrarily; each folder's canonical entry is -``/.md``, and slug (folder name) is globally unique across -the whole tree. Pending detection and graph machinery treat nodes at any -depth uniformly. - -Drives a ReAct agent with a read/lookup/graph/write toolkit; the -agent follows the protocol in ``protocol.md`` (the opinionated -default schema + R-M-W decision tree). The schema is convention-driven -— reme core only reserves ``name`` / ``description``, so the agent -owns its own discipline rather than relying on a post-write linter. - -Distillation state lives in the daily note's ``status`` frontmatter -— a **daily-tier convention owned by this digester** (reme core -reserves only ``name`` / ``description``; ``status`` is just an -extra). After processing each daily, the agent must call -``frontmatter_update`` with ``metadata={"status": "completed"}`` -(or ``metadata={"status": "skipped"}`` when intentionally bypassed). Convention: absent -≡ ``pending``, so the next pass finds residual work via -``file_list path=daily recursive=true`` + per-item ``frontmatter_read`` to filter for absent ``status``. -Only the digester writes ``status``; AutoMemory / hand-edits must -leave it alone. - -No degraded path — distillation strictly requires an LLM. When ``as_llm`` -is unavailable, the step short-circuits with ``skipped=True`` and an -error message. - -Override interface — schema is a service-consumption concern, not a -core invariant, so this step ships an **opinionated default** that any -caller can fully replace without touching reme4: - -* ``protocol`` / ``protocol_path`` constructor args replace the - ``protocol.md`` injected as ``{protocol}`` in the system prompt - (use when keeping the default prompt template but swapping schema). -* ``prompt_dict`` (inherited from ``BaseStep``) replaces the - ``system_prompt`` / ``user_message`` templates wholesale (use when - the prompt structure itself needs to change). -* ``toolkit`` replaces the tool surface ``_DIGESTER_TOOLS`` builds. - -Service layers (e.g. plugin-side configs) wire these in via component -config; ``digester.py`` / ``protocol.md`` shipped here are just a -reference implementation of one viable convention. - -Toolkit. Each entry in ``_DIGESTER_TOOLS`` is a job name registered -in the active config; ``add_as_tool`` wraps ``job(**kwargs)`` into a -``ToolResponse``. The job indirection means the agent sees the same -tool surface (rich descriptions + JSON schema) as the L2 MCP layer. -""" - -import datetime -import zoneinfo -from pathlib import Path - -from agentscope.agent import ReActAgent -from agentscope.message import Msg -from agentscope.tool import Toolkit -from pydantic import BaseModel, Field - -from ..base_step import BaseStep - -from ...components import R - - -_DIGESTER_TOOLS: tuple[str, ...] = ( - "file_list", - "file_read", - "file_stat", - "frontmatter_read", - "traverse", - "file_write", - "file_append", - "file_move", - "frontmatter_update", - "frontmatter_delete", -) - - -def _pack_daily(file_store, daily_path: str) -> str: - """Render one daily note file's body into a prompt-friendly block. - - A daily note is a single self-contained markdown file at - ``daily//.md`` — everything the originating task wanted - the digester to see is inline (no sibling materials). External - assets land in ``resource//`` and are linked from the note's - ``## References`` section; the LLM opens those on demand via - ``file_read``. - """ - try: - absolute = (Path(file_store.vault_path or ".") / daily_path).resolve() - except Exception as e: - return f"### {daily_path}\n(error resolving path: {type(e).__name__}: {e})\n" - - if not absolute.is_file(): - return f"### {daily_path}\n(note file not found)\n" - - parts: list[str] = [f"### {daily_path}"] - try: - parts.append(absolute.read_text(encoding="utf-8")) - except Exception as e: - parts.append(f"(error reading note: {type(e).__name__}: {e})") - return "\n".join(parts) + "\n" - - -class DistillResult(BaseModel): - """Outcome of a single distillation call. - - Without per-tool audit (the agent's toolkit is the job surface, - which doesn't expose per-call records back to the orchestrator), - the structured outcome is just the inputs the call was asked to - process plus the LLM's free-form summary. Per-file write outcomes - can be verified by re-reading vault_dir afterwards if needed. - - Field semantics: - * ``daily_read`` — daily paths actually processed (input - order, deduped) - * ``summary`` — LLM's free-form one-paragraph summary - * ``skipped`` — True when no LLM available, no daily - paths provided, or the LLM reported ``SKIP`` - * ``error`` — short error string when skipped due - to misconfiguration (e.g. no LLM) - """ - - used_llm: bool = False - skipped: bool = False - daily_read: list[str] = Field(default_factory=list) - summary: str = "" - error: str = "" - - -@R.register("digester") -class Digester(BaseStep): - """Knowledge digester: daily/ → digest/ via a ReAct agent. - - Inputs (from RuntimeContext): - daily_paths (list[str], required): vault-relative paths to - daily note files (``daily//.md``) to distill. - Pass ``[]`` to no-op. - hint (str, optional): caller guidance to the LLM - (e.g. "focus on the auth-related decisions"). - - Output (written to context.response.answer): - DistillResult JSON — see model docstring. - """ - - def __init__( - self, - toolkit: Toolkit | None = None, - console_enabled: bool = False, - timezone: str | None = None, - protocol: str | None = None, - protocol_path: str | None = None, - **kwargs, - ): - """Constructor overrides (service layer customization points): - - * ``toolkit`` — replace the agent's tool surface; default builds - one from ``_DIGESTER_TOOLS``. - * ``protocol`` — inline protocol document (highest precedence); - overrides whatever the agent sees under ``{protocol}`` in the - system prompt. - * ``protocol_path`` — path (relative to the vault or absolute) to a protocol - markdown file; used when ``protocol`` is not given. - * ``prompt_dict`` (inherited via ``BaseStep``) — override the - ``system_prompt`` / ``user_message`` templates wholesale, e.g. - to swap in a service-layer prompt that hardcodes a different - schema entirely. - - With none of the above, falls back to the opinionated default - (the ``protocol.md`` and ``digester.yaml`` shipped alongside - this module). - """ - super().__init__(**kwargs) - self.toolkit = toolkit - self.console_enabled = console_enabled - self.timezone = timezone - self._protocol = self._load_protocol(protocol, protocol_path) - - @staticmethod - def _load_protocol(protocol: str | None, protocol_path: str | None) -> str: - """Resolve the protocol document; explicit string > path > default.""" - if protocol is not None: - return protocol - if protocol_path: - path = Path(protocol_path) - if path.exists(): - return path.read_text(encoding="utf-8") - default_path = Path(__file__).parent / "protocol.md" - return default_path.read_text(encoding="utf-8") if default_path.exists() else "" - - def _now(self) -> datetime.datetime: - if self.timezone: - try: - return datetime.datetime.now(zoneinfo.ZoneInfo(self.timezone)) - except Exception as e: - self.logger.error(f"Invalid timezone: {self.timezone}, error={e}") - return datetime.datetime.now() - - def _vault_dir(self) -> Path: - vr = getattr(self.file_store, "vault_path", None) - return Path(vr).resolve() if vr else Path.cwd().resolve() - - def _llm_available(self) -> bool: - """Pre-flight check: can ``self.as_llm`` resolve without raising? - - ``BaseStep.as_llm`` asserts when no model is registered, so we - wrap the access here to avoid hard-failing at the call site.""" - try: - return self.as_llm is not None - except Exception: - return False - - def _build_toolkit(self) -> Toolkit: - """Bind every digester-relevant job as a tool function.""" - toolkit = self.toolkit or Toolkit() - for job_name in _DIGESTER_TOOLS: - self.add_as_tool(toolkit, job_name) - return toolkit - - async def execute(self): - assert self.context is not None - daily_paths: list[str] = list(self.context.get("daily_paths") or []) - hint: str = (self.context.get("hint", "") or "").strip() - - # No work to do: no daily paths supplied. - if not daily_paths: - result = DistillResult(used_llm=False, skipped=True) - self.context.response.success = True - self.context.response.answer = "Skipped: no daily paths supplied" - self.context.response.metadata.update(result.model_dump()) - return - - # No LLM available: distillation strictly requires one. - if not self._llm_available(): - result = DistillResult( - used_llm=False, - skipped=True, - error="no as_llm configured; distillation requires an LLM", - ) - self.context.response.success = False - self.context.response.answer = f"Error: {result.error}" - self.context.response.metadata.update(result.model_dump()) - return - - # Dedupe daily_paths while preserving order. - seen: set[str] = set() - deduped: list[str] = [] - for p in daily_paths: - if p and p not in seen: - seen.add(p) - deduped.append(p) - daily_paths = deduped - - # Build the per-daily blob the agent will see. - daily_blob = "\n\n".join(_pack_daily(self.file_store, p) for p in daily_paths) - - vault_dir = self._vault_dir() - toolkit = self._build_toolkit() - - agent = ReActAgent( - name="reme_digester", - model=self.as_llm, - sys_prompt=self.prompt_format( - "system_prompt", - vault_dir=str(vault_dir), - protocol=self._protocol, - ), - formatter=self.as_llm_formatter, - toolkit=toolkit, - ) - agent.set_console_output_enabled(self.console_enabled) - - user_message: str = self.prompt_format( - "user_message", - today=self._now().strftime("%Y-%m-%d"), - hint=hint or "(none)", - daily_blob=daily_blob or "(none)", - ) - - final_msg: Msg = await agent.reply( - Msg(name="reme", role="user", content=user_message), - ) - summary = (final_msg.get_text_content() or "").strip() - - result = DistillResult( - used_llm=True, - daily_read=list(daily_paths), - summary=summary, - skipped=summary.upper().startswith("SKIP"), - ) - self.context.response.success = True - self.context.response.answer = summary or "Distillation completed" - self.context.response.metadata.update(result.model_dump()) diff --git a/reme4/steps/jobs/digester.yaml b/reme4/steps/jobs/digester.yaml deleted file mode 100644 index cf311ec6..00000000 --- a/reme4/steps/jobs/digester.yaml +++ /dev/null @@ -1,114 +0,0 @@ -system_prompt: | - You are the digester. You read completed work in `daily//.md` - note files and lift entities, concepts, claims, and methods into - canonical entries under `digest/<…>//.md`. The main agent - retrieves them later via search and graph traversal. - - vault_dir: {vault_dir} - - ## Five steps per call - - ### Step 1 — Read each daily - - For every daily note passed in, the full body is already packed - in the user message below. Each note is a single self-contained - markdown file — anything the originating task wanted you to see is - inline. If a note's `## References` section points at - `[[resource//]]` items and you need them, open them via - `file_read` on demand. - - ### Step 2 — Lookup candidates - - Identify each entity / concept / claim / method named in the daily. - Before deciding to CREATE anything, look it up in `digest/`: - - - `file_list path=digest/ recursive=true` to scan the tree, or - - `graph_traverse path=.md depth=1` for neighborhood. - - **Slugs are globally unique under `digest/`** — a prior occurrence - at any nesting depth means the node already exists. Find and reuse - it; do not CREATE a duplicate under a new path. This is the worst - failure mode of the digester. - - ### Step 3 — R-M-W decision - - Per candidate, pick exactly one branch. Every branch writes only - the node being authored in this step — never sideways into other - nodes' bodies. A relation worth recording is captured by a typed - wikilink in the source body; the inbound view is queried later via - `graph_traverse direction=in`. - - - **CREATE** — no hit anywhere under `digest/`: - `file_write digest/<…>//.md`. Place it under the - closest existing semantic parent scope; top-level if none applies. - - - **UPDATE** — exact match exists: - Merge new facts into the right section. Prefer - `frontmatter_update` for metadata; `file_append` for purely - additive trailing sections; `file_read` + `file_write` for - mid-body edits. Override stale claims rather than stacking - contradictions. Don't restructure unrelated parts. - - - **MOVE / promote** — rename or relocate an existing node: - `file_move` (the retarget pass rewrites inbound wikilinks - atomically; never `file_write` to new + `file_delete` old). - - When CREATE / UPDATE writes a relation into the body, prefer a - typed wikilink (`predicate:: [[X]]` or `[predicate:: [[X]]]`) when - the relation has clear semantic weight; default to bare `[[X]]` for - plain mention. See protocol.md §4.1 for the recommended predicate - vocabulary. - - If a candidate is only a passing mention with no new fact to write, - do nothing — the mention stays in the daily, search will still - find it, and a future digester pass can lift it when it accumulates - substance worth CREATE/UPDATE. - - ### Step 4 — Flip status per daily (mandatory) - - After processing each daily, call: - - frontmatter_update path=daily//.md metadata={status: completed} - - Use `status=skipped` if the daily had nothing worth lifting (chitchat, - dead end). This flip is the daily-tier convention this digester owns - — absent ≡ pending, so the next pass uses - `file_list path=daily recursive=true` + per-item `frontmatter_read` - to find leftover work. Forgetting the flip leaves the daily in the - queue forever. - - ### Step 5 — Reply - - One short paragraph: which dailies you read, what you CREATEd / - UPDATEd / MOVEd (with paths), and which dailies you flipped to - `completed` vs `skipped`. Don't replay every tool call — those are - on the audit trail. - - ## Wikilinks - - Always full path relative to the vault with `.md`: - `[[digest/<…>//.md]]` for digest entries, - `[[daily//.md]]` for daily notes, - `[[resource//]]` for ingested resources. Short forms or - extension-less forms don't resolve. - - ## Boundaries - - - **Never write under `daily/`** except for the Step 4 `status` flip. - - **Only this digester writes `status`** — sync / hand-edits leave - it alone (absent ≡ pending is exactly how this digester finds its - workload). - - ## Memory protocol - - {protocol} - -user_message: | - today: {today} - hint: {hint} - - # Daily notes to distill - - {daily_blob} - - Run the five steps from the system prompt. Reply = one paragraph audit. diff --git a/reme4/steps/jobs/protocol.md b/reme4/steps/jobs/protocol.md deleted file mode 100644 index 58e42105..00000000 --- a/reme4/steps/jobs/protocol.md +++ /dev/null @@ -1,112 +0,0 @@ -# Memory Protocol - -Opinionated default contract for writing memory into a reme vault. -reme core reserves only `name` / `description` (both optional); -everything below is convention that consumers may replace. - -## 1. Directory architecture - -``` -/ -├── daily/ -│ └── / -│ └── / -│ ├── .md # hot summary note (markdown) -│ └── .* # sibling materials (any file type) -└── digest/ - └── / - ├── .md # cold canonical entry - ├── .md # supporting docs - └── / # nested narrower scope - └── .md -``` - -- **Hot tier** (`daily/…`) — streaming. One upstream writer per - folder; every other consumer treats it as read-only. The - summary note `.md` is markdown; siblings may be any - file type the writer chooses. -- **Cold tier** (`digest/…`) — curated. Each folder is a - scope and must contain `/.md` as its canonical - entry. A scope's other children are sibling material files or - narrower scope subfolders; nesting depth is unconstrained. - Slugs are globally unique — a folder name appears at most once - anywhere under `digest/`. - -Facts flow one-way, `daily/` → `digest/`. References use the -full path relative to the vault: `[[digest///.md]]`. - -## 2. Frontmatter - -Reserved (typed; all optional): - -| key | type | -|---|---| -| `name` | string | -| `description` | string | - -Opinionated default axes (closed enums): - -| key | values | -|---|---| -| `lifecycle` | `streaming` / `evolving` / `frozen` | -| `scope` | `instance` / `class` | -| `source` | `auto` / `curated` / `derived` | -| `role` | `profile` / `concept` / `claim` / `method` / `reference` / `observation` / `question` / `fundamentals` | - -Any other keys consumers want (e.g. a workflow `status` flag) live -as extras — write them, read them with the `where` filter on `list` -tools (`null` matches absent-or-null); the protocol does not name -or enumerate them. - -## 3. Body - -Section structure is **advisory** — `## Summary`, `## Key Facts`, -`## Decisions`, `## Related` are convenient defaults but the -protocol mandates no specific section. - -## 4. Wikilinks - -Three recognized forms: - -| Form | Example | Meaning | -|---|---|---| -| Bare | `See [[张三.md]]` | weakest layer — "mention" | -| Line-level Dataview | `colleague:: [[李四.md]]` | typed relation, queryable by predicate | -| Inline-bracketed Dataview | `主导 [负责:: [[项目X.md]]] 的重构` | typed relation, embedded inline | - -Targets are stored **verbatim** as full paths relative to the vault. -`[[digest/zhang-san/zhang-san.md]]` resolves; short or -extension-less forms do not — no implicit `.md` completion, no -basename search, no folder-note expansion. - -Renaming a node requires atomically rewriting every inbound -wikilink. - -### 4.1 Typed predicates (half-open) - -Recommended core vocabulary — writers may extend beyond this set, -but new predicates should be reused consistently: - -| predicate | meaning | -|---|---| -| `is_a` | hierarchical (X is a kind of Y) | -| `part_of` | containment (X is part of Y) | -| `depends_on` | dependency (X requires Y) | -| `manages` | authority / responsibility | -| `alias_of` | equivalence (X and Y are the same thing) | -| `references` | citation / external pointer | - -Use typed `predicate:: [[X]]` only when the relation has clear -semantic weight; default to bare `[[X]]` for plain mentions. Typed -edges become queryable via `graph_traverse predicate=`. - -### 4.2 One-way write rule - -A wikilink lives in the **source** node's body only — the node whose -prose introduces the relation. The **target** is never modified to -record the inbound relation. Backlinks are discovered at query time -via `graph_traverse direction=in`, never written into target bodies. - -This keeps every write authoritative: a node's body reflects only -what its own author/writer chose to say, never sideways annotations -from other nodes' writers. diff --git a/reme4/steps/transfer/ingest.py b/reme4/steps/transfer/ingest.py index 4ecf87c1..f9f36b52 100644 --- a/reme4/steps/transfer/ingest.py +++ b/reme4/steps/transfer/ingest.py @@ -63,7 +63,7 @@ Parameters: only. * ``description`` (required) — analysis hint for downstream agents: where the asset came from, what kind of content it carries, and how - it should be interpreted. The digester / auto_memory reads this + it should be interpreted. The dreamer / auto_memory reads this verbatim from ``meta.json`` to decide how to read the asset (skim vs. deep parse, structured extraction vs. summarization, etc.), so callers should write enough detail to drive that decision — not diff --git a/tests4/smoke/_dreamer_fixture.py b/tests4/smoke/_dreamer_fixture.py new file mode 100644 index 00000000..b8b95e4f --- /dev/null +++ b/tests4/smoke/_dreamer_fixture.py @@ -0,0 +1,273 @@ +"""Fixture for the dreamer smoke tests. + +Seeds a vault with: + + - 4 pre-existing digest/ nodes (3 concepts/procedures, 1 preference) — + these are the **recall targets**; the new material partially overlaps + them so Phase 2 must find them via search_step + file_read and decide + UPDATE (with E-1 edge conservation in play). + - 4 small daily/ stubs the digest nodes already link to — they exist + only so the seeded digest bodies don't dangle (the conservation check + doesn't validate link targets, but a realistic graph reads better). + - 1 NEW daily note (the file the dreamer will be invoked on). It covers + all three outcome cases — CREATE / UPDATE / SKIP — across 4 kinds: + * concept : UPDATE digest/concept/jwt.md + CREATE digest/concept/kid-versioning.md + + SKIP (an OAuth2 restatement that adds nothing over + the existing digest/concept/oauth2.md) + * procedure : UPDATE digest/procedure/key-rotation.md + * observation: CREATE digest/observation/soc2-30day-finding.md + * preference : UPDATE digest/preference/no-trailing-summary.md + CREATE digest/preference/small-pr.md + + Total budget per smoke run: 1 Phase 1 + up-to-4 Phase 2 = up to 5 + ReAct sessions, each with several tool turns (search → file_read → + digest_* / SKIP). + +Idempotent: re-running does NOT overwrite existing files. To re-seed +from scratch, delete the vault and rerun. + +Usage as a script: + python tests4/smoke/_dreamer_fixture.py /tmp/my-vault + +Usage as a module: + from _dreamer_fixture import clean_vault, seed_vault, INPUT_PATH + clean_vault(Path("/tmp/my-vault")) + seed_vault(Path("/tmp/my-vault")) +""" + +import shutil +import sys +from pathlib import Path + +INPUT_PATH = "daily/2026-05-28/auth-refactor/notes.md" + + +_FILES: dict[str, str] = { + # ----- pre-existing digest nodes (recall targets) ----- + "digest/concept/jwt.md": """\ +--- +name: jwt +description: JSON Web Token — signed authentication token format +--- + +# JWT + +JSON Web Token (RFC 7519). A compact, signed (JWS) or encrypted (JWE) +token used to assert identity and claims between parties. + +## Structure +- Header — `alg`, `typ`, `kid` +- Payload — claims: `iss`, `sub`, `aud`, `exp`, `iat` +- Signature + +## Related +Often issued by [[digest/concept/oauth2.md]] flows. + +derived_from:: [[daily/2026-05-15/auth-design/notes.md]] +""", + "digest/concept/oauth2.md": """\ +--- +name: oauth2 +description: OAuth 2.0 — delegated authorization framework +--- + +# OAuth 2.0 + +RFC 6749. A delegated authorization framework: a resource owner grants +a client limited access to a protected resource via an access token +issued by an authorization server. + +## Grant types +- Authorization code (with PKCE for public clients) +- Client credentials +- Refresh token + +derived_from:: [[daily/2026-05-10/oauth-intro/notes.md]] +""", + "digest/procedure/key-rotation.md": """\ +--- +name: key-rotation +description: Rotating signing keys for JWT issuance +--- + +# Key rotation (current — pre 2026-05-28 refactor) + +Procedure for rotating the signing key used by [[digest/concept/jwt.md]] +issuance. + +## Steps +1. Generate new keypair offline. +2. Publish the public key to the JWKS endpoint with a fresh `kid`. +3. Wait 24h for clients to refresh their JWKS cache. +4. Cut over the signer to the new private key. +5. Mark the old `kid` as deprecated; remove after 30 days. + +## Cadence +Default rotation cadence is **30 days**. Driven by historical practice; +no formal compliance requirement has tightened this so far. + +derived_from:: [[daily/2026-05-20/rotation-plan/notes.md]] +""", + "digest/preference/no-trailing-summary.md": """\ +--- +name: no-trailing-summary +description: 不要在回复末尾加总结段落 +--- + +# 不要在回复末尾加总结段落 + +用户能看 diff,不需要在回复末尾重述刚做的事。 + +derived_from:: [[daily/2026-05-01/style-feedback/notes.md]] +""", + # ----- daily provenance stubs (so the digest links don't dangle) ----- + "daily/2026-05-01/style-feedback/notes.md": """\ +--- +name: notes +description: style feedback to Claude on 2026-05-01 +--- + +# Style feedback (2026-05-01) + +每次任务结束都重述了一遍刚做的事——不需要,我能看 diff。以后直接停。 +""", + "daily/2026-05-10/oauth-intro/notes.md": """\ +--- +name: notes +description: OAuth 2.0 intro session +--- + +# OAuth 2.0 intro + +简介 grant types: authorization code (with PKCE), client credentials, +refresh token。重点放在 PKCE 是给 public clients 用的。 +""", + "daily/2026-05-15/auth-design/notes.md": """\ +--- +name: notes +description: initial auth design discussion +--- + +# Auth design + +讨论 JWT 的结构 (header / payload / signature) 和我们项目里的 claim +约定 (iss, sub, aud, exp, iat)。 +""", + "daily/2026-05-20/rotation-plan/notes.md": """\ +--- +name: notes +description: key rotation plan v1 +--- + +# Key rotation plan v1 + +定下当前的 5 步轮换流程:offline 生成 keypair → 发布到 JWKS (新 kid) +→ 等 24h cache → 切签发 → 30 天后清旧 kid。周期定 30 天。 +""", + # ----- the NEW daily note dreamer will be invoked on ----- + INPUT_PATH: """\ +--- +name: notes +description: auth refactor working notes — 2026-05-28 +--- + +# Auth refactor — 2026-05-28 + +## 决定:JWT 轮换周期改为 24 小时 + +今天确定把 JWT 签名密钥的轮换周期从 30 天压到 **24 小时**。原因是 +SOC2 合规审计批评:30 天的会话 token 太长,不满足"短期凭证"原则。 + +新流程不再依赖 JWKS cache 的 24h 等待,改成走 Redis 里的 `kid` +版本号实时下发。客户端在 token 验证失败时主动拉新 JWKS,而不是定 +时轮询。 + +(这条同时更新 JWT 概念笔记和 key-rotation 流程笔记。) + +## 新概念:kid 版本号机制 + +`kid` (key ID) 是 JWT header 里的字段。我们把它当成版本号来用: +Redis key `auth:jwks:current_kid` 保存当前活跃 kid;Auth Service +在签发 token 时读这个 key,客户端验证失败时也读这个 key 再拉对应 +的 public key。这样无须等 cache TTL。 + +## 顺带复习:OAuth 2.0 是什么 + +(为了帮新同学接住上下文,这里把 OAuth 2.0 简单重述一下,不引入 +新事实。)OAuth 2.0 (RFC 6749) 是一个委托授权框架:资源所有者 +允许 client 通过 authorization server 颁发的 access token 来有 +限度地访问受保护资源。常见 grant types: authorization code +(public client 用 PKCE)、client credentials、refresh token。 +——这一段没有任何新内容,纯粹是给后面 JWT 24h 轮换决定铺垫读者 +的背景知识。 + +## 观察:SOC2 审计在 30 天周期上的具体批评 + +审计员引用 SOC2 CC6.1 控制点:"会话凭证应有合理的短期有效期"。 +30 天对应于人类工作周期,但对自动化客户端 token 来说过长。审计 +要求 24h 或更短,且必须能在事件响应时立即吊销 (kid 切换可满足)。 + +## 偏好:小 PR 优先 + +后续这个 refactor 拆 PR 时,每个 PR 控制在 < 300 行。原因是 review +负担太大时容易被拍脑袋通过,这违背了 SOC2 审计中变更管理的精神。 + +## 偏好:回复结尾再补充 + +之前说过不要总结段落 (我能看 diff),今天再补充一点:也不要"接下来 +的步骤"列表,除非我明确问 next steps。直接回答问题然后停。 +""", +} + + +_CLEAN_DIRS = ("daily", "digest", "reme_metadata") + + +def clean_vault(vault: Path) -> list[str]: + """Remove fixture-managed subdirs (`daily/`, `digest/`, `reme_metadata/`) + under `vault` so the next `seed_vault` starts from a clean slate. + + Returns the relative paths that were actually removed.""" + removed: list[str] = [] + for rel in _CLEAN_DIRS: + target = vault / rel + if target.exists(): + shutil.rmtree(target) + removed.append(rel) + return removed + + +def seed_vault(vault: Path) -> list[str]: + """Write any missing fixture files under `vault`. Return relative paths + that were actually written (skipped existing ones).""" + seeded: list[str] = [] + for rel, body in _FILES.items(): + target = vault / rel + if target.exists(): + continue + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(body, encoding="utf-8") + seeded.append(rel) + return seeded + + +def main() -> None: + if len(sys.argv) < 2: + print(f"usage: {sys.argv[0]} ", file=sys.stderr) + sys.exit(2) + vault = Path(sys.argv[1]).resolve() + vault.mkdir(parents=True, exist_ok=True) + removed = clean_vault(vault) + if removed: + print(f"cleaned {len(removed)} dir(s) under {vault}: {', '.join(removed)}") + seeded = seed_vault(vault) + if seeded: + print(f"seeded {len(seeded)} file(s) under {vault}:") + for f in seeded: + print(f" + {f}") + else: + print(f"vault {vault} already seeded — no changes") + print(f"\nDream this file:\n {vault}/{INPUT_PATH}") + + +if __name__ == "__main__": + main() diff --git a/tests4/smoke/test_dreamer_cli.sh b/tests4/smoke/test_dreamer_cli.sh new file mode 100755 index 00000000..eadcbe60 --- /dev/null +++ b/tests4/smoke/test_dreamer_cli.sh @@ -0,0 +1,68 @@ +#!/usr/bin/env bash +# dreamer CLI smoke test (option B). +# +# Seeds a rich workspace via _dreamer_fixture.py, starts `reme start` +# bound to that vault, reindexes so Phase 2 recall can hit the pre- +# seeded digest nodes, then calls `reme dream`. +# +# Usage (from anywhere): +# VAULT_PATH=/tmp/reme-dreamer-test bash tests4/smoke/test_dreamer_cli.sh +# VAULT_PATH=/tmp/reme-dreamer-test bash tests4/smoke/test_dreamer_cli.sh daily/2026-05-28/auth-refactor/notes.md +# +# Defaults: +# VAULT_PATH unset → /tmp/reme-dreamer-test +# Workspace seeded on first run (idempotent). +# +# Required env (from .env or shell): +# LLM_API_KEY, LLM_BASE_URL, LLM_MODEL_NAME +set -euo pipefail + +VAULT="${VAULT_PATH:-/tmp/reme-dreamer-test}" +SMOKE_DIR="$(cd "$(dirname "$0")" && pwd)" +REPO="$(cd "$SMOKE_DIR/../.." && pwd)" +LOG="/tmp/test_dreamer_cli_server.log" + +# Resolve input from arg, else default to fixture's input path. +DEFAULT_INPUT="$(python -c "import sys; sys.path.insert(0, '$SMOKE_DIR'); from _dreamer_fixture import INPUT_PATH; print(INPUT_PATH)")" +INPUT="${1:-$DEFAULT_INPUT}" + +mkdir -p "$VAULT" +echo "--- seeding fixture under $VAULT" +python "$SMOKE_DIR/_dreamer_fixture.py" "$VAULT" + +cd "$REPO" + +echo "" +echo "--- starting reme server (log: $LOG)" +reme start "vault_dir=$VAULT" >"$LOG" 2>&1 & +SERVER_PID=$! +trap 'echo "--- stopping reme server (pid $SERVER_PID)"; kill "$SERVER_PID" 2>/dev/null || true; wait "$SERVER_PID" 2>/dev/null || true' EXIT + +echo "--- waiting for server" +for _ in $(seq 1 30); do + if curl -s -o /dev/null http://localhost:8000/docs 2>/dev/null; then + echo "--- server up" + break + fi + sleep 0.5 +done + +echo "" +echo "--- reindexing vault so Phase 2 recall has something to hit" +reme reindex + +echo "" +echo "=== reme dream path=$INPUT ===" +reme dream "path=$INPUT" +echo "" + +echo "=== digest/ tree after dream ===" +if [ -d "$VAULT/digest" ]; then + find "$VAULT/digest" -name "*.md" | sort | while read -r f; do + echo "" + echo "--- ${f#$VAULT/} ---" + cat "$f" + done +else + echo " (no digest/ created)" +fi diff --git a/tests4/smoke/test_dreamer_inproc.py b/tests4/smoke/test_dreamer_inproc.py new file mode 100644 index 00000000..25516934 --- /dev/null +++ b/tests4/smoke/test_dreamer_inproc.py @@ -0,0 +1,102 @@ +"""dreamer in-process smoke test (option C). + +Loads the default reme4 config, seeds a rich workspace (pre-existing +digest nodes + a new daily that should drive both UPDATE and CREATE +through Phase 1 classify → Phase 2 per-kind recall+write), reindexes +so search_step can hit the pre-existing nodes, then calls `dream` and +prints what happened. + +Usage (from anywhere): + VAULT_PATH=/tmp/reme-dreamer-test python tests4/smoke/test_dreamer_inproc.py + VAULT_PATH=/tmp/reme-dreamer-test python tests4/smoke/test_dreamer_inproc.py daily/2026-05-28/auth-refactor/notes.md + +Defaults: + VAULT_PATH unset → /tmp/reme-dreamer-test + Each run wipes `daily/`, `digest/`, and `reme_metadata/` under the + vault before reseeding, so the dreamer always starts from the same + fixture state. See _dreamer_fixture.py for what gets created. + +Required env (from .env or shell): + LLM_API_KEY, LLM_BASE_URL, LLM_MODEL_NAME — for the Phase 1/2 agents +""" + +import asyncio +import os +import sys +from pathlib import Path + +# Make `reme4` importable regardless of the caller's cwd; and make the +# fixture module importable as a top-level name. +REPO_ROOT = Path(__file__).resolve().parents[2] +SMOKE_DIR = Path(__file__).resolve().parent +sys.path.insert(0, str(REPO_ROOT)) +sys.path.insert(0, str(SMOKE_DIR)) + +from _dreamer_fixture import clean_vault, seed_vault, INPUT_PATH # noqa: E402 + +VAULT = os.environ.get("VAULT_PATH", "/tmp/reme-dreamer-test") + + +async def main() -> None: + from reme4 import ReMe # noqa: E402 + from reme4.config import resolve_app_config # noqa: E402 + from reme4.utils import load_env # noqa: E402 + + os.chdir(REPO_ROOT) # so load_env() picks up the repo's .env + load_env() + + vault = Path(VAULT).resolve() + vault.mkdir(parents=True, exist_ok=True) + + removed = clean_vault(vault) + if removed: + print(f"--- cleaned {len(removed)} dir(s) under {vault}: {', '.join(removed)}") + seeded = seed_vault(vault) + print(f"--- seeded {len(seeded)} fixture file(s) under {vault}") + + rel_input = sys.argv[1] if len(sys.argv) > 1 else INPUT_PATH + + cfg = resolve_app_config(vault_dir=str(vault)) + print(f"--- vault_dir: {cfg.get('vault_dir')}") + print(f"--- input: {rel_input}") + + app = ReMe(**cfg) + await app.start() + try: + # Reindex first so search_step can actually find the pre-seeded + # digest/ nodes — otherwise Phase 2 recall returns empty and + # every atomic unit ends up as CREATE (UPDATE path not exercised). + print("\n--- reindexing vault so Phase 2 recall has something to hit") + await app.run_job("reindex") + + print(f"\n--- running dream path={rel_input}") + resp = await app.run_job("dream", path=rel_input) + + print("\n=== Response.success ===") + print(resp.success) + print("\n=== Response.answer ===") + print(resp.answer) + print("\n=== Response.metadata (DreamResult fields) ===") + for k, v in (resp.metadata or {}).items(): + if isinstance(v, list) and len(v) > 8: + print(f" {k}: list({len(v)} items) head={v[:3]!r}") + else: + print(f" {k}: {v!r}") + finally: + await app.close() + + print("\n=== digest/ tree after dream ===") + digest_root = vault / "digest" + if not digest_root.exists(): + print(" (no digest/ created)") + return + files = sorted(digest_root.rglob("*.md")) + if not files: + print(" (digest/ is empty)") + for p in files: + print(f"\n--- {p.relative_to(vault)} ---") + # print(p.read_text(encoding="utf-8")) + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/tests4/unit/test_resource_steps.py b/tests4/unit/test_resource_steps.py index dbdf77b8..9fba9f00 100644 --- a/tests4/unit/test_resource_steps.py +++ b/tests4/unit/test_resource_steps.py @@ -542,7 +542,7 @@ def test_upload_preserves_description_verbatim_in_meta(): payload = _metadata(step) assert "error" not in payload, payload - # meta.json preserves the original — downstream digester sees full hint. + # meta.json preserves the original — downstream dreamer sees full hint. meta = _meta(tmp, payload["date"]) assert meta[0]["front_matter"]["description"] == multi