diff --git a/reme4/config/default.yaml b/reme4/config/default.yaml index dd9bc653..4f1f5552 100644 --- a/reme4/config/default.yaml +++ b/reme4/config/default.yaml @@ -387,9 +387,9 @@ jobs: steps: - backend: dreamer_step - dream_today: + auto-dream: backend: base - description: "Dream-today: scan // + // and dream each file (LLM). Subroots resolved from app config." + description: "Auto-dream: scan today's day-index /.md and session notes under //*.md — run dream on each (Phase 1 extract+classify, Phase 2 per-bucket integrate)." parameters: type: object properties: diff --git a/reme4/steps/evolve/auto_dream.py b/reme4/steps/evolve/auto_dream.py index abbd949d..9b61f3ae 100644 --- a/reme4/steps/evolve/auto_dream.py +++ b/reme4/steps/evolve/auto_dream.py @@ -16,14 +16,14 @@ Pipeline (external loop in Python, two distinct ReAct agent invocations, **light Phase 1 / heavy Phase 2**): execute(): - _extract(material_blob) # 1× ReAct: identify abstractions - # agent emits ExtractedUnits - # ({units: [{name, bucket, summary}, ...]}) - for unit in self._units: # Python loop, K iterations - _integrate_unit(unit) # 1× ReAct per abstraction, dispatched - # to integrate_system_prompt_; - # recalls cross-bucket, decides write, - # uses canonical write/edit tools. + units, _ = _extract(material_blob) # 1× ReAct: identify abstractions + # agent emits ExtractedUnits + # ({units: [{name, bucket, summary}, ...]}) + for unit in units: # Python loop, K iterations + _integrate_unit(unit) # 1× ReAct per abstraction, dispatched + # to integrate_system_prompt_; + # recalls cross-bucket, decides write, + # uses canonical write/edit/frontmatter_update tools. The bucket vocabulary is hard-coded (:data:`BUCKETS`) — three buckets, each with a dedicated Phase 2 prompt: @@ -52,8 +52,8 @@ import zoneinfo from pathlib import Path from typing import Literal -from agentscope.message import Msg, TextBlock -from agentscope.tool import Toolkit, ToolResponse +from agentscope.message import Msg +from agentscope.tool import Toolkit from pydantic import BaseModel, Field from ._evolve import FlexReActAgent @@ -71,18 +71,18 @@ BUCKETS: tuple[str, ...] = ("procedure", "personal", "wiki") Bucket = Literal["procedure", "personal", "wiki"] -_EXTRACT_READ_TOOLS: tuple[str, ...] = ("read",) +_EXTRACT_TOOLS: tuple[str, ...] = ("read",) -_INTEGRATE_READ_TOOLS: tuple[str, ...] = ( +_INTEGRATE_TOOLS: tuple[str, ...] = ( + # read "search", "traverse", "read", "frontmatter_read", -) - -_INTEGRATE_WRITE_TOOLS: tuple[str, ...] = ( + # write "write", "edit", + "frontmatter_update", ) @@ -240,10 +240,6 @@ class Dreamer(BaseStep): self.toolkit = toolkit self.console_enabled = console_enabled self.timezone = timezone - # Per-invocation outcome trackers, populated by tool callbacks. - self._units: list[dict] = [] - self._created: list[str] = [] - self._updated: list[str] = [] def _now(self) -> datetime.datetime: if self.timezone: @@ -263,97 +259,33 @@ class Dreamer(BaseStep): except Exception: return False - def _make_write_tool(self): - """Wrap the canonical ``write`` job with create-tracking. - - Same shape as the underlying job; the wrapper just records - successful paths into ``self._created`` so the dreamer can - reconstruct outcomes when the LLM drops its structured emission. - """ - job = self.get_job("write") - if job is None: - raise RuntimeError("write job not registered") - - async def write(path: str, name: str, description: str, content: str) -> ToolResponse: - resp = await job( - path=path, - name=name, - description=description, - content=content, - ) - if resp.success: - self._created.append(path) - return ToolResponse(content=[TextBlock(type="text", text=resp.answer)]) - - return write, job - - def _make_edit_tool(self): - """Wrap the canonical ``edit`` job with update-tracking.""" - job = self.get_job("edit") - if job is None: - raise RuntimeError("edit job not registered") - - async def edit(path: str, old: str, new: str) -> ToolResponse: - resp = await job(path=path, old=old, new=new) - if resp.success: - self._updated.append(path) - return ToolResponse(content=[TextBlock(type="text", text=resp.answer)]) - - return edit, job - def _build_extract_toolkit(self) -> Toolkit: """Read-only toolkit for the extract agent. Sub-units come back via :class:`ExtractedUnits` structured output, not via a tool call.""" toolkit = Toolkit() - for job_name in _EXTRACT_READ_TOOLS: + for job_name in _EXTRACT_TOOLS: self.add_as_tool(toolkit, job_name) return toolkit def _build_integrate_toolkit(self) -> Toolkit: - """Full read + canonical write/edit toolkit for the integrate agent. - - write/edit are wrapped in tracker closures (created/updated paths) - so the outer loop can reconstruct outcomes when an LLM call drops - its structured emission. Read-only tools go through ``add_as_tool`` - unchanged. - """ + """Full read + canonical write/edit/frontmatter_update toolkit for + the integrate agent. All tools are registered via :meth:`add_as_tool` + — same as every other step in this codebase. Outcome tracking is + driven by the agent's :class:`IntegrateOutcome` structured emission, + not by per-tool callbacks.""" toolkit = self.toolkit or Toolkit() - for job_name in _INTEGRATE_READ_TOOLS: + for job_name in _INTEGRATE_TOOLS: self.add_as_tool(toolkit, job_name) - - write_tool, write_job = self._make_write_tool() - toolkit.register_tool_function( - tool_func=write_tool, - func_name="write", - func_description=write_job.description, - json_schema={ - "type": "function", - "function": { - "name": "write", - "description": write_job.description, - "parameters": write_job.parameters, - }, - }, - ) - - edit_tool, edit_job = self._make_edit_tool() - toolkit.register_tool_function( - tool_func=edit_tool, - func_name="edit", - func_description=edit_job.description, - json_schema={ - "type": "function", - "function": { - "name": "edit", - "description": edit_job.description, - "parameters": edit_job.parameters, - }, - }, - ) return toolkit - async def _extract(self, material_blob: str, hint: str, vault_dir: Path) -> str: - """Phase 1: one ReAct invocation — read material + emit ExtractedUnits. Returns LLM summary.""" + async def _extract(self, material_blob: str, hint: str, vault_dir: Path) -> tuple[list[dict], str]: + """Phase 1: one ReAct invocation — read material + emit ExtractedUnits. + + Returns ``(units, llm_summary)`` where ``units`` is the cleaned + sub-unit list (each entry has ``name`` / ``bucket`` / ``summary``) + and ``llm_summary`` is whatever free-form text the agent produced + alongside its structured emission. + """ toolkit = self._build_extract_toolkit() agent = FlexReActAgent( name="reme_dreamer_extract", @@ -399,21 +331,14 @@ class Dreamer(BaseStep): ) bucket = "wiki" cleaned.append({"name": name, "summary": summary, "bucket": bucket}) - self._units = cleaned - return (msg.get_text_content() or "").strip() + return cleaned, (msg.get_text_content() or "").strip() async def _integrate_unit(self, unit: dict, material_blob: str, hint: str, vault_dir: Path) -> IntegrateOutcome: """One ReAct invocation per memory sub-unit, dispatched to the bucket-specific system prompt. Returns the parsed - :class:`IntegrateOutcome` reported by the agent. - - File writes happen as side effects via the canonical ``write`` / - ``edit`` tool calls (which populate ``self._created`` / - ``self._updated`` via the tracker closures); the structured - outcome here is the agent's own summary of what it decided — - useful for rendering and for catching hallucinations (action= - CREATE without the matching write call landing in trackers). - """ + :class:`IntegrateOutcome` reported by the agent — that's the + single source of truth for what got written (action + + target_path).""" bucket = unit.get("bucket") or "wiki" toolkit = self._build_integrate_toolkit() digest_dir = getattr(self.app_context.app_config, "digest_dir", "") @@ -438,29 +363,12 @@ class Dreamer(BaseStep): unit_summary=unit.get("summary", ""), material_blob=material_blob, ) - # Snapshot trackers so we can reconstruct the outcome from the - # filesystem side effects if the agent's structured emission slips. - created_before = len(self._created) - updated_before = len(self._updated) msg = await agent.reply( Msg(name="reme", role="user", content=user_message), structured_model=IntegrateOutcome, ) meta = msg.metadata if isinstance(msg.metadata, dict) else {} - try: - return IntegrateOutcome.model_validate(meta) - except Exception: - # The LLM occasionally drops the final structured emission even - # after a successful tool call. The trackers are the source of - # truth — reconstruct the outcome from the new entries this - # session added. - new_created = self._created[created_before:] - new_updated = self._updated[updated_before:] - if new_created: - return IntegrateOutcome(action="CREATE", target_path=new_created[-1]) - if new_updated: - return IntegrateOutcome(action="CORROBORATE", target_path=new_updated[-1]) - raise + return IntegrateOutcome.model_validate(meta) async def dream_one(self, path: str, hint: str = "") -> DreamResult: """Run the full extract + integrate pipeline on one vault-relative @@ -486,19 +394,14 @@ class Dreamer(BaseStep): material_blob = _pack_material(self.file_store, path) - # Reset per-invocation trackers. - self._units.clear() - self._created.clear() - self._updated.clear() - vault_dir = self._vault_dir() # Phase 1 — extract (light). Agent emits ExtractedUnits structured output to commit the # memory sub-units worth lifting. Each unit carries its own bucket. self.logger.info(f"[{self.name}] extract phase: path={path!r}") - extract_summary = await self._extract(material_blob, hint, vault_dir) + units, extract_summary = await self._extract(material_blob, hint, vault_dir) - if not self._units: + if not units: return DreamResult( used_llm=True, path=path, @@ -506,42 +409,47 @@ class Dreamer(BaseStep): skipped=True, ) - unit_handles = ", ".join(f"{u['name']}/{u['bucket']}" for u in self._units) - self.logger.info(f"[{self.name}] integrate phase: {len(self._units)} sub-unit(s): {unit_handles}") + unit_handles = ", ".join(f"{u['name']}/{u['bucket']}" for u in units) + self.logger.info(f"[{self.name}] integrate phase: {len(units)} sub-unit(s): {unit_handles}") # Phase 2 — integrate, one fresh ReAct per sub-unit, dispatched to # the bucket-specific system prompt. Python-level loop, not agent - # loop. Each session emits a structured IntegrateOutcome; file - # writes happen as side effects via the canonical write / edit - # tool calls. + # loop. Each session emits a structured IntegrateOutcome whose + # action + target_path are the source of truth for what landed. + nodes_created: list[str] = [] + nodes_updated: list[str] = [] per_unit_lines: list[str] = [] - for i, unit in enumerate(self._units, start=1): + for i, unit in enumerate(units, start=1): name = unit.get("name", "?") bucket = unit.get("bucket", "?") try: outcome = await self._integrate_unit(unit, material_blob, hint, vault_dir) except Exception as e: self.logger.error( - f"[{self.name}] integrate {i}/{len(self._units)} " + f"[{self.name}] integrate {i}/{len(units)} " f"(unit={name}, bucket={bucket}) failed: {type(e).__name__}: {e}", ) per_unit_lines.append(f"[{name}/{bucket}] FAILED: {type(e).__name__}: {e}") continue + if outcome.action == "CREATE": + nodes_created.append(outcome.target_path) + else: + nodes_updated.append(outcome.target_path) per_unit_lines.append(_render_outcome_line(name, bucket, outcome)) per_unit_block = "\n".join(per_unit_lines) summary = ( - f"Declared {len(self._units)} sub-unit(s) ({unit_handles}); " - f"created {len(self._created)}, updated {len(self._updated)}.\n" + f"Declared {len(units)} sub-unit(s) ({unit_handles}); " + f"created {len(nodes_created)}, updated {len(nodes_updated)}.\n" f"{per_unit_block}" ) return DreamResult( used_llm=True, path=path, - units=list(self._units), - nodes_created=list(self._created), - nodes_updated=list(self._updated), + units=units, + nodes_created=nodes_created, + nodes_updated=nodes_updated, summary=summary, skipped=False, ) diff --git a/reme4/steps/evolve/auto_dream.yaml b/reme4/steps/evolve/auto_dream.yaml index 64bc3731..b91ee2f9 100644 --- a/reme4/steps/evolve/auto_dream.yaml +++ b/reme4/steps/evolve/auto_dream.yaml @@ -1,98 +1,92 @@ extract_system_prompt: | - You are the **dreamer** — EXTRACT phase. Your ONLY job here is to - read the material, identify the ABSTRACTIONS it teaches — the - principles, patterns, decisions-as-precedent, cognitive takeaways - that belong in long-term memory — and **classify each into one - of three buckets**. You commit the result via the structured - output schema attached to this call (an `ExtractedUnits` object). - You do NOT do recall, integrate, or write. A separate downstream - invocation processes each unit (with its bucket determining which - Phase 2 prompt runs). + You are Phase 1 of dream — read the material, identify the + ABSTRACTIONS it teaches, and tag each with a bucket. Phase 2 + picks the slug and writes; you only declare what's worth lifting. vault_dir: {vault_dir} ## What digest memory is for Digest is the **abstract memory layer** — analogous to the - prefrontal cortex aggregating cognition. The raw details of - what happened (numbers, narratives, who said what, full - procedure text) STAY IN THE MATERIAL. Digest holds the - generalized lesson the reader should recall next time — - the part that survives once the specific event fades. + prefrontal cortex aggregating cognition. Raw details (numbers, + narratives, who said what, full procedure text) STAY IN THE + MATERIAL. Digest holds the generalized lesson the reader should + recall next time — the part that survives once the specific + event fades. - When you cluster, you are NOT cataloguing the material's - contents — you are answering: *"What abstractions does - this material teach that I'd want a future agent / human - to have at-hand when facing a similar situation?"* + When you cluster, you are NOT cataloguing the material's contents + — you are answering: *"what abstractions does this material teach + that I'd want a future agent / human to have at-hand when facing + a similar situation?"* ## What is a memory sub-unit? - One sub-unit = one abstraction the material teaches. **One - sub-unit maps to exactly one digest node** — Phase 2 will - make one write decision per sub-unit (CREATE or one of the - three UPDATE flavors). Phase 1 is the gate for "not worth - memorizing"; once a sub-unit reaches Phase 2 it WILL be - written. + One sub-unit = one abstraction the material teaches. **One sub-unit + maps to exactly one digest node** — Phase 2 makes one write + decision per sub-unit (CREATE or one of the three UPDATE flavors). + Phase 1 is the gate for "not worth memorizing"; once a sub-unit + reaches Phase 2 it WILL be written. - Multiple raw facts in the material that all illustrate the - same abstraction collapse to ONE sub-unit. The Redis-kid - versioning mechanism, the SOC2 CC6.1 rationale, and the new - 24h cadence are three FACTS, but they teach one abstraction: - "JWT rotation cadence is driven by short-credential - compliance, not by procedural convenience". That's one - sub-unit. The mechanism / numbers / RFC citation are - details — they stay in the daily note, the digest reaches - them through `derived_from::` provenance edges. + Multiple raw facts in the material that all illustrate the same + abstraction collapse to ONE sub-unit. Example: the kid-versioning + mechanism, the SOC2 CC6.1 rationale, and the new 24h cadence are + three FACTS, but they teach one abstraction — "short-credential + compliance drives auth cadence, not procedural convenience". + That's one sub-unit. The mechanism / numbers / RFC citation are + details — they stay in the daily note; the digest reaches them + through `derived_from::` provenance edges. - Sub-units are NOT bucket names, NOT kinds, NOT the eventual - digest slug — they are an agent-internal handle for the - abstraction you've identified. Phase 2 picks the slug + write - decision per sub-unit; YOU pick the bucket here in Phase 1. + Sub-units are NOT bucket names, NOT kinds, NOT the eventual digest + slug — they're an agent-internal handle for the abstraction you've + identified. Phase 2 picks the slug + write decision; YOU pick the + bucket here. ### Bias: fewer, richer sub-units over many narrow ones - This is the abstract layer — heavy lifting toward few - high-leverage sub-units, not toward exhaustive coverage. - Heuristic for splitting two pieces into two sub-units vs - one: + This is the abstract layer — heavy lifting toward few high-leverage + units, not toward exhaustive coverage. Heuristic for splitting two + pieces into two units vs one: - * Same abstraction shown by different facts? → ONE sub-unit. - * Genuinely different abstractions that a future reader - would invoke in DIFFERENT situations? → TWO sub-units. + * Same abstraction shown by different facts? → ONE unit. + * Genuinely different abstractions a future reader would invoke + in DIFFERENT situations? → TWO units. * Will they evolve independently as more materials arrive? - → TWO sub-units. + → TWO units. - When in doubt, KEEP TOGETHER (or DROP one of them entirely). + When in doubt, MERGE (or drop one of them entirely). ### What NOT to declare - - Passing mentions with no new abstraction — daily-note - indexing already covers detail-level recall. + - Passing mentions with no new abstraction (e.g. an OAuth recap + that restates a known concept) — daily-note indexing already + covers detail-level recall. - Facts whose only audience is the material itself (one-off timestamps, single meeting attendance) — not an abstraction. - - Event-level umbrella sub-units (e.g. "X-event-summary") — - every sub-unit already carries a `derived_from:: [[]]` - wikilink, so the material itself is the fan-out point. + - Event-level umbrella sub-units (e.g. `X-event-summary`) — + every sub-unit already carries + `derived_from:: [[]]`, so the material itself + is the fan-out point linking to all its derived nodes; the + umbrella adds nothing. - ## Bucket vocabulary (HARD-CODED — pick exactly one per unit) + ## Bucket — pick exactly one per unit The bucket determines which specialized Phase 2 prompt processes - this sub-unit. Three buckets, picked by *what kind of abstraction* - this is — NOT by the material's surface topic. + this sub-unit. Pick by *kind of abstraction*, not by surface + topic. - **`procedure`** — *how to do X*. Steps, methods, recipes, - workflows, runbooks, executable patterns. The reader's question - is "how do I accomplish Y?" Pick this when the abstraction - is an actionable sequence or technique. + workflows, runbooks, executable patterns. Reader's question: + "how do I accomplish Y?" Pick when the abstraction is an + actionable sequence or technique. Examples: "key-rotation procedure", "incident triage flow", "how to wire up a new MCP tool". - **`personal`** — *user/team-specific facts about how WE work*. - Identity ("who is X"), preferences ("user prefers terse replies"), - conventions ("we use kebab-case for slug names"), things to - avoid ("don't run schema migrations on Friday"), collaboration - style. The reader's question is "what does THIS user / team - want / do / dislike?" Pick this when the abstraction is only + Identity ("who is X"), preferences ("user prefers terse + replies"), conventions ("we use kebab-case for slug names"), + things to avoid ("don't run schema migrations on Friday"), + collaboration style. Reader's question: "what does THIS user / + team want / do / dislike?" Pick when the abstraction is only valid in the context of this user / team / project. Examples: "huangsen prefers short PRs", "team avoids mocking the DB in integration tests", "we don't write @@ -100,58 +94,31 @@ extract_system_prompt: | - **`wiki`** — *general knowledge*. Definitions, principles, observations, decisions-as-precedent, factual claims, mental - models. The reader's question is "what IS X / what happened - / what was decided?" Pick this when the abstraction is true - independent of which user is reading. Also the **default - catch-all** when nothing else fits cleanly. + models. Reader's question: "what IS X / what happened / what + was decided?" Pick when the abstraction is true independent + of who's reading. Also the **default catch-all** when nothing + else fits cleanly. Examples: "JWT is a signed token format", "short-credential compliance drives auth cadence", "moving to 24h refresh reduced p99 latency by 12%". - Picking heuristic when a unit straddles two buckets: pick by - CENTER OF GRAVITY — what would a future reader most likely be - searching for, and from which mindset? "User prefers small PRs" - is *personal*, not *wiki*, because the lesson only applies to - this user. "Small PRs are easier to review" is *wiki* — it's a - general claim. "Steps to split a large PR" is *procedure*. + Straddling two buckets → pick by **center of gravity** (which + bucket the future reader will search from): + - "User prefers small PRs" → personal (rule for THIS user). + - "Small PRs are easier to review" → wiki (general claim). + - "Steps to split a large PR" → procedure. - Available buckets: {buckets} + Available: {buckets} - ## What to do + ## Output - 1. **Read the material** — its body is packed in the user - message below. If it references `[[resource//]]` - and that asset is critical to understanding what - abstractions are present, you MAY open it via `read`; - otherwise skip external reads (this is the light phase). + Each unit's `summary` should name the abstraction AND point at + where in the material the supporting evidence lives — Phase 2 + cites it as provenance without re-reading. Field shapes are + enforced by the structured-output schema. - 2. **Identify the abstractions** the material teaches. - For each candidate, ask: *if I forgot all the details - of this material in 6 months, what one-line lesson - would I still want to recall?* That lesson is a - sub-unit candidate. - - 3. **Tag each unit with a bucket** (procedure / personal / - wiki). This routes Phase 2 to the right specialized prompt. - - 4. **Emit the surviving list** as your structured output. Each - entry's `summary` should be concrete about WHERE in the - material the supporting evidence lives, so Phase 2 can cite - it as provenance without re-reading. Field shapes are - enforced by the schema attached to this call. - - If the material teaches no new abstraction worth long-term - memory (e.g. routine status updates, pure logs), emit an empty - unit list. - - ## Boundaries - - - You CANNOT write to digest in this phase (no write/edit tools - here). - - You CANNOT do recall in this phase (no search/traverse here). - - You declare ABSTRACTIONS (sub-units), not detail copies. - - The structured output you emit is the final scope for this - dream call. + You have read-only access (`read` for inline `[[resource/...]]` + references that genuinely matter); no recall, no writing. extract_user_message: | today: {today} @@ -161,518 +128,297 @@ extract_user_message: | {material_blob} - Identify the ABSTRACTIONS this material teaches, classify each + Identify the abstractions this material teaches, classify each into one of {{procedure, personal, wiki}}, and emit via the structured output schema. Empty list if nothing new is taught. # ============================================================ # Phase 2 — bucket-specific INTEGRATE prompts. -# Each bucket has its OWN fully self-contained system prompt. -# Dispatcher in dreamer.py picks integrate_system_prompt_ -# based on the bucket Phase 1 assigned to the unit. +# Dispatcher picks integrate_system_prompt_ from the +# bucket Phase 1 assigned to the unit. # ============================================================ integrate_system_prompt_procedure: | - You are the **dreamer** — INTEGRATE phase, **procedure bucket**. + You are Phase 2 of dream, **procedure** bucket. The unit is a + how-to-do-X (steps, methods, recipes, runbooks, executable + patterns). Recall cross-bucket, decide CREATE / CORROBORATE / + REFINE / CORRECT, write exactly once. Sub-unit ↔ digest node + is 1:1; no SKIP outcome — Phase 1 already gated. - This invocation processes ONE memory sub-unit whose abstraction - is a *procedure*: a how-to-do-X — steps, methods, recipes, - workflows, runbooks. The reader's question against this node - later will be "how do I accomplish Y?" The completed material - is in the user message; Phase 1 already pointed you at the - supporting evidence. Your job: recall existing digest nodes - (cross-bucket), decide between CREATE and the three UPDATE - flavors (CORROBORATE / REFINE / CORRECT), and write — **with - a procedure-shaped body**. + vault_dir: {vault_dir} + digest_dir: {digest_dir} - **Sub-unit maps 1:1 to a digest node.** Exactly ONE write per - session — there is no "no-write" outcome. + ## Digest is the abstract memory layer + + Digest is NOT a faithful copy of the material — it's the cognitive + aggregation (think prefrontal cortex). Details stay in the daily + / resource file; digest holds the principle, pattern, or + precedent the agent should recall later. + + - **Body is SHORT and abstract** (≈ 50-200 words for most nodes; + longer only when the concept genuinely needs it). If your draft + starts copying paragraphs from the material, you're filing + detail in the wrong layer. + - **Provenance edges carry the details.** Whenever this + abstraction is illustrated by a specific material, add a + `derived_from:: [[daily/...]]` or `[[resource/...]]` wikilink + — readers drill down through the edge, not through re-stated + facts in the body. + - **Wikilinks between digest nodes** carry the conceptual graph + (`relates_to::`, `depends_on::`, `is_a::`, …). ## Procedure-bucket body shape - A procedure node body should read like a runbook, not a recap. - Keep it short and actionable: + A runbook, not a recap: - - **Trigger / when to use**: 1 line — under what conditions + - **Trigger / when to use** (1 line) — under what conditions does the reader reach for this procedure? - - **Steps**: a numbered or terse bulleted list. Each step is - one verb-led imperative. Optional inline justification - ("because X locks the row before Y commits"). - - **Pre-conditions / inputs**: one short list, not prose. - - **Failure modes / caveats**: brief — "if step 3 returns - ROLLBACK, restart from step 1" type notes. NOT a transcript - of every observed failure. - - **At least one `derived_from:: [[]]`** so - the procedure is traceable to the material that taught it. + - **Steps** — numbered or terse bullets; each is one verb-led + imperative. Optional inline justification ("because X locks + the row before Y commits") is fine. + - **Pre-conditions / inputs** — short list, not prose. + - **Failure modes / caveats** — brief ("if step 3 returns + ROLLBACK, restart from step 1"); NOT a transcript of every + observed failure. + - **`derived_from:: [[]]`** — at least one. + Plain-prose provenance does NOT count (only wikilinks survive + future updates). - Body is short (≈ 50-200 words). If your draft starts copying - paragraphs of conversation or full code blocks from the - material, you're filing detail in the wrong layer — those - belong in the daily / resource file; the digest only carries - the generalizable runbook. + ## Recall → decide → write - ## What to do + 1. **Recall** — `search` (include verb stems: rotate, migrate, + deploy…) + `traverse depth=2 direction=both` on any hit under + `{digest_dir}/`. Cross-bucket on purpose: an existing match + filed elsewhere beats a duplicate. + 2. **Hit** — `frontmatter_read` triage, then `read` body for + survivors. Same procedure = same trigger + substantially + overlapping steps. New step or one-line nuance is REFINE, + not "different procedure". + 3. **Decide** (exactly one): + - empty hit set → **CREATE** at + `{digest_dir}/procedure/.md`. + - hit non-empty → **UPDATE** the best match: + - **CORROBORATE** — same procedure observed again; append + `derived_from::`, optionally strengthen wording + ("consistently used across N runs"). + - **REFINE** — new pre-condition / edge case / failure + mode; expand the relevant span, slot new steps into + the right position. + - **CORRECT** — wrong order, missing critical step, bad + outcome; tighten or annotate inline (`> note: + contradicted by [[new-material]] — `). - Two-stage flow: **RECALL** (cross-bucket; assemble candidate - paths) → **HIT** (confirm whether any candidate carries this - procedure): + ## Discipline - hit set empty ⇒ CREATE under {digest_dir}/procedure/ - hit set non-empty ⇒ UPDATE the best match - (CORROBORATE / REFINE / CORRECT) + - CREATE writes inside `{digest_dir}/procedure/`. Phase 1 chose + your bucket — don't pivot. + - UPDATE may target any bucket if RECALL legitimately matched. + - `edit` is body-only and **only-add, not-delete**: never drop + wikilinks the `old` span contained (provenance must accumulate, + not evaporate). + - `frontmatter_update` is the only way to change frontmatter + (e.g. tighten `description`, set `kind: procedure`). + - One target per session. Never edit other nodes sideways. - ### Stage 1 — RECALL (search + traverse; CROSS-BUCKET) - - Goal: surface candidate paths under `{digest_dir}/`. Recall is - intentionally cross-bucket — the same procedure may have been - filed elsewhere by an earlier dream pass, and updating it in - place beats creating a duplicate. - - - **`search`** — keyword + vector hits using the sub-unit's - likely slug + summary. Include verb stems ("rotate", - "migrate", "deploy") since procedure slugs lean that way. - - - **`traverse path= depth=2 direction=both`** — - graph expansion. Run this whenever `search` returned ANY - hit under `{digest_dir}/`, even if the top hit looks unrelated - by snippet alone — adjacent procedures often link from - related concept nodes. - - ### Stage 2 — HIT (frontmatter_read + read) - - Goal: for each candidate, decide whether it carries the same - procedure as your sub-unit. - - - **`frontmatter_read`** — peek `name` + `description`. Drop - candidates that are clearly different procedures (different - domain, different trigger). - - **`read`** — full body for survivors. Same procedure means: - same trigger AND substantially overlapping steps. Slight - variations in wording or one extra step is REFINE territory, - not "different procedure". - - Hit set = candidates whose body confirms the same procedure. - - ### Decision - - - **Hit set empty** ⇒ CREATE a new digest node under - `{digest_dir}/procedure/.md`. - - **Hit set non-empty** ⇒ UPDATE the best-matching hit: - - **CORROBORATE**: same procedure observed again → - append a `derived_from::` link, optionally strengthen - wording ("consistently used across N runs"); steps - unchanged. - - **REFINE**: procedure has new pre-condition, edge case, - or failure-mode addition → expand the relevant span; - new step or guard goes into the right slot of the - ordering; add provenance. - - **CORRECT**: procedure as previously stated has a - wrong order, missing critical step, or bad outcome → - tighten or annotate inline (`> note: contradicted by - [[new-material]] — `); add provenance. - - ### Tools - - - **`write(path, name, description, content)`** — for CREATE. - Canonical write job (no path-shape validation; you are - responsible for placing it correctly). - - `path` MUST be `{digest_dir}/procedure/.md`. Do - NOT write outside `procedure/` — your prompt is bucket- - specific because Phase 1 classified this unit as a - procedure. - - `name` is the frontmatter name (usually the slug). - - `description` is the one-line summary of the procedure. - - `content` is the body (no leading `---` frontmatter — - the step composes frontmatter from `name` + `description` - automatically). - If the path already exists, the write will overwrite — but - that should never happen for a CREATE, because RECALL would - have surfaced it as a hit. Re-do RECALL in that case. - - - **`edit(path, old, new)`** — for CORROBORATE / REFINE / - CORRECT. Body-only find-and-replace on an existing digest - node. - - `path` is the existing digest node (any bucket — UPDATE - may target a procedure node filed elsewhere if recall - legitimately matched). - - Pick `old` narrow but unique. Composition rule for - `new`: only-add, not-delete. Never drop wikilinks the - old span contained — provenance must accumulate, not - evaporate. - - You MAY issue more than one `edit` against the SAME - target if multiple sections need updating; never write - to a different target as a side-effect. - - Write only the target you committed to for this sub-unit. - Never edit other nodes' bodies sideways. - - ## Wikilink form - - Always full vault-relative path with `.md`: - - - `[[{digest_dir}//.md]]` - - `[[daily///.md]]` - - `[[resource//]]` - - Optional Dataview-style typed predicates (predicate sits - outside the brackets): - - - `derived_from:: [[daily/2026/05/15/auth-refactor.md]]` - - `depends_on:: [[{digest_dir}/wiki/jwt.md]]` - - inline: `relies on [depends_on:: [[{digest_dir}/wiki/jwt.md]]]` - - ## Provenance - - Body MUST weave at least one `derived_from:: [[daily/...]]` - or `[[resource/...]]` wikilink. Plain prose ("from yesterday's - notes") does NOT count — only wikilinks survive future updates. - This is the only way the procedure can be traced to its source. - - ## Frontmatter - - Reserved fields (both optional, both useful): - - - `name` — basename without extension - - `description` — one-line summary - - Do NOT write a `status` field — there is no distill-pass marker. - - ## Reporting your outcome - - After the file write lands, emit your decision through the - `IntegrateOutcome` schema attached to this call. `action` is - one of CREATE / CORROBORATE / REFINE / CORRECT, `target_path` - is the digest path you wrote to. Both fields mandatory. + Wikilinks are full vault-relative paths with `.md` + (`[[{digest_dir}//.md]]`, `[[daily/...]]`, + `[[resource/...]]`). Predicates are open + (`[A-Za-z][A-Za-z0-9_]*`) and live outside the brackets. integrate_system_prompt_personal: | - You are the **dreamer** — INTEGRATE phase, **personal bucket**. + You are Phase 2 of dream, **personal** bucket. The unit is + user/team-specific (identity, preference, convention, avoid-rule, + collaboration style). Recall cross-bucket, decide CREATE / + CORROBORATE / REFINE / CORRECT, write exactly once. Sub-unit ↔ + digest node is 1:1; no SKIP — Phase 1 already gated. - This invocation processes ONE memory sub-unit whose abstraction - is *user/team-specific*: identity (who someone is), preferences - (how they like to work), conventions (what the team follows), - avoid-rules (what they explicitly said NOT to do), collaboration - style. The reader's question against this node later will be - "what does THIS user / team want / do / dislike?" The full - material is in the user message; Phase 1 pointed you at the - evidence. Your job: recall existing digest nodes (cross-bucket), - decide between CREATE and the three UPDATE flavors, and write — - **with a personal-shaped body**. + vault_dir: {vault_dir} + digest_dir: {digest_dir} - **Sub-unit maps 1:1 to a digest node.** Exactly ONE write per - session. + ## Digest is the abstract memory layer + + Digest is NOT a faithful copy of the material — it's the cognitive + aggregation (think prefrontal cortex). Details stay in the daily + / resource file; digest holds the rule, identity, or convention + the agent should recall later. + + - **Body is SHORT and abstract** (≈ 50-200 words). If your draft + starts narrating *what the user said in detail*, you're filing + detail in the wrong layer. + - **Provenance edges carry the details.** Whenever this rule is + set, restated, or revised by a specific material, add a + `derived_from:: [[daily/...]]` wikilink — readers drill down + through the edge, not through re-stated context. + - **Wikilinks between digest nodes** carry the conceptual graph + (`applies_to::`, `relates_to::`, …). ## Personal-bucket body shape - Personal nodes are short rules-of-engagement, not biographies: + A short rule of engagement, not a biography: - - **The rule / fact**: one sentence stating the preference, + - **Rule / fact** — one sentence stating the preference, convention, or identity claim. - - **`Why:`**: the reason the user gave (or the inferable - motivation — e.g. a past incident, a constraint they - care about, a strong preference). Knowing *why* lets a - future reader judge edge cases instead of blindly applying. - - **`How to apply:`**: when this rule kicks in — which - contexts, which tasks, which boundaries. - - **At least one `derived_from:: [[]]`** so - the rule is traceable to the conversation / decision that - set it. + - **`Why:`** — the reason (a past incident, a constraint, a + strong preference). Knowing *why* lets future readers judge + edge cases instead of blindly applying. + - **`How to apply:`** — when this rule kicks in: which contexts, + tasks, boundaries. + - **`derived_from:: [[]]`** — at least one. + Plain-prose provenance does NOT count. - Body is short (≈ 50-200 words). If your draft starts narrating - *what the user said in detail*, you're filing detail in the - wrong layer — those belong in the daily file; the digest holds - the rule. - - ## Personal-bucket sub-shape: identity vs preference - - Two common kinds inside `personal/`: - - - *Identity* — biographical / role facts about the user or - team ("X is a backend engineer focused on observability"; - "team Y owns the auth subsystem"). Reader question: "who - is X?". - - *Preference / convention / avoid-rule* — how they like to - work / what they avoid. Reader question: "how does X like + Two common sub-shapes filed in this bucket: + - *Identity* — biographical / role facts ("X is a backend + engineer focused on observability"). Reader's question: + "who is X?". + - *Preference / convention / avoid-rule* — how someone likes + to work / what to skip. Reader's question: "how does X like to work / what should I not do?". - Both file under `{digest_dir}/personal/.md` — the - distinction is for your body framing, not for path layout. - When the same person has many preferences, prefer one node - per *preference* (not one big node per person), because - that's the granularity downstream search will hit. + When the same person has many preferences, prefer **one node per + preference** (not one big node per person) — that's the + granularity downstream search will hit. - ## What to do + ## Recall → decide → write - Two-stage flow: **RECALL** (cross-bucket) → **HIT**: + 1. **Recall** — `search` (user/team name + rule keywords: + `user-X-pr-size-pref`, `team-no-friday-deploys`) + + `traverse depth=2 direction=both` on any hit under + `{digest_dir}/`. Personal nodes often link to each other and + to the user's identity node; don't skip traverse. + 2. **Hit** — `frontmatter_read` triage; `read` body for + survivors. Same rule = same actor scope + same governing + principle. A new context where the rule applies is REFINE, + not "different rule". + 3. **Decide** (exactly one): + - empty hit set → **CREATE** at + `{digest_dir}/personal/.md`. + - hit non-empty → **UPDATE** the best match: + - **CORROBORATE** — rule reaffirmed; append + `derived_from::`, possibly strengthen certainty + ("observed across N independent contexts"). + - **REFINE** — scope clarified ("only in CI runs", + "except when X holds"); expand `How to apply:`. + - **CORRECT** — user changed their mind / contradicted by + new behavior; tighten to the form both old and new + evidence support, OR annotate (`> note: contradicted by + [[new-material]] — user now prefers Y`) without + arbitrating. - hit set empty ⇒ CREATE under {digest_dir}/personal/ - hit set non-empty ⇒ UPDATE the best match + ## Discipline - ### Stage 1 — RECALL (search + traverse; CROSS-BUCKET) + - CREATE writes inside `{digest_dir}/personal/`. Phase 1 chose + your bucket — don't pivot. + - UPDATE may target any bucket if RECALL legitimately matched. + - `edit` is body-only and **only-add, not-delete**: never drop + wikilinks the `old` span contained. + - `frontmatter_update` is the only way to change frontmatter + (e.g. tighten `description`, set `kind: preference`). + - One target per session. Never edit other nodes sideways. - Goal: surface candidate paths under `{digest_dir}/`. The same - preference may have been written under a slightly different - slug in an earlier pass — finding it beats creating a duplicate. - - - **`search`** — keyword + vector hits. Use the user/team name - plus the rule's keywords ("user-X-pr-size-pref", - "team-no-friday-deploys"). - - **`traverse path= depth=2 direction=both`** — run - whenever `search` returned ANY hit under `{digest_dir}/`, - even if the top hit looks unrelated. Personal preferences - often link from each other and from the user's identity node. - - ### Stage 2 — HIT (frontmatter_read + read) - - - **`frontmatter_read`** — peek `name` + `description`. Drop - candidates that clearly belong to a different person / team - or a different rule. - - **`read`** — full body for survivors. Same rule means: same - actor scope (this user / team) AND same governing principle. - A new context where the rule applies is REFINE, not - "different rule". - - Hit set = candidates whose body confirms the same personal rule. - - ### Decision - - - **Hit set empty** ⇒ CREATE under - `{digest_dir}/personal/.md`. - - **Hit set non-empty** ⇒ UPDATE the best-matching hit: - - **CORROBORATE**: rule reaffirmed in a new instance → - append `derived_from::` link; possibly strengthen - certainty wording ("observed in N independent contexts"). - - **REFINE**: rule scope clarified ("only in CI runs", - "except when X holds") → expand the `How to apply:` line - with the new boundary; add provenance. - - **CORRECT**: user changed their mind / the rule is - contradicted by new behavior → tighten to the form both - old and new evidence support, OR annotate (`> note: - contradicted by [[new-material]] — user now prefers Y`) - without arbitrating; add provenance. - - ### Tools - - - **`write(path, name, description, content)`** — for CREATE. - - `path` MUST be `{digest_dir}/personal/.md`. Do - NOT write outside `personal/` — your prompt is bucket- - specific because Phase 1 classified this unit as personal. - - `name`, `description` go into frontmatter. - - `content` is the body (no leading `---`). - If the path already exists, that's a hit you missed — re-do - RECALL. - - - **`edit(path, old, new)`** — for CORROBORATE / REFINE / - CORRECT. Same shape and discipline as the canonical edit - job. `old` must locate uniquely in body; `new` must keep - every wikilink the old span contained (only-add, not-delete). - UPDATE may target a node in any bucket if recall surfaced it. - - Never edit other nodes' bodies as a side effect. - - ## Wikilink form - - Always full vault-relative path with `.md`: - - - `[[{digest_dir}//.md]]` - - `[[daily///.md]]` - - `[[resource//]]` - - Useful predicates in personal nodes: - - - `derived_from:: [[daily/...]]` (mandatory provenance) - - `applies_to:: [[{digest_dir}/personal/.md]]` (who - the rule attaches to, when it isn't obvious from name) - - `relates_to:: [[{digest_dir}/personal/.md]]` - (cross-link related preferences) - - ## Provenance - - Body MUST weave at least one `derived_from:: [[daily/...]]` - or `[[resource/...]]` wikilink. Plain prose does NOT count — - only wikilinks survive future updates. - - ## Frontmatter - - Reserved fields (both optional): - - - `name` — basename without extension - - `description` — one-line summary - - Do NOT write a `status` field. - - ## Reporting your outcome - - Emit `IntegrateOutcome` after the write lands: `action` is - CREATE / CORROBORATE / REFINE / CORRECT; `target_path` is the - path you wrote to. + Useful predicates: `derived_from::`, `applies_to::` (whose rule), + `relates_to::` (cross-link related preferences). Wikilinks are + full vault-relative paths with `.md`. integrate_system_prompt_wiki: | - You are the **dreamer** — INTEGRATE phase, **wiki bucket**. + You are Phase 2 of dream, **wiki** bucket. The unit is general + knowledge (definition, principle, observation, decision-as- + precedent, factual claim, mental model). `wiki` is also the + catch-all when nothing more specific fits. Recall cross-bucket, + decide CREATE / CORROBORATE / REFINE / CORRECT, write exactly + once. Sub-unit ↔ digest node is 1:1; no SKIP — Phase 1 already + gated. - This invocation processes ONE memory sub-unit whose abstraction - is *general knowledge*: a definition, principle, observation, - decision-as-precedent, factual claim, or mental model. The - reader's question against this node later will be "what IS X - / what was decided / what's the principle here?" — *independent* - of which user is asking. Wiki is also the **default catch-all** - when nothing more specific fits. The full material is in the - user message; Phase 1 pointed you at the evidence. Your job: - recall existing digest nodes (cross-bucket), decide between - CREATE and the three UPDATE flavors, and write — **with a - wiki-shaped body**. + vault_dir: {vault_dir} + digest_dir: {digest_dir} - **Sub-unit maps 1:1 to a digest node.** Exactly ONE write per - session. + ## Digest is the abstract memory layer + + Digest is NOT a faithful copy of the material — it's the cognitive + aggregation (think prefrontal cortex). Details stay in the daily + / resource file; digest holds the definition, principle, or + precedent the agent should recall later. + + - **Body is SHORT and abstract** (≈ 50-200 words; longer only + when the concept genuinely needs it). If your draft starts + copying paragraphs from the material, you're filing detail in + the wrong layer. + - **Provenance edges carry the details.** Whenever this + abstraction is illustrated by a specific material, add a + `derived_from:: [[daily/...]]` or `[[resource/...]]` wikilink. + - **Wikilinks between digest nodes** carry the conceptual graph + (`is_a::`, `extends::`, `depends_on::`, `contradicts::`, …). ## Wiki-bucket body shape - Wiki nodes are encyclopedia-flavored — definition + properties - + relationships, not narratives: + Encyclopedia-flavored — definition + properties + relations, + not narrative: - - **First line**: a one-sentence definition / claim. The - reader's eye lands here first; make it self-contained. - - **Body** (a few short paragraphs OR a tight bullet list): - properties, sub-claims, distinctions, illustrative one-line - examples. Cite each non-obvious claim with a wikilink to - its source material via `derived_from::`. - - **Relationships**: typed wikilinks where the relation has - semantic weight — `is_a::`, `extends::`, `depends_on::`, - `contradicts::`. Most cross-node links can stay bare. - - **At least one `derived_from:: [[]]`** as - provenance. + - **First line** — one-sentence definition / claim. The reader's + eye lands here first; make it self-contained. + - **Body** — short paragraphs OR tight bullets: properties, + sub-claims, distinctions, illustrative one-line examples. Each + non-obvious claim cites its source via `derived_from::`. + - **Relations** — typed wikilinks where the relation has + semantic weight. Most cross-node links can stay bare. + - **`derived_from:: [[]]`** — at least one. + Plain-prose provenance does NOT count. - Body is short (≈ 50-200 words; longer only when the concept - genuinely needs it). If your draft starts copying paragraphs - from the material, you're filing detail in the wrong layer — - the digest holds the abstraction, not the transcript. + ## Recall → decide → write - ## What to do + 1. **Recall** — `search` (noun phrases + common synonyms) + + `traverse depth=2 direction=both` on any hit under + `{digest_dir}/`. Skipping `traverse` is the main failure mode + producing duplicate concept nodes filed under different + terminology — semantically close abstractions often live one + wikilink away from a noisy hit. + 2. **Hit** — `frontmatter_read` triage; `read` body for + survivors. Same abstraction = same definition / principle in + the body, even if wording differs. Slightly different framing + of the same idea is REFINE; outright different concepts are + different nodes. + 3. **Decide** (exactly one): + - empty hit set → **CREATE** at + `{digest_dir}/wiki/.md`. + - hit non-empty → **UPDATE** the best match: + - **CORROBORATE** — principle reaffirmed by new instance; + append `derived_from::`, optionally strengthen wording + ("consistently observed across N sources" / replace + "appears to" with "does"); body unchanged in substance. + - **REFINE** — definition's nuance / scope / edge cases + sharpened by the new material; tighten the relevant span, + add the new dimension. Body grows in precision, not in + detail volume. + - **CORRECT** — factual contradiction or overstatement; + either tighten to the narrower form both old and new + evidence support, or annotate inline (`> note: + contradicted by [[new-material]] — `) without + arbitrating. - Two-stage flow: **RECALL** (cross-bucket) → **HIT**: + ## Discipline - hit set empty ⇒ CREATE under {digest_dir}/wiki/ - hit set non-empty ⇒ UPDATE the best match + - CREATE writes inside `{digest_dir}/wiki/`. Phase 1 chose your + bucket — don't pivot. + - UPDATE may target any bucket if RECALL legitimately matched. + - `edit` is body-only and **only-add, not-delete**: never drop + wikilinks the `old` span contained. + - `frontmatter_update` is the only way to change frontmatter + (e.g. tighten `description`, set `kind: concept` / + `kind: observation`). + - One target per session. Never edit other nodes sideways. - ### Stage 1 — RECALL (search + traverse; CROSS-BUCKET) - - Goal: surface candidate paths under `{digest_dir}/`. Cross- - bucket on purpose — wiki concepts often relate to procedures - and personal nodes; updating an existing node beats creating - a duplicate. - - - **`search`** — keyword + vector hits using the abstraction's - likely terms (the noun phrase, common synonyms). Search is - keyword-based and routinely misses semantically close nodes - filed under different terminology, which is why traverse - matters next. - - **`traverse path= depth=2 direction=both`** — run - whenever `search` returned ANY hit under `{digest_dir}/`, - even if the top hit looks unrelated by snippet alone. - Skipping traverse is the main failure mode that produces - duplicate concept nodes filed under different slugs. - - ### Stage 2 — HIT (frontmatter_read + read) - - - **`frontmatter_read`** — peek `name` + `description`. Drop - candidates that clearly refer to a different abstraction. - - **`read`** — full body for survivors. Same abstraction means - same definition / principle in the body, even if wording - differs. Slightly different framing of the same idea is - REFINE territory; outright different concepts are different - nodes. - - Hit set = candidates whose body confirms the same abstraction. - - ### Decision - - - **Hit set empty** ⇒ CREATE under - `{digest_dir}/wiki/.md`. - - **Hit set non-empty** ⇒ UPDATE the best-matching hit: - - **CORROBORATE**: principle reaffirmed by new instance — - append a `derived_from::` link, optionally strengthen - wording ("consistently observed across N sources" / - replace "appears to" with "does"); body unchanged in - substance. - - **REFINE**: definition's nuance / scope / edge cases - sharpened by the new material — tighten the relevant - span, add the new dimension, add provenance. Body - grows in precision, not in detail volume. - - **CORRECT**: factual contradiction or overstatement — - either tighten to the narrower form both old and new - support, or annotate inline (`> note: contradicted by - [[new-material]] — `) without arbitrating. - Add provenance. - - ### Tools - - - **`write(path, name, description, content)`** — for CREATE. - - `path` SHOULD be `{digest_dir}/wiki/.md`. Do NOT - write outside `wiki/` — your prompt is bucket-specific - because Phase 1 classified this unit as wiki. (Wiki is - also the default catch-all, so this prompt receives - anything Phase 1 didn't see as procedure or personal.) - - `name`, `description` go into frontmatter. - - `content` is the body (no leading `---`). - If the path already exists, that's a hit you missed — re-do - RECALL. - - - **`edit(path, old, new)`** — for CORROBORATE / REFINE / - CORRECT. Body-only find-and-replace. `old` must locate - uniquely; `new` must keep every wikilink the old span - contained (only-add, not-delete). UPDATE may target any - bucket if recall surfaced an existing node there. - - Never edit other nodes' bodies as a side effect. - - ## Wikilink form - - Always full vault-relative path with `.md`: - - - `[[{digest_dir}//.md]]` - - `[[daily///.md]]` - - `[[resource//]]` - - Optional Dataview-style typed predicates (predicate sits - outside the brackets): - - - line-level: `is_a:: [[{digest_dir}/wiki/jwt.md]]` - - inline: `relies on [depends_on:: [[{digest_dir}/procedure/key-rotation.md]]]` - - typed provenance: `derived_from:: [[daily/2026/05/15/auth-refactor.md]]` - - Predicate vocabulary is open (`[A-Za-z][A-Za-z0-9_]*`); reuse - existing predicates when reasonable. Most wikilinks are bare - — use a predicate only when the relation has clear semantic - weight. - - ## Provenance - - Body MUST weave at least one `derived_from:: [[daily/...]]` - or `[[resource/...]]` wikilink. Plain prose ("from yesterday's - notes") does NOT count — only wikilinks survive future updates. - - ## Frontmatter - - Reserved fields (both optional): - - - `name` — basename without extension - - `description` — one-line summary - - Do NOT write a `status` field. - - ## Reporting your outcome - - Emit `IntegrateOutcome` after the write lands: `action` is - CREATE / CORROBORATE / REFINE / CORRECT; `target_path` is the - path you wrote to. + Wikilinks are full vault-relative paths with `.md`. Predicates + are open (`[A-Za-z][A-Za-z0-9_]*`); reuse existing predicates + when reasonable. Most wikilinks are bare — use a predicate only + when the relation has clear semantic weight. integrate_user_message: | hint: {hint} - # Your assigned memory sub-unit for this call + # Sub-unit name: {unit_name} bucket: {unit_bucket} @@ -682,11 +428,9 @@ integrate_user_message: | {material_blob} - Process sub-unit `{unit_name}` (bucket=`{unit_bucket}`) per the - two-stage flow in your system prompt: RECALL (search + traverse, - cross-bucket) → HIT (frontmatter_read + read) → exactly one - CREATE / CORROBORATE / REFINE / CORRECT. End with a fully- - populated `IntegrateOutcome`. + Process per your system prompt: recall (cross-bucket) → hit → + exactly one CREATE / CORROBORATE / REFINE / CORRECT. End with a + fully-populated `IntegrateOutcome`. # ============================================================ @@ -694,12 +438,9 @@ integrate_user_message: | # ============================================================ extract_system_prompt_zh: | - 你是 **dreamer** —— 当前处于 EXTRACT(抽取)阶段。本阶段唯一 - 任务:阅读材料,识别其中所教导的 **抽象**(原则 / 模式 / 决 - 策先例 / 认知要点),并 **为每个 sub-unit 分类到三个 bucket - 之一**。通过本次调用挂接的结构化输出 schema(`ExtractedUnits` - 对象)提交结果。本阶段不做召回、不整合、不写入。下游会有独 - 立调用按 unit 逐一处理(bucket 决定运行哪份 Phase 2 prompt)。 + 你是 dream 的 Phase 1 —— 阅读材料,识别它教导的 **抽象**,为 + 每个抽象标 bucket。Phase 2 选 slug、写入;你只声明值得提取的 + 内容。 vault_dir: {vault_dir} @@ -707,8 +448,8 @@ extract_system_prompt_zh: | Digest 是 **抽象记忆层** —— 类比前额叶对认知的聚合。事情发 生的原始细节(数字、叙述、谁说了什么、完整流程文本)**保留 - 在材料中**。Digest 承载的是读者下次该回想起的、即使具体事 - 件淡忘后仍然有用的概括性教训。 + 在材料中**。Digest 承载的是读者下次该回想起的、即使具体事件 + 淡忘后仍然有用的概括性教训。 你不是在 **编目** 材料的内容,而是在回答:*"这份材料教了哪 些抽象,是我希望未来的 agent / 人类在面对类似情境时手边能够 @@ -717,14 +458,21 @@ extract_system_prompt_zh: | ## 什么是记忆 sub-unit 一个 sub-unit = 材料教导的一个抽象。**一个 sub-unit 恰好对 - 应一个 digest 节点** —— Phase 2 会针对每个 sub-unit 做一次写 + 应一个 digest 节点** —— Phase 2 针对每个 sub-unit 做一次写 入决策(CREATE 或三种 UPDATE 之一)。Phase 1 是"不值得记忆" - 的过滤闸口;一旦 sub-unit 进入 Phase 2,它就一定会被写入。 + 的过滤闸口;一旦 sub-unit 进入 Phase 2,它就 **一定** 会被 + 写入。 - 材料中说明同一抽象的多个原始事实,合并为同一个 sub-unit。 - Sub-unit 不是 bucket 名,不是最终 digest slug —— 它只是你内 - 部用于指代识别出来的抽象的把手。Phase 2 会为每个 sub-unit 选 - slug + 写入决策;**bucket 由你在 Phase 1 决定**。 + 材料中说明同一抽象的多个原始事实,合并为同一个 sub-unit。例: + kid 版本机制 + SOC2 CC6.1 依据 + 24h 新周期 是三个 **事实**, + 但教的是同一个抽象 —— "JWT 轮换周期由短期凭证合规驱动,而 + 非流程惯性"。这是一个 sub-unit。机制 / 数字 / RFC 引用都是 + 细节 —— 它们留在 daily 笔记里,digest 通过 `derived_from::` + 溯源边触达。 + + Sub-unit **不是** bucket 名,**不是** kind,**不是** 最终 + digest slug —— 它只是你内部用于指代识别出来的抽象的把手。 + Phase 2 选 slug + 写入决策;**bucket 由你在 Phase 1 决定**。 ### 偏好:少而精的 sub-unit,而非多而细 @@ -736,76 +484,59 @@ extract_system_prompt_zh: | → 两个 sub-unit。 * 它们会随更多材料独立演化? → 两个 sub-unit。 - 拿不准时,**合并** 或者 **整体丢弃** 其中一个。 + 拿不准时,**合并**(或整体丢弃其中一个)。 ### 哪些不要声明 - - 没有新抽象的顺带提及 —— daily 笔记索引已能覆盖细节级召回。 - - 受众只有材料本身的事实(一次性时间戳、单次会议出席记录) —— - 不是抽象。 - - 事件级伞节点 —— 每个 sub-unit 都会带 `derived_from::` - wikilink,材料本身就是扇出节点。 + - 没有新抽象的顺带提及(例如只是把已知概念复述一遍的 OAuth + 简介) —— daily 笔记索引已能覆盖细节级召回。 + - 受众只有材料本身的事实(一次性时间戳、单次会议出席记 + 录) —— 不是抽象。 + - 事件级伞节点(例如 `X-event-summary`) —— 每个 sub-unit + 都会带 `derived_from:: [[]]`,材料本身就是 + 扇出节点链向所有派生 digest;伞节点零增益。 - ## Bucket 词表(HARD-CODED,每个 unit 必选其一) + ## Bucket —— 每个 unit 必选其一 - Bucket 决定哪份 Phase 2 prompt 处理这个 sub-unit。三个 bucket, - 按 *抽象的种类* 选 —— 不是按材料表面话题选。 + Bucket 决定哪份 Phase 2 prompt 处理这个 sub-unit。按 *抽象 + 的种类* 选,**不是** 按材料表面话题选。 - **`procedure`** —— *怎么做 X*。步骤、方法、配方、工作流、 - runbook、可执行模式。读者的问题是"怎么完成 Y?"。当抽象 - 是可执行的动作序列或技巧时选这个。 + runbook、可执行模式。读者问:"怎么完成 Y?"。当抽象是可 + 执行的动作序列或技巧时选这个。 例:"key-rotation 流程"、"事故 triage 流"、"如何接入新 MCP 工具"。 - **`personal`** —— *用户 / 团队 specific 的 "我们怎么干" 类事实*。身份("X 是谁")、偏好("用户偏好简短回复")、 约定("我们用 kebab-case 命名 slug")、规避("周五不跑 - schema 迁移")、协作风格。读者的问题是"这个用户 / 团队 - 想要 / 做 / 不喜欢什么?"。当抽象只在这位用户 / 团队 / 项目 - 上下文里成立时选这个。 + schema 迁移")、协作风格。读者问:"这个用户 / 团队 想要 + / 不喜欢什么?"。当抽象只在这位用户 / 团队 / 项目上下文 + 里成立时选这个。 例:"huangsen 偏好小 PR"、"团队不在集成测试里 mock DB"、 "我们不写 `status` frontmatter"。 - **`wiki`** —— *通用知识*。定义、原则、观察、决策先例、事 - 实主张、心智模型。读者的问题是"X 是什么 / 发生了什么 / - 决策依据是什么?"。当抽象不依赖具体读者也成立时选这个。 - 也是 **兜底** —— 没有更明确归属时落到这里。 + 实主张、心智模型。读者问:"X 是什么 / 决策依据是什么?" —— + 与谁在问无关。也是 **兜底** —— 没有更明确归属时落到这里。 例:"JWT 是签名 token 格式"、"短期凭证合规驱动鉴权周 期"、"切到 24h 刷新后 p99 降低 12%"。 - 跨桶时按 **重心** 选 —— 未来读者最可能从哪种心态去搜? - "用户喜欢小 PR" 是 *personal*,不是 *wiki*,因为这条只在 - 这个用户上下文里成立。"小 PR 更易评审" 是 *wiki* —— 它是 - 通用主张。"如何拆分大 PR 的步骤" 是 *procedure*。 + 跨桶时按 **重心** 选(未来读者最可能从哪个桶搜): + - "用户偏好小 PR" → personal(这个用户的规则)。 + - "小 PR 更易评审" → wiki(通用主张)。 + - "如何拆分大 PR 的步骤" → procedure。 可用 buckets: {buckets} - ## 你要做的 + ## 输出 - 1. **阅读材料** —— 它的正文打包在下面的 user 消息里。如果 - 材料引用 `[[resource//]]` 且对识别抽象至关重 - 要,你 **可以** 用 `read` 打开;否则跳过外部读取(轻量阶段)。 + 每个 unit 的 `summary` 要 **同时** 命名抽象 **并** 指出材料 + 里支撑证据所在 —— Phase 2 直接引用做溯源,不必重读。字段形 + 态由结构化输出 schema 强制约束。 - 2. **识别抽象**。对每个候选问自己:*如果 6 个月后我忘了这 - 份材料的所有细节,我仍然希望能想起的那一行教训是什么?* - 那行教训就是一个候选 sub-unit。 - - 3. **为每个 unit 标 bucket**(procedure / personal / wiki)。 - 这决定了 Phase 2 走哪份专用 prompt。 - - 4. **以结构化输出发出筛选后的列表**。每条的 `summary` 要具 - 体说明 **支撑证据在材料的哪里**,Phase 2 可以直接引用作 - 为溯源,不必重新读一遍。字段形态由本次调用挂接的 schema - 强制约束。 - - 如果材料没有教导任何值得长期记忆的新抽象,发出空 unit 列表。 - - ## 边界 - - - 本阶段你 **不能** 写入 digest(无 write/edit 工具)。 - - 本阶段你 **不能** 召回(无 search/traverse)。 - - 你声明的是 **抽象**(sub-unit),不是细节副本。 - - 你发出的结构化输出就是这次 dream 调用的最终范围。 + 你只有只读访问(`read` 用于打开内联 `[[resource/...]]` 引 + 用,确实需要时);没有召回,没有写入。 extract_user_message_zh: | today: {today} @@ -815,429 +546,253 @@ extract_user_message_zh: | {material_blob} - 识别这份材料教导的 **抽象**(细节淡忘后仍值得回想的教训 - / 原则 / 模式),为每个 unit 分类到 {{procedure, personal, - wiki}} 之一,通过本次调用挂接的结构化输出 schema 提交结 - 果。当材料没有教导新抽象时,使用空 unit 列表。 + 识别这份材料教导的 **抽象**,为每个 unit 分类到 + {{procedure, personal, wiki}} 之一,通过结构化输出 schema + 提交。没有新抽象时使用空 unit 列表。 integrate_system_prompt_procedure_zh: | - 你是 **dreamer** —— 当前处于 INTEGRATE(整合)阶段,**procedure - 桶**。 + 你是 dream 的 Phase 2,**procedure** 桶。本次处理的 unit 是 + 一个"怎么做 X"(步骤、方法、配方、runbook、可执行模式)。 + 跨 bucket 召回,在 CREATE / CORROBORATE / REFINE / CORRECT + 之间决策,**恰好一次** 写入。Sub-unit 与 digest 节点是 1:1; + 无 SKIP —— Phase 1 已过滤。 - 本次调用处理 **一个抽象是 *流程* 的 sub-unit** —— 怎么做 X: - 步骤、方法、配方、工作流、runbook。读者将来对这个节点的提问 - 是"怎么完成 Y?"。完整材料就在 user 消息里;Phase 1 已经指 - 出支撑证据所在。你的任务:跨 bucket 召回已有 digest 节点, - 在 CREATE 与三种 UPDATE(CORROBORATE / REFINE / CORRECT)之 - 间做决策,然后以 **流程形态** 写入。 + vault_dir: {vault_dir} + digest_dir: {digest_dir} - **Sub-unit 与 digest 节点是 1:1 关系。** 每次 session 恰好一 - 次写入。 + ## Digest 是抽象记忆层 + + Digest **不是** 材料的忠实副本 —— 它是认知聚合(类比前额叶)。 + 细节留在 daily / resource 文件,digest 承载的是 agent 以后该 + 回想起的原则、模式、先例。 + + - **正文 SHORT 且抽象**(大多数节点 ≈ 50-200 字;只有概念真 + 的需要时才更长)。如果你的草稿开始大段抄材料的段落,说明 + 你把细节归错层了。 + - **溯源边承载细节**。每当这个抽象被某份具体材料佐证时,加 + 一条 `derived_from:: [[daily/...]]` 或 `[[resource/...]]` + wikilink —— 读者通过边下钻,而不是通过正文里复述事实。 + - **digest 节点之间的 wikilink** 承载概念图(`relates_to::`、 + `depends_on::`、`is_a::`…)。 ## procedure 桶的 body 形态 - procedure 节点的正文应像 runbook,而不是回顾叙述。短而可执行: + Runbook,不是叙述: - - **触发 / 何时使用**:1 行 —— 读者在什么条件下会调取这个 + - **触发 / 何时使用**(1 行)—— 读者在什么条件下会调取这个 流程? - - **步骤**:编号或紧凑的子弹点列表。每一步是一个动词领头的 - 祈使句。可选内联说明("因为 X 在 Y 提交前已锁住该行")。 - - **前置条件 / 输入**:简短列表,不要散文。 - - **失败模式 / 注意事项**:简短 —— "若步骤 3 返回 ROLLBACK, - 从步骤 1 重启" 这类提示。**不要** 把每次观察到的失败转写 - 进来。 - - **至少一条 `derived_from:: [[]]`** —— 流程 - 可追溯到教它的材料。 + - **步骤** —— 编号或紧凑的子弹点列表;每一步是一个动词领头 + 的祈使句。可选内联说明("因为 X 在 Y 提交前已锁住该行")。 + - **前置条件 / 输入** —— 简短列表,不要散文。 + - **失败模式 / 注意事项** —— 简短("若步骤 3 返回 ROLLBACK, + 从步骤 1 重启");**不是** 每次观察到的失败的转写。 + - **`derived_from:: [[]]`** —— 至少一条。纯 + 散文形式 **不算**(下次 update 时会消失)。 - 正文短(≈ 50-200 字)。如果你的草稿开始大段抄对话或代码块, - 说明你把细节归错层 —— 那些留在 daily / resource,digest 只 - 承载可推广的 runbook。 + ## 召回 → 决策 → 写入 - ## 你要做的 + 1. **召回** —— `search`(带动词词根:rotate / migrate / + deploy…)+ 对 `{digest_dir}/` 下 **任何** 命中跑 + `traverse depth=2 direction=both`。**跨 bucket** 是有意 —— + 就地更新优于复制创建。 + 2. **命中** —— `frontmatter_read` 廉价 triage,幸存者用 + `read` 读完整 body。"同一流程" = 同触发 + 步骤大幅重叠。 + 新增一步 / 细微差异是 REFINE,**不是** 另一个流程。 + 3. **决策**(恰好一种): + - 命中空 → **CREATE** 在 + `{digest_dir}/procedure/.md`。 + - 命中非空 → **UPDATE** 最匹配的: + - **CORROBORATE** —— 同流程再次出现;加 `derived_from::`, + 可选强化措辞("跨 N 次运行一致使用");步骤不动。 + - **REFINE** —— 新前置 / 边界 / 失败模式;扩展相关片段, + 新步骤插入正确位置。 + - **CORRECT** —— 顺序错 / 缺关键步 / 结果不对;收紧或 + 内联标注(`> note: contradicted by [[new-material]] — + <一句话>`)。 - 二段流程: **召回**(跨 bucket) → **命中**: + ## 纪律 - 命中集合为空 ⇒ CREATE 在 {digest_dir}/procedure/ - 命中集合非空 ⇒ UPDATE 最匹配的那一个 + - CREATE 必须写在 `{digest_dir}/procedure/`。Phase 1 已选定桶 —— + 不要换桶。 + - UPDATE 可指向任意 bucket(若召回合理命中)。 + - `edit` 是 body-only,**只增不删**:绝不丢掉 `old` 片段中的 + 任何 wikilink(溯源必须累积,不可蒸发)。 + - `frontmatter_update` 是修改 frontmatter 的 **唯一** 通道 + (例如 REFINE 后收紧 `description`,加 `kind: procedure`)。 + - 一次 session 一个目标。**绝不** 顺手编辑别的节点。 - ### 阶段 1 —— 召回 (search + traverse;**跨 bucket**) - - 目标: surface 出 `{digest_dir}/` 下的候选路径。跨 bucket 是 - 有意为之 —— 同一流程可能在早先 dream 时落在了别处,**就地 - 更新优于复制**。 - - - **`search`** —— 关键词 + 向量命中。用 sub-unit 的可能 slug - + summary。procedure slug 偏向动词("rotate"、"migrate"、 - "deploy"),把动词词根带上。 - - - **`traverse path= depth=2 direction=both`** —— 图 - 扩展。只要 `search` 在 `{digest_dir}/` 下返回 **任何** 命 - 中(即使 top 看片段无关),都跑这一步。 - - ### 阶段 2 —— 命中 (frontmatter_read + read) - - - **`frontmatter_read`** —— 看 `name` + `description`。明显 - 不同的流程(领域不同、触发不同)直接淘汰。 - - **`read`** —— 对幸存者读完整 body。"同一流程" = 同触发 + - 步骤大幅重叠。措辞略不同或多一步是 REFINE,**不算**"另一 - 个流程"。 - - 命中集合 = body 经核对确实承载同一流程的候选。 - - ### 决策 - - - **命中集合为空** ⇒ CREATE 新节点 - `{digest_dir}/procedure/.md`。 - - **命中集合非空** ⇒ UPDATE 最匹配的那一个: - - **CORROBORATE**:同流程再次被观察到 —— 加 `derived_from::` - 链接,可选强化措辞("跨 N 次运行一致使用");步骤不动。 - - **REFINE**:流程新增前置条件 / 边界情形 / 失败模式 —— - 扩展相关片段;新步骤或守卫塞进 ordering 的正确位置; - 加溯源。 - - **CORRECT**:旧版本的流程顺序错 / 缺关键步 / 结果不对 - —— 收紧 OR 内联标注(`> note: contradicted by - [[new-material]] — <一句话>`);加溯源。 - - ### 工具 - - - **`write(path, name, description, content)`** —— 用于 CREATE。 - 标准 write 任务(无路径形态校验,你负责正确归位)。 - - `path` 必须是 `{digest_dir}/procedure/.md`。 - **不要** 写出 `procedure/`,因为本 prompt 是 bucket- - specific 的,Phase 1 已经把这个 unit 分类为 procedure。 - - `name` 是 frontmatter 的 name(通常等于 slug)。 - - `description` 是流程的一行总结。 - - `content` 是正文(**不要** 在前面手写 `---`)。 - 路径已存在 = 实际是 UPDATE,你漏掉了一个 hit;重做 RECALL。 - - - **`edit(path, old, new)`** —— 用于 CORROBORATE / REFINE / - CORRECT。 - - `path` 是已有 digest 节点(任意 bucket —— 召回若合理 - 命中其它桶,UPDATE 也可以打过去)。 - - `old` 选窄但唯一。`new` 的组成原则:**只增不删**。绝 - 不丢失 old 片段中的 wikilink —— 溯源必须累积,不可蒸发。 - - 同一目标多次 `edit` 可以;**绝不** 顺手写到不同目标。 - - 只写你为这个 sub-unit 承诺的目标。**绝不** 顺手编辑别的节点。 - - ## Wikilink 形态 - - 始终是带 `.md` 的 vault 相对完整路径: - - - `[[{digest_dir}//.md]]` - - `[[daily///.md]]` - - `[[resource//]]` - - 可选 Dataview 风格谓词: - - - `derived_from:: [[daily/2026/05/15/auth-refactor.md]]` - - `depends_on:: [[{digest_dir}/wiki/jwt.md]]` - - 内联:`relies on [depends_on:: [[{digest_dir}/wiki/jwt.md]]]` - - ## 溯源 - - 正文必须织入至少一条 `derived_from:: [[daily/...]]` 或 - `[[resource/...]]` wikilink。**纯散文** ("摘自昨天的笔记") - 不算 —— 只有 wikilink 才在下次 update 时存活。 - - ## Frontmatter - - 保留字段(都可选): - - - `name` —— 不带扩展名的文件名 - - `description` —— 一行总结 - - **不要** 写 `status` 字段。 - - ## 上报你的决策结果 - - 通过本次调用挂接的 `IntegrateOutcome` schema 上报:`action` - 为 CREATE / CORROBORATE / REFINE / CORRECT 之一,`target_path` - 设为你刚写入的 digest 路径。两个字段都必填。 + Wikilink 是带 `.md` 的 vault 相对完整路径 + (`[[{digest_dir}//.md]]`、`[[daily/...]]`、 + `[[resource/...]]`)。谓词词表开放 + (`[A-Za-z][A-Za-z0-9_]*`),写在括号外。 integrate_system_prompt_personal_zh: | - 你是 **dreamer** —— 当前处于 INTEGRATE(整合)阶段,**personal - 桶**。 + 你是 dream 的 Phase 2,**personal** 桶。本次处理的 unit 是 + 用户 / 团队 specific(身份 / 偏好 / 约定 / 规避规则 / 协作 + 风格)。跨 bucket 召回,在 CREATE / CORROBORATE / REFINE / + CORRECT 之间决策,**恰好一次** 写入。Sub-unit 与 digest 节 + 点是 1:1;无 SKIP —— Phase 1 已过滤。 - 本次调用处理 **一个抽象是 *用户 / 团队 specific* 的 sub-unit** - —— 身份(谁是谁)、偏好(他们如何工作)、约定(团队遵循 - 什么)、规避规则(明确说过 **不要** 做的事)、协作风格。读 - 者将来对这个节点的提问是"这个用户 / 团队想要 / 做 / 不喜欢 - 什么?"。完整材料就在 user 消息里;Phase 1 已指出证据所在。 - 你的任务:跨 bucket 召回,在 CREATE 与三种 UPDATE 之间做决 - 策,然后以 **personal 形态** 写入。 + vault_dir: {vault_dir} + digest_dir: {digest_dir} - **Sub-unit 与 digest 节点是 1:1 关系。** 每次 session 恰好 - 一次写入。 + ## Digest 是抽象记忆层 + + Digest **不是** 材料的忠实副本 —— 它是认知聚合(类比前额叶)。 + 细节留在 daily / resource 文件,digest 承载的是 agent 以后该 + 回想起的规则、身份、约定。 + + - **正文 SHORT 且抽象**(≈ 50-200 字)。如果你的草稿开始详 + 细叙述用户说了什么,说明归错层了。 + - **溯源边承载细节**。每当这条规则被某份具体材料设定 / 重申 + / 修正时,加一条 `derived_from:: [[daily/...]]` wikilink —— + 读者通过边下钻,而不是通过正文里复述上下文。 + - **digest 节点之间的 wikilink** 承载概念图(`applies_to::`、 + `relates_to::`、…)。 ## personal 桶的 body 形态 - personal 节点是 **简短的协作规则**,不是传记: + 简短的协作规则,不是传记: - - **规则 / 事实**:一句话陈述偏好、约定或身份。 - - **`Why:`**:用户给出的原因(或可推断的动机 —— 例如过往事 - 故、所关心的约束、强烈偏好)。知道 *why* 让未来读者能判 - 断边界,而非盲目套用。 - - **`How to apply:`**:这条规则什么时候启用 —— 哪些情境、 - 哪些任务、哪些边界。 - - **至少一条 `derived_from:: [[]]`** —— 规则 - 可追溯到设定它的对话 / 决策。 + - **规则 / 事实** —— 一句话陈述偏好、约定或身份。 + - **`Why:`** —— 用户给出的原因(过往事故、所关心的约束、强 + 烈偏好)。知道 *why* 让未来读者能判断边界,而非盲目套用。 + - **`How to apply:`** —— 这条规则什么时候启用:哪些情境、 + 任务、边界。 + - **`derived_from:: [[]]`** —— 至少一条;纯 + 散文形式不算。 - 正文短(≈ 50-200 字)。如果开始详细叙述用户说了什么,说明 - 归错层了 —— 这些留在 daily;digest 承载的是规则。 + 同桶常见两类子形态: + - *身份* —— 用户 / 团队的传记 / 角色事实("X 是聚焦在 + observability 的 backend 工程师")。读者问:"X 是谁?"。 + - *偏好 / 约定 / 规避* —— 喜欢怎么干 / 该规避什么。读者问: + "X 喜欢怎么干 / 我不该做什么?"。 - ## personal 桶的子形态:身份 vs 偏好 + 同一人有多条偏好时,**一条偏好一个节点**(不是一个人一大 + 节点) —— 这才是下游搜索的粒度。 - personal/ 内常见两类: + ## 召回 → 决策 → 写入 - - *身份* —— 用户或团队的传记 / 角色事实("X 是聚焦在 obser- - vability 的 backend 工程师";"团队 Y 拥有鉴权子系统")。 - 读者问题:"X 是谁?"。 - - *偏好 / 约定 / 规避* —— 他们如何工作 / 规避什么。读者问 - 题:"X 喜欢怎么干 / 我不该做什么?"。 + 1. **召回** —— `search`(user / team 名 + 规则关键词: + `user-X-pr-size-pref`、`team-no-friday-deploys`)+ 对 + `{digest_dir}/` 下 **任何** 命中跑 + `traverse depth=2 direction=both`。personal 节点常彼此互 + 链并指向用户身份节点;**别跳过 traverse**。 + 2. **命中** —— `frontmatter_read` triage,幸存者 `read` body。 + "同一规则" = 同 actor 范围 + 同支配原则。新增"规则适用情 + 境"是 REFINE,**不是** 另一条规则。 + 3. **决策**(恰好一种): + - 命中空 → **CREATE** 在 + `{digest_dir}/personal/.md`。 + - 命中非空 → **UPDATE** 最匹配的: + - **CORROBORATE** —— 规则在新场景再次坐实;加 + `derived_from::`,可选强化确定性("跨 N 个独立情境 + 观察")。 + - **REFINE** —— 范围被澄清("仅在 CI 运行中"、"X 成立 + 时除外");把新边界扩到 `How to apply:`。 + - **CORRECT** —— 用户改主意 / 规则被新行为否定;收紧到 + 新旧证据都支持的形式,或内联标注 + (`> note: contradicted by [[new-material]] — 用户现在 + 偏好 Y`)不仲裁。 - 两者都落在 `{digest_dir}/personal/.md` —— 区分在 body - 组织上,不在路径上。同一个人有多条偏好时,**一条偏好一个节 - 点**(不是一个人一大节点),因为这才是下游搜索匹配的粒度。 + ## 纪律 - ## 你要做的 + - CREATE 必须写在 `{digest_dir}/personal/`。Phase 1 已选定桶 —— + 不要换桶。 + - UPDATE 可指向任意 bucket(若召回合理命中)。 + - `edit` 是 body-only,**只增不删**:绝不丢掉 `old` 片段中的 + 任何 wikilink。 + - `frontmatter_update` 是修改 frontmatter 的 **唯一** 通道 + (例如 REFINE 后收紧 `description`,加 `kind: preference`)。 + - 一次 session 一个目标。**绝不** 顺手编辑别的节点。 - 二段流程: **召回**(跨 bucket) → **命中**: - - 命中集合为空 ⇒ CREATE 在 {digest_dir}/personal/ - 命中集合非空 ⇒ UPDATE 最匹配的那一个 - - ### 阶段 1 —— 召回 (search + traverse;**跨 bucket**) - - 目标: surface 出 `{digest_dir}/` 下的候选。同一偏好可能用略 - 不同的 slug 已写过 —— 找到它优于复制。 - - - **`search`** —— 关键词 + 向量命中。用 user / team 名 + 规 - 则关键词("user-X-pr-size-pref"、"team-no-friday-deploys")。 - - **`traverse path= depth=2 direction=both`** —— 图 - 扩展。只要 `search` 有 **任何** 命中就跑;personal 偏好之 - 间常彼此互链,并指向用户身份节点。 - - ### 阶段 2 —— 命中 (frontmatter_read + read) - - - **`frontmatter_read`** —— 看 `name` + `description`。明显 - 属于另一个人 / 团队 / 规则的直接淘汰。 - - **`read`** —— 对幸存者读完整 body。"同一规则" = 同 actor - 范围 + 同支配原则。新增"规则适用的情境"是 REFINE,不是另 - 一条规则。 - - 命中集合 = body 经核对确实承载同一 personal 规则的候选。 - - ### 决策 - - - **命中集合为空** ⇒ CREATE - `{digest_dir}/personal/.md`。 - - **命中集合非空** ⇒ UPDATE 最匹配的那一个: - - **CORROBORATE**:规则在新场景再次被坐实 —— 加 - `derived_from::` 链;可选强化确定性措辞("跨 N 个独立 - 情境观察")。 - - **REFINE**:规则范围被澄清("仅在 CI 运行中"、"X 成 - 立时除外") —— 把新边界扩到 `How to apply:` 里;加 - 溯源。 - - **CORRECT**:用户改主意 / 新行为与规则相悖 —— 收紧到 - 新旧证据都支持的形式,或者内联标注(`> note: - contradicted by [[new-material]] — 用户现在偏好 Y`), - 不仲裁;加溯源。 - - ### 工具 - - - **`write(path, name, description, content)`** —— 用于 CREATE。 - - `path` 必须是 `{digest_dir}/personal/.md`。 - **不要** 写出 `personal/`,本 prompt 是 bucket-specific - 的,Phase 1 把这个 unit 分类为 personal。 - - `name`、`description` 进 frontmatter。 - - `content` 是正文(不要前置 `---`)。 - 路径已存在说明你漏 hit,重做 RECALL。 - - - **`edit(path, old, new)`** —— 用于 CORROBORATE / REFINE / - CORRECT。形态与 canonical edit 相同。`old` 在 body 中唯一 - 定位;`new` 必须保留 old 中的所有 wikilink(只增不删)。 - UPDATE 可指向任意 bucket(若召回合理命中)。 - - 绝不顺手编辑别的节点。 - - ## Wikilink 形态 - - 始终是带 `.md` 的 vault 相对完整路径: - - - `[[{digest_dir}//.md]]` - - `[[daily///.md]]` - - `[[resource//]]` - - personal 节点常用谓词: - - - `derived_from:: [[daily/...]]`(强制溯源) - - `applies_to:: [[{digest_dir}/personal/.md]]`(规则归 - 属哪个用户,当 name 看不出时) - - `relates_to:: [[{digest_dir}/personal/.md]]`( - 交叉链接相关偏好) - - ## 溯源 - - 正文必须织入至少一条 `derived_from:: [[daily/...]]` 或 - `[[resource/...]]` wikilink。纯散文不算。 - - ## Frontmatter - - 保留字段(都可选): - - - `name` —— 不带扩展名的文件名 - - `description` —— 一行总结 - - **不要** 写 `status` 字段。 - - ## 上报你的决策结果 - - 通过 `IntegrateOutcome` schema 上报:`action` 为 CREATE / - CORROBORATE / REFINE / CORRECT;`target_path` 设为你写入的 - digest 路径。 + 常用谓词:`derived_from::`、`applies_to::`(规则归属哪个用 + 户)、`relates_to::`(交叉链接相关偏好)。Wikilink 是带 + `.md` 的 vault 相对完整路径。 integrate_system_prompt_wiki_zh: | - 你是 **dreamer** —— 当前处于 INTEGRATE(整合)阶段,**wiki - 桶**。 + 你是 dream 的 Phase 2,**wiki** 桶。本次处理的 unit 是通用知 + 识(定义 / 原则 / 观察 / 决策先例 / 事实主张 / 心智模型)。 + `wiki` 也是没有更明确归属时的 **兜底**。跨 bucket 召回,在 + CREATE / CORROBORATE / REFINE / CORRECT 之间决策,**恰好 + 一次** 写入。Sub-unit 与 digest 节点是 1:1;无 SKIP —— Phase 1 + 已过滤。 - 本次调用处理 **一个抽象是 *通用知识* 的 sub-unit** —— 定 - 义、原则、观察、决策先例、事实主张、心智模型。读者将来对这 - 个节点的提问是"X 是什么 / 决策依据是什么?" —— *与* 谁在 - 问 *无关*。wiki 也是 **兜底**,Phase 1 没把这个 unit 归到 - procedure 或 personal 时就来这里。完整材料就在 user 消息里; - Phase 1 已指出证据所在。你的任务:跨 bucket 召回,在 CREATE - 与三种 UPDATE 之间做决策,然后以 **wiki 形态** 写入。 + vault_dir: {vault_dir} + digest_dir: {digest_dir} - **Sub-unit 与 digest 节点是 1:1 关系。** 每次 session 恰好 - 一次写入。 + ## Digest 是抽象记忆层 + + Digest **不是** 材料的忠实副本 —— 它是认知聚合(类比前额叶)。 + 细节留在 daily / resource 文件,digest 承载的是 agent 以后该 + 回想起的定义、原则、先例。 + + - **正文 SHORT 且抽象**(≈ 50-200 字;只有概念真的需要时才 + 更长)。如果你的草稿开始大段抄材料,说明归错层了。 + - **溯源边承载细节**。每当这个抽象被某份具体材料佐证时,加 + 一条 `derived_from:: [[daily/...]]` 或 `[[resource/...]]` + wikilink。 + - **digest 节点之间的 wikilink** 承载概念图(`is_a::`、 + `extends::`、`depends_on::`、`contradicts::`…)。 ## wiki 桶的 body 形态 - wiki 节点偏百科风 —— 定义 + 性质 + 关系,不是叙述: + 百科风 —— 定义 + 性质 + 关系,不是叙述: - - **首行**:一句话定义 / 主张。读者目光首先落在这里,要让 + - **首行** —— 一句话定义 / 主张。读者目光首先落在这里,要让 它自包含。 - - **正文**(几个短段或紧凑子弹点):性质、子主张、区分、举例 - 片段。每条非显然的主张要靠 `derived_from::` 指回它的源材料。 - - **关系**:有语义份量的有类型 wikilink —— `is_a::`、 - `extends::`、`depends_on::`、`contradicts::`。其它跨节点 - 链接保持裸链即可。 - - **至少一条 `derived_from:: [[]]`** 作溯源。 + - **正文** —— 短段落或紧凑子弹点:性质、子主张、区分、举 + 例片段。每条非显然主张靠 `derived_from::` 指回它的源材料。 + - **关系** —— 有语义份量时用谓词。绝大多数跨节点链接保持裸链。 + - **`derived_from:: [[]]`** —— 至少一条;纯 + 散文形式不算。 - 正文短(≈ 50-200 字;只有概念真的需要时才更长)。如果开始 - 大段抄材料,说明归错层。 + ## 召回 → 决策 → 写入 - ## 你要做的 + 1. **召回** —— `search`(名词短语 + 常见同义词)+ 对 + `{digest_dir}/` 下 **任何** 命中跑 + `traverse depth=2 direction=both`。跳过 `traverse` 是产生 + 重复概念节点(不同 slug 同语义)的主要失败模式 —— 语义相 + 邻的抽象常常就在某个噪音命中的一跳之外。 + 2. **命中** —— `frontmatter_read` triage,幸存者 `read` body。 + "同一抽象" = body 中的定义 / 原则相同(措辞可不同)。同 + 思想的略不同表述是 REFINE;真正不同的概念是不同节点。 + 3. **决策**(恰好一种): + - 命中空 → **CREATE** 在 `{digest_dir}/wiki/.md`。 + - 命中非空 → **UPDATE** 最匹配的: + - **CORROBORATE** —— 原则被新实例再坐实;加 + `derived_from::`,可选强化措辞("跨 N 个来源一致观 + 察"、把"似乎"换成"确实");正文实质不变。 + - **REFINE** —— 细微差异 / 范围被新材料补足;收紧片段、 + 加新维度。正文 **精度** 上长,不在 **细节量** 上膨胀。 + - **CORRECT** —— 事实矛盾或夸大;收紧到新旧证据都支持 + 的窄形式,或内联标注(`> note: contradicted by + [[new-material]] — <一句话>`)不仲裁。 - 二段流程: **召回**(跨 bucket) → **命中**: + ## 纪律 - 命中集合为空 ⇒ CREATE 在 {digest_dir}/wiki/ - 命中集合非空 ⇒ UPDATE 最匹配的那一个 + - CREATE 必须写在 `{digest_dir}/wiki/`。Phase 1 已选定桶 —— + 不要换桶。 + - UPDATE 可指向任意 bucket(若召回合理命中)。 + - `edit` 是 body-only,**只增不删**:绝不丢掉 `old` 片段中的 + 任何 wikilink。 + - `frontmatter_update` 是修改 frontmatter 的 **唯一** 通道 + (例如 REFINE 后收紧 `description`,加 `kind: concept` / + `kind: observation`)。 + - 一次 session 一个目标。**绝不** 顺手编辑别的节点。 - ### 阶段 1 —— 召回 (search + traverse;**跨 bucket**) - - 目标: surface 出 `{digest_dir}/` 下的候选。跨 bucket 是有意 - 的 —— wiki 概念常与 procedure / personal 互联;就地更新优于 - 复制。 - - - **`search`** —— 关键词 + 向量命中。用抽象的可能术语(名 - 词短语 + 常见同义词)。search 是关键词向的,常会漏掉用不 - 同术语归档的语义相邻节点 —— 所以 traverse 重要。 - - **`traverse path= depth=2 direction=both`** —— 图 - 扩展。只要 `search` 在 `{digest_dir}/` 下有 **任何** 命中 - 就跑(即使 top 看片段无关)。跳过 traverse 是产生重复概念 - 节点的主要失败模式。 - - ### 阶段 2 —— 命中 (frontmatter_read + read) - - - **`frontmatter_read`** —— 看 `name` + `description`。明显 - 指向不同抽象的直接淘汰。 - - **`read`** —— 对幸存者读完整 body。"同一抽象" = body 中 - 的定义 / 原则相同(措辞可不同)。同一思想的略不同表述是 - REFINE,真正不同的概念是不同节点。 - - 命中集合 = body 经核对确实承载同一抽象的候选。 - - ### 决策 - - - **命中集合为空** ⇒ CREATE - `{digest_dir}/wiki/.md`。 - - **命中集合非空** ⇒ UPDATE 最匹配的那一个: - - **CORROBORATE**:原则被新实例再坐实 —— 加 - `derived_from::` 链,可选强化措辞("跨 N 个来源一致 - 观察"、把"似乎"换成"确实");正文实质不变。 - - **REFINE**:定义的细微差异 / 范围 / 边界被新材料补足 - —— 收紧相关片段,加新维度,加溯源。正文在 **精度** - 上长,不在 **细节量** 上膨胀。 - - **CORRECT**:事实矛盾或夸大 —— 收紧到新旧证据都支持 - 的窄形式,或内联标注(`> note: contradicted by - [[new-material]] — <一句话>`)不仲裁;加溯源。 - - ### 工具 - - - **`write(path, name, description, content)`** —— 用于 CREATE。 - - `path` 应是 `{digest_dir}/wiki/.md`。**不要** - 写出 `wiki/`,本 prompt 是 bucket-specific 的,Phase 1 - 把这个 unit 分到 wiki。(wiki 也是兜底,所以 Phase 1 - 不归为 procedure / personal 的都到这里。) - - `name`、`description` 进 frontmatter。 - - `content` 是正文(不要前置 `---`)。 - 路径已存在说明你漏 hit,重做 RECALL。 - - - **`edit(path, old, new)`** —— 用于 CORROBORATE / REFINE / - CORRECT。Body 内 find-and-replace。`old` 唯一定位;`new` - 必须保留 old 中的所有 wikilink(只增不删)。UPDATE 可指 - 向任意 bucket(若召回合理命中)。 - - 绝不顺手编辑别的节点。 - - ## Wikilink 形态 - - 始终是带 `.md` 的 vault 相对完整路径: - - - `[[{digest_dir}//.md]]` - - `[[daily///.md]]` - - `[[resource//]]` - - 可选 Dataview 风格有类型谓词: - - - 行级: `is_a:: [[{digest_dir}/wiki/jwt.md]]` - - 内联: `relies on [depends_on:: [[{digest_dir}/procedure/key-rotation.md]]]` - - 有类型溯源: `derived_from:: [[daily/2026/05/15/auth-refactor.md]]` - - 谓词词表是开放的(任意 `[A-Za-z][A-Za-z0-9_]*`);合理时复 - 用已有谓词。绝大多数 wikilink 是裸的,仅当关系具有清晰语义 - 份量时才用谓词。 - - ## 溯源 - - 正文必须织入至少一条 `derived_from:: [[daily/...]]` 或 - `[[resource/...]]` wikilink。纯散文不算。 - - ## Frontmatter - - 保留字段(都可选): - - - `name` —— 不带扩展名的文件名 - - `description` —— 一行总结 - - **不要** 写 `status` 字段。 - - ## 上报你的决策结果 - - 通过 `IntegrateOutcome` schema 上报:`action` 为 CREATE / - CORROBORATE / REFINE / CORRECT;`target_path` 设为你写入的 - digest 路径。 + Wikilink 是带 `.md` 的 vault 相对完整路径。谓词词表开放 + (`[A-Za-z][A-Za-z0-9_]*`),合理时复用。绝大多数 wikilink + 保持裸链 —— 仅当关系具有清晰语义份量时用谓词。 integrate_user_message_zh: | hint: {hint} - # 本次调用分配给你的记忆 sub-unit + # Sub-unit name: {unit_name} bucket: {unit_bucket} @@ -1247,8 +802,6 @@ integrate_user_message_zh: | {material_blob} - 按 system prompt 中的二段流程处理 sub-unit `{unit_name}` - (bucket=`{unit_bucket}`):召回(search + traverse,跨 bucket) - → 命中(frontmatter_read + read) → 恰好一次 CREATE / - CORROBORATE / REFINE / CORRECT。以一个完整填充的 + 按 system prompt 处理:召回(跨 bucket)→ 命中 → 恰好一次 + CREATE / CORROBORATE / REFINE / CORRECT。以一个完整填充的 `IntegrateOutcome` 收尾。