From 7d0bec60be8cfeacf0535490f0d66329a7f2800f Mon Sep 17 00:00:00 2001 From: Sen Huang <48879559+ployts@users.noreply.github.com> Date: Fri, 22 May 2026 18:10:29 +0800 Subject: [PATCH] feat: rename working_dir to vault_dir and update documentation (#254) * feat: rename working_dir to vault_dir and update documentation - Rename working_dir to vault_dir across the application - Update documentation to reflect vault_dir instead of working_dir - Change FileFrontMatter title field to name field - Update .gitignore to include vault directory - Modify file path descriptions to reference vault instead of working_dir - Update related configuration and property names accordingly * refactor(steps): rename working_path to vault_path in CRUD operations - Rename parameter from `working_path` to `vault_path` in `resolve_path` function - Update all usages in append, edit, read, and write steps to use `self.vault_path` - Update documentation comments to reflect the new parameter name - Update docstring in read.py to mention `vault_dir` instead of `vault` test(chunked_file_parser): update frontmatter field from title to name - Change frontmatter field from `title` to `name` in test cases - Update comment in background steps test to reference `vault_path` instead of `working_path` * refactor(schema): remove unused ResourceEntry import * feat(file_graph): add link scope filtering to get_inlinks/get_outlinks * feat(file-store): add scope parameter to link methods --- .gitignore | 1 + docs4/reme4_index.md | 6 +- docs4/reme4_report.md | 29 +- docs4/reme_design.md | 18 +- example.env | 16 +- reme4/application.py | 11 +- reme4/components/base_component.py | 16 +- .../embedding/base_embedding_model.py | 2 +- .../components/file_graph/base_file_graph.py | 33 +- .../components/file_graph/local_file_graph.py | 37 +- .../components/file_graph/neo4j_file_graph.py | 71 +- reme4/components/file_graph/nx_file_graph.py | 32 +- .../file_parser/linked_file_parser.py | 124 +- .../components/file_store/base_file_store.py | 20 +- .../components/file_store/local_file_store.py | 17 +- .../keyword_index/base_keyword_index.py | 2 +- reme4/enumeration/__init__.py | 2 + reme4/enumeration/link_scope_enum.py | 17 + reme4/schema/application_config.py | 6 +- reme4/schema/file_chunk.py | 2 +- reme4/schema/file_front_matter.py | 2 +- reme4/schema/file_node.py | 2 +- reme4/schema/request.py | 2 +- reme4/steps/background/index_changes.py | 2 +- reme4/steps/background/update_store.py | 6 +- reme4/steps/background/watch_changes.py | 4 +- reme4/steps/base_step.py | 6 +- reme4/steps/crud/_file_io.py | 8 +- reme4/steps/crud/append.py | 2 +- reme4/steps/crud/edit.py | 2 +- reme4/steps/crud/read.py | 4 +- reme4/steps/crud/write.py | 2 +- tests4/unittest/test_background_steps.py | 2 +- tests4/unittest/test_chunked_file_parser.py | 4 +- tests4/unittest/test_crud_md_steps.py | 1071 ----------------- tests4/unittest/test_file_graph.py | 85 +- tests4/unittest/test_linked_file_parser.py | 178 +-- tests4/unittest/test_neo4j_file_graph.py | 6 +- 38 files changed, 404 insertions(+), 1446 deletions(-) create mode 100644 reme4/enumeration/link_scope_enum.py delete mode 100644 tests4/unittest/test_crud_md_steps.py diff --git a/.gitignore b/.gitignore index 1e52cf66..04a536d5 100644 --- a/.gitignore +++ b/.gitignore @@ -46,3 +46,4 @@ meta_memory/* *.db memories/* .reme/* +vault \ No newline at end of file diff --git a/docs4/reme4_index.md b/docs4/reme4_index.md index 18ed6f36..00bf6c1e 100644 --- a/docs4/reme4_index.md +++ b/docs4/reme4_index.md @@ -29,12 +29,12 @@ | 类 | 文件 | 说明 | | --- | --- | --- | -| `ApplicationConfig` / `ComponentConfig` / `JobConfig` | `application_config.py` | 顶层配置模型;包含 service/jobs/components/working_dir 等字段。 | +| `ApplicationConfig` / `ComponentConfig` / `JobConfig` | `application_config.py` | 顶层配置模型;包含 service/jobs/components/vault_dir 等字段。 | | `EmbNode` | `emb_node.py` | 文本+embedding 节点基类,`np.ndarray` 序列化为列表存储。 | | `FileChunk` | `file_chunk.py` | 文件切片(继承 `EmbNode`):`path/start_line/end_line/scores`,含 `set_hash_id()`。 | | `FileNode` | `file_node.py` | 文件级图节点:`path/st_mtime/links/chunk_ids/front_matter`。 | | `FileLink` | `file_link.py` | 文件间 wikilink 边:`source_path → target_path`,可选 `target_anchor` / `predicate`。 | -| `FileFrontMatter` | `file_front_matter.py` | YAML 头:`title/description/tags`,`extra="allow"` 保留未知键。 | +| `FileFrontMatter` | `file_front_matter.py` | YAML 头:`name/description`,`extra="allow"` 保留未知键。 | | `Request` / `Response` | `request.py` / `response.py` | 服务端请求/响应封装。 | | `StreamChunk` | `stream_chunk.py` | 流式分块:`chunk_type/chunk/done/metadata`。 | @@ -43,7 +43,7 @@ > 所有组件继承 `BaseComponent`(`reme4/components/base_component.py`),提供: > - `start/close/restart` 生命周期; > - `bind(name, base_cls, default_factory, optional)` 声明依赖(启动时通过拓扑排序解析); -> - `working_path` / `working_metadata_path` 工作目录; +> - `vault_path` / `working_metadata_path` 工作目录; > - `dump/load` 持久化钩子。 > > 组件通过 `ComponentRegistry`(`R`,`reme4/components/component_registry.py`)按 `(ComponentEnum, name)` 注册和查找;上下文容器为 `ApplicationContext`(`application_context.py`),运行期上下文为 `RuntimeContext`(`runtime_context.py`)。 diff --git a/docs4/reme4_report.md b/docs4/reme4_report.md index b9983961..1c80e97a 100644 --- a/docs4/reme4_report.md +++ b/docs4/reme4_report.md @@ -307,7 +307,7 @@ ReMe 同时跑三种检索通路,结果通过 RRF(Reciprocal Rank Fusion) `reme4/steps/common/search.py`)是**分跳**的,且每一跳的"信息密度"刻意不同: - **第一跳:直接命中的切片**——返回 chunk 全文 + 章节骨架。 -- **第二跳:1-hop 邻居**——只返回邻居的 path + meta(title/tags)+ 边的语义(predicate/anchor),**不展开正文**。 +- **第二跳:1-hop 邻居**——只返回邻居的 path + meta(name/description)+ 边的语义(predicate/anchor),**不展开正文**。 - **第 N 跳:Agent 主动追问**——基于二跳的"目录",挑出真正相关的邻居,再发起新一次 search 拿正文。 **一个具体例子:分析师查询"钴的下游应用"** @@ -321,20 +321,20 @@ ReMe 同时跑三种检索通路,结果通过 RRF(Reciprocal Rank Fusion) 钴是锂电正极材料的关键原料,主要用于动力电池、消费电子和储能…… → outlinks (3): - → digest/knowledge/financial/矿产/刚果(金).md title="刚果(金) - 钴矿主产区" tags=['矿产', '非洲'] + → digest/knowledge/financial/矿产/刚果(金).md name="刚果(金) - 钴矿主产区" via predicate=producer, anchor=#钴矿带 - → digest/knowledge/financial/公司/嘉能可.md title="嘉能可 Glencore" tags=['矿企', '海外'] + → digest/knowledge/financial/公司/嘉能可.md name="嘉能可 Glencore" via plain - → digest/knowledge/financial/产品/三元正极.md title="三元正极材料" tags=['锂电', '正极'] + → digest/knowledge/financial/产品/三元正极.md name="三元正极材料" via predicate=downstream ← inlinks (2): - ← digest/knowledge/financial/产业链/锂电产业链.md title="锂电产业链总览" tags=['新能源', '锂电'] + ← digest/knowledge/financial/产业链/锂电产业链.md name="锂电产业链总览" via predicate=upstream - ← daily/20260318/宁德调研纪要.md title="宁德时代调研纪要" tags=['公司调研'] + ← daily/20260318/宁德调研纪要.md name="宁德时代调研纪要" via plain ``` -注意第二跳的信息只有"路径 + 标题 + 标签 + 边的 predicate/anchor",**没有邻居正文**。这是关键设计: +注意第二跳的信息只有"路径 + 名称 + 边的 predicate/anchor",**没有邻居正文**。这是关键设计: - 一次检索就让 Agent 看到"这个主题周围长什么样"——上游是刚果(金)、嘉能可,下游是三元正极,被锂电产业链当作 upstream 引用,最近还在 3 月 18 日的宁德调研里被提到。 @@ -555,7 +555,7 @@ digest/knowledge/financial/公司/洛阳钼业.md:1-22 [score=0.0152] ... ``` -Agent 不需要把这些邻居正文都拉回来——光看"路径 + title + predicate"就足够拼出上下游骨架。 +Agent 不需要把这些邻居正文都拉回来——光看"路径 + name + predicate"就足够拼出上下游骨架。 ###### Step 2:Agent 基于检索结果合成总览,写回知识库 @@ -570,8 +570,7 @@ digest/knowledge/financial/产业链/ ```markdown --- -title: 锂电产业链总览 -tags: [产业链, 新能源, 锂电] +name: 锂电产业链总览 source: query-synthesized ← 标记来源是 query 合成 trigger_query: "分析锂电相关上下游" generated_at: 2026-05-22 @@ -641,8 +640,7 @@ Agent 拿这份合成结果,给王分析师的回复直接带**上下游骨架 ```markdown --- -title: 早间洞察 · 2026-05-19 -tags: [proactive, 早报] +name: 早间洞察 · 2026-05-19 --- ## 与您近期关注主题相关的事件 @@ -701,8 +699,7 @@ daily/ ```markdown --- -title: 2026-05-13 -tags: [daily] +name: 2026-05-13 --- ## 今日事件 @@ -825,8 +822,7 @@ ReMe 的 Auto-Dream 当晚把这次会话沉淀到: ```markdown # digest/procedural/webpack 编译卡死.md --- -title: webpack 编译卡死的排查路径 -tags: [agent_memory, programmatic, webpack] +name: webpack 编译卡死的排查路径 type: programmatic --- @@ -871,7 +867,6 @@ vite 的 esbuild 阶段也吃内存,先试 NODE_OPTIONS=--max-old-space-size=8 # digest/personal/代码风格.md --- type: personalization -tags: [user_profile] --- ## 命名 diff --git a/docs4/reme_design.md b/docs4/reme_design.md index 9f57f336..56d30120 100644 --- a/docs4/reme_design.md +++ b/docs4/reme_design.md @@ -42,7 +42,7 @@ reme4 version | 🔎 search | 🔍 `search` (`search_step`) | `call_server("search", query=…, …)` | 📥 `query:str` ⭐ | 🎚️ `limit:int=5`(>0) | 🎚️ `min_score:float=0.0` | ⚖️ `vector_weight:float=0.7` ∈[0,1](keyword 权 = 1-vw)| 🔀 `candidate_multiplier:float=3.0`(candidates = min(200, limit×mult))| 🔗 `expand_links:bool=True` | 🔢 `max_links_per_direction:int=10` | 🎚️ `search_filter:dict={}` | 📤 `answer` 每命中一行 `path:start-end [score=… vector=… keyword=…] text` + 缩进的 `→ outlinks (n)` / `← inlinks (n)` + `via predicate=… anchor=#…` | 📊 `metadata.results` / `metadata.link_expansion` / `metadata.counts={vector,keyword,returned,hybrid}` | 🛠️ 并行 `vector_search` + `keyword_search` → RRF 融合(K=60,按 chunk.id 合并)→ `min_score` 过滤 → `limit` 截断 → 邻居 meta 注入 | | 🧪 demo | 🪄 `demo_echo` (`demo_echo_step1` + `step2`) | `call_server("demo_echo", query=…, min_score=…)` | 📥 `query:str=""` | 🎚️ `min_score:float=0.5` | 🛠️ step1:`processed_query = query.strip().lower()`,`adjusted_min_score = min_score * 0.9`,写回 context | 📤 step2:`answer = "echo: {processed_query} (min_score={adjusted_min_score})"` | 📊 `metadata = {step, query, min_score, processed_query, adjusted_min_score}` | | 🌊 demo | 🌊 `stream_demo` (`stream_demo_step1` + `step2`) | `call_server("stream_demo", query=…, repeat=…, interval=…)` | 📥 `query:str=""` | 🎚️ `repeat:int=10` | 🎚️ `interval:float=0.1`(秒/字符)| 🛠️ step1:`stream_text = query * repeat` 写回 context | 📤 step2:按字符 `add_stream_string(ch, ChunkEnum.CONTENT)` 流式输出,`asyncio.sleep(interval)` 节流 | -| 📂 crud | 📖 `read` (`read_step`) | `call_server("read", path=…, …)` | 📥 `path:str` ⭐(**完整相对路径**,相对于 `working_dir`;绝对路径会被拒绝;非 `.md` 后缀拒绝)| 🎚️ `start_line:int=null`(1-based, 含端点)| 🎚️ `end_line:int=null`(1-based, 含端点)| 🎚️ `max_bytes:int=51200`(截断阈值)| 📤 `answer = 选中的行内容`,超过 `max_bytes` 时附加 `--- TRUNCATED ---` 续读指引(`start_line=…`)| 📊 `metadata.path` / `metadata.total_lines`(出错路径才会附带)| 🛠️ 流程:`BaseStep.resolve_path(raw, require_md=True)` → `aiofiles.os.stat` → `read_file_safe`(utf-8-sig BOM 容忍、UnicodeDecodeError fallback `errors=ignore`)→ `split("\n")` 切片 `[s-1:e]` → `truncate_text_output` 按字节截断保行 | +| 📂 crud | 📖 `read` (`read_step`) | `call_server("read", path=…, …)` | 📥 `path:str` ⭐(**完整相对路径**,相对于 vault;绝对路径会被拒绝;非 `.md` 后缀拒绝)| 🎚️ `start_line:int=null`(1-based, 含端点)| 🎚️ `end_line:int=null`(1-based, 含端点)| 🎚️ `max_bytes:int=51200`(截断阈值)| 📤 `answer = 选中的行内容`,超过 `max_bytes` 时附加 `--- TRUNCATED ---` 续读指引(`start_line=…`)| 📊 `metadata.path` / `metadata.total_lines`(出错路径才会附带)| 🛠️ 流程:`BaseStep.resolve_path(raw, require_md=True)` → `aiofiles.os.stat` → `read_file_safe`(utf-8-sig BOM 容忍、UnicodeDecodeError fallback `errors=ignore`)→ `split("\n")` 切片 `[s-1:e]` → `truncate_text_output` 按字节截断保行 | 使用示例: @@ -67,7 +67,7 @@ reme4 version reme4 reindex reme4 search query="latency 问题" limit=10 min_score=0.2 vector_weight=0.6 -# 读取 working_dir 下的 markdown(完整相对路径;无后缀自动补 .md;可按行切片或限制字节) +# 读取 vault 下的 markdown(完整相对路径;无后缀自动补 .md;可按行切片或限制字节) reme4 read path=Templates/Recipe.md reme4 read path=Notes start_line=1 end_line=20 reme4 read path=Big.md max_bytes=4096 @@ -77,13 +77,9 @@ reme4 search query="..." backend=mcp ``` @sen -| tags | stat | 返回特定tag信息 | -| tags | list | 返回所有tag列表 | -| crud | upload/download | 其他文件 | -| file | stat | path | -| file | list | path | - | -| graph | traverse | path="My Note" directtion=forward/backward depth=1 predicat=xxx | +| file | upload/download/move/delete/stat/list | 文件操作CRUD | +| property | read/update/delete | frontmatter CRUD | | +| graph | traverse/retarget | path="My Note" directtion=forward/backward depth=1 predicat=xxx | @wangce | crud | write | path="New Note" name="xxx" description="xxx" metadata={}, content="# Hello" (4 字段都必填,frontmatter 只写 name/description) | @@ -168,10 +164,8 @@ MemorySchema 1. markdown文件结构 @sen a. formatter: - ⅰ. title + ⅰ. name ⅱ. desc - ⅲ. tags - ⅳ. 2. memory文件结构目录 a. MEMORY.md b. msg/files -> daily/YYYYMMDD/YYYYMMDD.md + xxxx.md diff --git a/example.env b/example.env index 9824e6a4..5d91cb96 100644 --- a/example.env +++ b/example.env @@ -1,18 +1,6 @@ # LLM (required for most flows) LLM_API_KEY=sk-xxxx LLM_BASE_URL=https://xxxx/v1 - -# Embedding (optional; Application uses EMBEDDING_* or falls back to LLM_* when unset) +LLM_MODEL_NAME=xxxx #EMBEDDING_API_KEY=sk-xxxx -#EMBEDDING_BASE_URL=https://xxxx/v1 - -# Web search (optional) -#TAVILY_API_KEY=xxxx - -# Seekdb / pyseekdb (optional; requires Python >=3.11, installed automatically on 3.11+) -# Embedded: leave SEEKDB_HOST unset. Remote: set host (and port if not 2881). -#SEEKDB_HOST=127.0.0.1 -#SEEKDB_PORT=2881 -#SEEKDB_USER=root -#SEEKDB_PASSWORD= -#SEEKDB_DATABASE=test +#EMBEDDING_BASE_URL=https://xxxx/v1 \ No newline at end of file diff --git a/reme4/application.py b/reme4/application.py index 76adeccd..dc2a3fd4 100644 --- a/reme4/application.py +++ b/reme4/application.py @@ -18,11 +18,12 @@ class Application(BaseComponent): self.context = ApplicationContext(**kwargs) self._started_components: list[BaseComponent] = [] - working_path = Path(self.config.working_dir).absolute() - working_path.mkdir(parents=True, exist_ok=True) - (working_path / self.config.metadata_dir).mkdir(parents=True, exist_ok=True) - (working_path / self.config.daily_dir).mkdir(parents=True, exist_ok=True) - (working_path / self.config.digest_dir).mkdir(parents=True, exist_ok=True) + vault_path = Path(self.config.vault_dir).absolute() + vault_path.mkdir(parents=True, exist_ok=True) + (vault_path / self.config.metadata_dir).mkdir(parents=True, exist_ok=True) + (vault_path / self.config.daily_dir).mkdir(parents=True, exist_ok=True) + (vault_path / self.config.digest_dir).mkdir(parents=True, exist_ok=True) + (vault_path / self.config.resource_dir).mkdir(parents=True, exist_ok=True) if self.config.enable_logo: print_logo(self.config) diff --git a/reme4/components/base_component.py b/reme4/components/base_component.py index fa38dc96..9b59dcfe 100644 --- a/reme4/components/base_component.py +++ b/reme4/components/base_component.py @@ -121,24 +121,24 @@ class BaseComponent(ABC): # ----- Lookup -------------------------------------------------------- @property - def working_path(self) -> Path: - """Resolved working directory from app context or cwd.""" + def vault_path(self) -> Path: + """Resolved vault root path from app context or cwd.""" if self.app_context is None: return Path.cwd() - return Path(self.app_context.app_config.working_dir).absolute() + return Path(self.app_context.app_config.vault_dir).absolute() @property - def working_metadata_path(self) -> Path: - """Resolved metadata directory: working_path / metadata_dir, or absolute metadata_dir.""" + def vault_metadata_path(self) -> Path: + """Resolved metadata directory: vault_path / metadata_dir, or absolute metadata_dir.""" if self.app_context is None: return Path.cwd() / "metadata" - return self.working_path / self.app_context.app_config.metadata_dir + return self.vault_path / self.app_context.app_config.metadata_dir def to_vault_relative(self, path: str | Path) -> str: - """Return path relative to working_path; absolute path string if outside.""" + """Return path relative to vault_path; absolute path string if outside.""" abs_path = Path(path).absolute() try: - return str(abs_path.relative_to(self.working_path)) + return str(abs_path.relative_to(self.vault_path)) except ValueError: return str(abs_path) diff --git a/reme4/components/embedding/base_embedding_model.py b/reme4/components/embedding/base_embedding_model.py index 0d03ec61..64add5b5 100644 --- a/reme4/components/embedding/base_embedding_model.py +++ b/reme4/components/embedding/base_embedding_model.py @@ -52,7 +52,7 @@ class BaseEmbeddingModel(BaseComponent): @property def cache_path(self) -> Path: """Disk path for the embedding cache file.""" - return self.working_metadata_path / "embedding_cache" / f"{self.name}_{self.cache_version}.npz" + return self.vault_metadata_path / "embedding_cache" / f"{self.name}_{self.cache_version}.npz" async def _start(self) -> None: """Load cache from disk on startup.""" diff --git a/reme4/components/file_graph/base_file_graph.py b/reme4/components/file_graph/base_file_graph.py index a5ef4110..f68488fd 100644 --- a/reme4/components/file_graph/base_file_graph.py +++ b/reme4/components/file_graph/base_file_graph.py @@ -4,7 +4,7 @@ from abc import abstractmethod from pathlib import Path from ..base_component import BaseComponent -from ...enumeration import ComponentEnum +from ...enumeration import ComponentEnum, LinkScopeEnum from ...schema import FileLink, FileNode @@ -17,7 +17,7 @@ class BaseFileGraph(BaseComponent): super().__init__(**kwargs) self.graph_name: str = graph_name or self.name self.graph_version: str = graph_version - self.graph_path: Path = self.working_metadata_path / self.component_type.value + self.graph_path: Path = self.vault_metadata_path / self.component_type.value self.graph_path.mkdir(parents=True, exist_ok=True) # -- Lifecycle --------------------------------------------------------- @@ -67,9 +67,30 @@ class BaseFileGraph(BaseComponent): # -- Link access ------------------------------------------------------- @abstractmethod - async def get_outlinks(self, path: str) -> list[FileLink]: - """Return outgoing links for *path*.""" + async def get_outlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: + """Return outgoing links for *path*. + + ``scope=REAL`` (default) → edges whose target is an indexed + (real) node. ``scope=VIRTUAL`` → only dangling edges (target + was referenced but never upserted, or was deleted). ``ALL`` + → both. + """ @abstractmethod - async def get_inlinks(self, path: str) -> list[FileLink]: - """Return incoming links for *path*.""" + async def get_inlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: + """Return incoming links for *path*. + + ``scope=REAL`` (default) → returns the inbound edges only when + *path* itself is a real node. ``scope=VIRTUAL`` → returns + inbound edges only when *path* is virtual (useful for retarget + / lint that must see references to non-existent targets). + ``ALL`` → both. + """ diff --git a/reme4/components/file_graph/local_file_graph.py b/reme4/components/file_graph/local_file_graph.py index 0966ebfc..f34293de 100644 --- a/reme4/components/file_graph/local_file_graph.py +++ b/reme4/components/file_graph/local_file_graph.py @@ -4,6 +4,7 @@ from pathlib import Path from .base_file_graph import BaseFileGraph from ..component_registry import R +from ...enumeration import LinkScopeEnum from ...schema import FileLink, FileNode @@ -119,15 +120,39 @@ class LocalFileGraph(BaseFileGraph): # -- Link access ------------------------------------------------------- - async def get_outlinks(self, path: str) -> list[FileLink]: + async def get_outlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: + # Source must be real (only real nodes carry a ``links`` payload). + # Targets may be real or virtual; ``scope`` selects which to surface. node = self._nodes.get(path) if node is None: return [] - return [lnk for lnk in node.links if lnk.target_path and lnk.target_path in self._nodes] + return [lnk for lnk in node.links if lnk.target_path and _match_target(lnk.target_path, self._nodes, scope)] - async def get_inlinks(self, path: str) -> list[FileLink]: - if path not in self._nodes: - return [] + async def get_inlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: + # ``_inverse`` keys real targets; ``_pending`` keys virtual ones. + # The queried ``path`` lives in at most one bucket, so ``scope`` + # is satisfied by selecting which bucket to read. + sources: set[str] = set() + if scope in (LinkScopeEnum.REAL, LinkScopeEnum.ALL): + sources |= self._inverse.get(path, set()) + if scope in (LinkScopeEnum.VIRTUAL, LinkScopeEnum.ALL): + sources |= self._pending.get(path, set()) return [ - link for src in self._inverse.get(path, ()) for link in self._nodes[src].links if link.target_path == path + link for src in sources if src in self._nodes for link in self._nodes[src].links if link.target_path == path ] + + +def _match_target(target_path: str, nodes: dict, scope: LinkScopeEnum) -> bool: + """Whether an edge into ``target_path`` should be surfaced under ``scope``.""" + if scope is LinkScopeEnum.ALL: + return True + is_real = target_path in nodes + return is_real if scope is LinkScopeEnum.REAL else not is_real diff --git a/reme4/components/file_graph/neo4j_file_graph.py b/reme4/components/file_graph/neo4j_file_graph.py index 869c2b18..f580c9ed 100644 --- a/reme4/components/file_graph/neo4j_file_graph.py +++ b/reme4/components/file_graph/neo4j_file_graph.py @@ -2,7 +2,7 @@ Property-graph mapping: - Real node: (:File {path, st_mtime, title, description, tags, + Real node: (:File {path, st_mtime, name, description, chunk_ids, links_json, extra_json}) Virtual node: (:File {path}) — placeholder created when something links to a path that hasn't been upserted yet. @@ -26,7 +26,7 @@ graph from per-node payloads after backend repair / migration. Adjacency policy: trusts ``FileLink.path`` directly — no internal wikilink resolution. The parser pipeline (with the external resolver) produces safe links where ``link.path`` is already a -vault-relative target. +target relative to the vault. Conditional dependency: the ``neo4j`` driver loads lazily; the import error fires at ``_start`` (boot), not at first call. @@ -39,20 +39,20 @@ from typing import Any from .base_file_graph import BaseFileGraph from ..component_registry import R +from ...enumeration import LinkScopeEnum from ...schema import FileLink, FileNode from ...schema.file_node import FileFrontMatter -_TYPED_FRONTMATTER_FIELDS = {"title", "description", "tags"} +_TYPED_FRONTMATTER_FIELDS = {"name", "description"} _LINK_FIELDS = {"source_path", "target_path", "target_anchor", "predicate"} # Properties that distinguish a "real" node from a virtual placeholder. # Listed for the demote query (delete_nodes) so we can REMOVE them all. _REAL_PROPS = ( "st_mtime", - "title", + "name", "description", - "tags", "chunk_ids", "links_json", "extra_json", @@ -332,16 +332,22 @@ class Neo4jFileGraph(BaseFileGraph): # -- Link access ------------------------------------------------------- - async def get_outlinks(self, path: str) -> list[FileLink]: - """Outgoing links from ``path``. Source must be real; targets - into virtual nodes are excluded so dangling refs are invisible.""" + async def get_outlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: + """Outgoing links from ``path``. Source must be real; ``scope`` + selects which targets to surface (REAL / VIRTUAL / ALL). + """ + target_filter = _neo4j_scope_filter("t", scope) async with self._session() as session: rec = await session.run( - """ - MATCH (s:File {path: $path}) + f""" + MATCH (s:File {{path: $path}}) WHERE s.links_json IS NOT NULL MATCH (s)-[r:LINKS]->(t:File) - WHERE t.links_json IS NOT NULL + WHERE 1=1 {target_filter} RETURN t.path AS target, r.anchor AS anchor, r.predicate AS predicate, r.idx AS idx ORDER BY r.idx ASC @@ -359,15 +365,25 @@ class Neo4jFileGraph(BaseFileGraph): for row in rows ] - async def get_inlinks(self, path: str) -> list[FileLink]: - """Incoming links to ``path`` (must be real). Sources are always - real because virtual nodes never have outgoing edges.""" + async def get_inlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: + """Incoming links to ``path`` from real sources. + + ``scope`` selects whether ``path`` itself must be real / virtual / + either; sources are always real (virtual nodes have no outgoing + edges to begin with). + """ + target_filter = _neo4j_scope_filter("t", scope) async with self._session() as session: rec = await session.run( - """ - MATCH (t:File {path: $path}) - WHERE t.links_json IS NOT NULL + f""" + MATCH (t:File {{path: $path}}) + WHERE 1=1 {target_filter} MATCH (s:File)-[r:LINKS]->(t) + WHERE s.links_json IS NOT NULL RETURN r.anchor AS anchor, r.predicate AS predicate, r.idx AS idx, s.path AS source ORDER BY s.path ASC, r.idx ASC @@ -394,9 +410,8 @@ class Neo4jFileGraph(BaseFileGraph): return { "path": node.path, "st_mtime": float(node.st_mtime), - "title": fm.title or "", + "name": fm.name or "", "description": fm.description or "", - "tags": list(fm.tags or []), "chunk_ids": list(node.chunk_ids or []), "links_json": json.dumps( [link.model_dump(exclude_none=True) for link in node.links], @@ -434,9 +449,8 @@ class Neo4jFileGraph(BaseFileGraph): except Exception: continue fm_kwargs: dict[str, Any] = { - "title": d.get("title", "") or "", + "name": d.get("name", "") or "", "description": d.get("description", "") or "", - "tags": d.get("tags") or None, } fm_kwargs.update( {k: v for k, v in extras.items() if k not in _TYPED_FRONTMATTER_FIELDS}, @@ -448,3 +462,18 @@ class Neo4jFileGraph(BaseFileGraph): chunk_ids=[str(c) for c in (d.get("chunk_ids") or [])], front_matter=FileFrontMatter(**fm_kwargs), ) + + +def _neo4j_scope_filter(node_var: str, scope: LinkScopeEnum) -> str: + """Cypher predicate that restricts ``node_var`` to the requested scope. + + Encodes the same convention used everywhere in this backend: + ``links_json IS NOT NULL`` ⇔ real node; ``IS NULL`` ⇔ virtual + placeholder. ``ALL`` returns an empty fragment so callers can + splice it after ``WHERE 1=1``. + """ + if scope is LinkScopeEnum.REAL: + return f"AND {node_var}.links_json IS NOT NULL" + if scope is LinkScopeEnum.VIRTUAL: + return f"AND {node_var}.links_json IS NULL" + return "" diff --git a/reme4/components/file_graph/nx_file_graph.py b/reme4/components/file_graph/nx_file_graph.py index 0a007c45..e41cafe7 100644 --- a/reme4/components/file_graph/nx_file_graph.py +++ b/reme4/components/file_graph/nx_file_graph.py @@ -10,6 +10,7 @@ except ImportError: from .base_file_graph import BaseFileGraph from ..component_registry import R +from ...enumeration import LinkScopeEnum from ...schema import FileLink, FileNode @@ -97,18 +98,39 @@ class NxFileGraph(BaseFileGraph): # -- Link access ------------------------------------------------------- - async def get_outlinks(self, path: str) -> list[FileLink]: + async def get_outlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: + # Source must be real; targets may be virtual placeholders. + # ``scope`` picks real / virtual / both targets. nodes_view = self._graph.nodes if path not in nodes_view or "node" not in nodes_view[path]: return [] return [ d["link"] - for _, target, d in self._graph.out_edges(path, data=True) - if "link" in d and "node" in nodes_view[target] + for _, tgt, d in self._graph.out_edges(path, data=True) + if "link" in d and _match_node(nodes_view, tgt, scope) ] - async def get_inlinks(self, path: str) -> list[FileLink]: + async def get_inlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: + # ``path`` is a single node — its realness selects which scope + # produces a non-empty result (REAL ↔ real node, VIRTUAL ↔ + # virtual placeholder; ALL is always allowed). nodes_view = self._graph.nodes - if path not in nodes_view or "node" not in nodes_view[path]: + if path not in nodes_view or not _match_node(nodes_view, path, scope): return [] return [d["link"] for _, _, d in self._graph.in_edges(path, data=True) if "link" in d] + + +def _match_node(nodes_view, key: str, scope: LinkScopeEnum) -> bool: + """Whether ``key`` satisfies ``scope`` under the nx node-realness convention.""" + if scope is LinkScopeEnum.ALL: + return True + is_real = "node" in nodes_view[key] + return is_real if scope is LinkScopeEnum.REAL else not is_real diff --git a/reme4/components/file_parser/linked_file_parser.py b/reme4/components/file_parser/linked_file_parser.py index 6bac9c3f..2386e4b5 100644 --- a/reme4/components/file_parser/linked_file_parser.py +++ b/reme4/components/file_parser/linked_file_parser.py @@ -11,6 +11,15 @@ blocks (table / code / list / paragraph) split on internal boundaries and each piece is annotated ``[Part X/N]``. Wikilinks in the body are extracted as graph edges, with optional Dataview-style typed predicates (line-level ``predicate:: [[X]]`` or inline-bracketed ``[predicate:: [[X]]]``). + +Wikilink convention. Targets are taken **literally** — ``[[X]]`` +becomes ``target_path="X"`` with no implicit ``.md``, no short-form +basename search, no folder-note expansion. The recommended form is a +full path relative to the vault with extension, e.g. +``[[digest/alice/alice.md]]``. Anything else (``[[Alice]]``, +``[[digest/alice/alice]]``) is also stored verbatim and will be +flagged by ``lint:dangling`` because no node lives at that literal +path — the parser does no validation, lint is the contract enforcer. """ from __future__ import annotations @@ -25,8 +34,6 @@ import frontmatter from .base_file_parser import BaseFileParser from ..component_registry import R -from ..file_graph import BaseFileGraph -from ...enumeration import ComponentEnum from ...schema import ( FileChunk, FileLink, @@ -35,53 +42,6 @@ from ...schema import ( ) -# -- Wikilink resolution -------------------------------------------------- -# -# Wikilinks are a markdown user-facing convention: ``[[Alice]]`` should -# resolve to ``topics/Alice/Alice.md`` (or wherever the file lives). -# This short-form / implicit-``.md`` / folder-note resolution lives here -# at the markdown boundary rather than as a generic utility — file-IO -# steps require full vault-relative paths and never use these helpers. - - -def _complete_md(target: str) -> str: - """Apply implicit ``.md`` rule for wikilink targets.""" - if not target: - return target - last = target.rsplit("/", 1)[-1] - return target if "." in last else target + ".md" - - -def _filter_folder_note(target: str, paths: list[str]) -> list[str]: - """Apply folder-note rule: when both ``X.md`` and ``X/X.md`` exist, - prefer ``X/X.md``. Sorted for determinism. - """ - if not paths: - return [] - stem = Path(target).stem - folder_hits = sorted(p for p in paths if Path(p).parent.name == stem) - return folder_hits or sorted(paths) - - -async def _resolve_wikilink(graph: BaseFileGraph, target: str) -> list[str]: - """Resolve a wikilink target to vault-relative path(s). - - Returns: - ``[path]`` for an unambiguous match, - ``[path, path, ...]`` for short-form ambiguity (caller may - fan out one FileLink per candidate), or - ``[]`` when nothing matches (dangling — caller drops the link). - """ - if not target: - return [] - target = _complete_md(target) - if "/" in target: - nodes = await graph.get_nodes([target]) - return [target] if nodes else [] - matches = [n.path for n in await graph.get_nodes() if Path(n.path).name == target] - return _filter_folder_note(target, matches) - - # -- Wikilink extraction -------------------------------------------------- @@ -149,16 +109,12 @@ def _predicate_for( return None -async def _extract_links( - graph: BaseFileGraph, - text: str, - source_path: str, -) -> list[FileLink]: - """Find every wikilink in ``text``, resolve targets, emit FileLinks. +def _extract_links(text: str, source_path: str) -> list[FileLink]: + """Find every wikilink in ``text`` and emit FileLinks with literal targets. - Short-path ambiguity **expands** into one FileLink per candidate so - the body's wikilink is recorded against every plausible target. - Dangling targets are dropped. Results are deduped by + No resolution is performed: ``target_path`` is the bracket contents + verbatim. ``lint:dangling`` checks whether the literal target exists + in the graph. Results are deduped by ``(target_path, predicate, target_anchor)`` preserving order. """ if not text: @@ -173,22 +129,18 @@ async def _extract_links( anchor_raw = wm.group("anchor") anchor = anchor_raw.strip() if anchor_raw else None predicate = _predicate_for(text, wm.start(), inline_spans) - resolved_paths = await _resolve_wikilink(graph, target) - if not resolved_paths: + key = (target, predicate, anchor) + if key in seen: continue - for resolved in resolved_paths: - key = (resolved, predicate, anchor) - if key in seen: - continue - seen.add(key) - out.append( - FileLink( - source_path=source_path, - target_path=resolved, - target_anchor=anchor, - predicate=predicate, - ), - ) + seen.add(key) + out.append( + FileLink( + source_path=source_path, + target_path=target, + target_anchor=anchor, + predicate=predicate, + ), + ) return out @@ -276,33 +228,12 @@ class LinkedFileParser(BaseFileParser): encoding: str = "utf-8", chunk_chars: int = 2000, embed_toc: bool = True, - file_graph: str = "default", **kwargs, ): super().__init__(**kwargs) self.encoding = encoding self.chunk_chars = max(100, chunk_chars) self.embed_toc = embed_toc - self._file_graph_name: str = file_graph - - def _resolve_file_graph(self) -> BaseFileGraph | None: - """Lazily fetch the configured file_graph from app_context. - - Lazy (rather than ``_start``) so the parser doesn't impose a - component start-order constraint, and so tests can construct - the parser without a graph wired up. - """ - if self.app_context is None: - return None - graphs = self.app_context.components.get(ComponentEnum.FILE_GRAPH, {}) - graph = graphs.get(self._file_graph_name) - if graph is None: - return None - if not isinstance(graph, BaseFileGraph): - raise TypeError( - f"Expected BaseFileGraph, got {type(graph).__name__}", - ) - return graph async def parse(self, path: str | Path) -> tuple[FileNode, list[FileChunk]]: from mistletoe.markdown_renderer import MarkdownRenderer @@ -318,10 +249,7 @@ class LinkedFileParser(BaseFileParser): tree = self._build_tree(Document(post.content), renderer) chunks = self._chunk_node(tree, "", "", rel_path, renderer) - links: list[FileLink] = [] - graph = self._resolve_file_graph() - if graph is not None: - links = await _extract_links(graph, post.content, rel_path) + links = _extract_links(post.content, rel_path) if post.content else [] node = FileNode( path=rel_path, diff --git a/reme4/components/file_store/base_file_store.py b/reme4/components/file_store/base_file_store.py index 7237c60a..7788487e 100644 --- a/reme4/components/file_store/base_file_store.py +++ b/reme4/components/file_store/base_file_store.py @@ -3,7 +3,7 @@ from abc import abstractmethod from ..base_component import BaseComponent -from ...enumeration import ComponentEnum +from ...enumeration import ComponentEnum, LinkScopeEnum from ...schema import FileChunk, FileLink, FileNode @@ -22,7 +22,7 @@ class BaseFileStore(BaseComponent): super().__init__(**kwargs) self.store_name = store_name or self.name self.store_version = store_version - self.store_path = self.working_metadata_path / self.component_type.value / self.store_name + self.store_path = self.vault_metadata_path / self.component_type.value / self.store_name self.store_path.mkdir(parents=True, exist_ok=True) # -- CRUD ------------------------------------------------------------ @@ -40,12 +40,20 @@ class BaseFileStore(BaseComponent): """Return file nodes; None = all nodes; missing paths are skipped.""" @abstractmethod - async def get_outlinks(self, path: str) -> list[FileLink]: - """Return outgoing links for *path*.""" + async def get_outlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: + """Return outgoing links for *path*. See ``BaseFileGraph.get_outlinks`` for scope semantics.""" @abstractmethod - async def get_inlinks(self, path: str) -> list[FileLink]: - """Return incoming links for *path*.""" + async def get_inlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: + """Return incoming links for *path*. See ``BaseFileGraph.get_inlinks`` for scope semantics.""" @abstractmethod async def clear(self) -> None: diff --git a/reme4/components/file_store/local_file_store.py b/reme4/components/file_store/local_file_store.py index d3867038..92c92e02 100644 --- a/reme4/components/file_store/local_file_store.py +++ b/reme4/components/file_store/local_file_store.py @@ -8,6 +8,7 @@ from ..component_registry import R from ..embedding import BaseEmbeddingModel from ..file_graph import BaseFileGraph from ..keyword_index import BaseKeywordIndex +from ...enumeration import LinkScopeEnum from ...schema import FileChunk, FileLink, FileNode from ...utils import batch_cosine_similarity @@ -159,13 +160,21 @@ class LocalFileStore(BaseFileStore): assert self.file_graph is not None return await self.file_graph.get_nodes(paths) - async def get_outlinks(self, path: str) -> list[FileLink]: + async def get_outlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: assert self.file_graph is not None - return await self.file_graph.get_outlinks(path) + return await self.file_graph.get_outlinks(path, scope) - async def get_inlinks(self, path: str) -> list[FileLink]: + async def get_inlinks( + self, + path: str, + scope: LinkScopeEnum = LinkScopeEnum.REAL, + ) -> list[FileLink]: assert self.file_graph is not None - return await self.file_graph.get_inlinks(path) + return await self.file_graph.get_inlinks(path, scope) async def clear(self) -> None: assert self.file_graph is not None diff --git a/reme4/components/keyword_index/base_keyword_index.py b/reme4/components/keyword_index/base_keyword_index.py index 0be21c58..86433e97 100644 --- a/reme4/components/keyword_index/base_keyword_index.py +++ b/reme4/components/keyword_index/base_keyword_index.py @@ -19,7 +19,7 @@ class BaseKeywordIndex(BaseComponent): self.tokenizer = self.bind(tokenizer, BaseTokenizer, default_factory=RegexTokenizer) self.index_version = index_version - self.index_path = self.working_metadata_path / self.component_type.value + self.index_path = self.vault_metadata_path / self.component_type.value self.index_path.mkdir(parents=True, exist_ok=True) async def _start(self) -> None: diff --git a/reme4/enumeration/__init__.py b/reme4/enumeration/__init__.py index 9887ddee..06e6075a 100644 --- a/reme4/enumeration/__init__.py +++ b/reme4/enumeration/__init__.py @@ -2,8 +2,10 @@ from .chunk_enum import ChunkEnum from .component_enum import ComponentEnum +from .link_scope_enum import LinkScopeEnum __all__ = [ "ChunkEnum", "ComponentEnum", + "LinkScopeEnum", ] diff --git a/reme4/enumeration/link_scope_enum.py b/reme4/enumeration/link_scope_enum.py new file mode 100644 index 00000000..c00163e6 --- /dev/null +++ b/reme4/enumeration/link_scope_enum.py @@ -0,0 +1,17 @@ +"""Link-scope enumeration: which edges to return from get_inlinks / get_outlinks.""" + +from enum import Enum + + +class LinkScopeEnum(str, Enum): + """Which subset of edges a link query should return. + + A file-graph edge is **real** when both endpoints are indexed nodes, + and **virtual** (dangling / pending) when one endpoint is a placeholder + — a path referenced by a wikilink but never upserted, or upserted + then deleted. + """ + + REAL = "real" + VIRTUAL = "virtual" + ALL = "all" diff --git a/reme4/schema/application_config.py b/reme4/schema/application_config.py index c56b336e..6e82e527 100644 --- a/reme4/schema/application_config.py +++ b/reme4/schema/application_config.py @@ -28,10 +28,14 @@ class ApplicationConfig(BaseModel): """Root config for the ReMe application.""" app_name: str = Field(default=os.getenv("APP_NAME", "ReMe"), description="Application display name") - working_dir: str = Field(default=".reme", description="Working directory for runtime files") + vault_dir: str = Field(default=".reme", description="Vault root directory (knowledge base) for runtime files") metadata_dir: str = Field(default="reme_metadata", description="Subdirectory for ReMe persistent state") daily_dir: str = Field(default="daily", description="Subdirectory for daily memory") digest_dir: str = Field(default="digest", description="Subdirectory for digest") + resource_dir: str = Field( + default="resource", + description="Subdirectory for passively-received external assets (upload bucket)", + ) enable_logo: bool = Field(default=True, description="Show ASCII logo on startup") language: str = Field(default="", description="Default language for LLM interactions") log_to_console: bool = Field(default=True, description="Log to console") diff --git a/reme4/schema/file_chunk.py b/reme4/schema/file_chunk.py index f7de723b..c02cc98f 100644 --- a/reme4/schema/file_chunk.py +++ b/reme4/schema/file_chunk.py @@ -8,7 +8,7 @@ from .emb_node import EmbNode class FileChunk(EmbNode): """A chunk of a file with positional info and per-stage retrieval scores.""" - path: str = Field(default="", description="Vault-relative file path") + path: str = Field(default="", description="Path relative to the vault") start_line: int = Field(default=0, description="Inclusive start line (0-based)") end_line: int = Field(default=0, description="Exclusive end line") scores: dict[str, float] = Field(default_factory=dict, description="Retrieval scores keyed by stage") diff --git a/reme4/schema/file_front_matter.py b/reme4/schema/file_front_matter.py index 830fc72e..febc80d0 100644 --- a/reme4/schema/file_front_matter.py +++ b/reme4/schema/file_front_matter.py @@ -10,7 +10,7 @@ class FileFrontMatter(BaseModel): model_config = ConfigDict(extra="allow") - name: str = Field(default="", description="Document title") + name: str = Field(default="", description="Document name") description: str = Field(default="", description="Document description") @property diff --git a/reme4/schema/file_node.py b/reme4/schema/file_node.py index 9be276f8..76690fa3 100644 --- a/reme4/schema/file_node.py +++ b/reme4/schema/file_node.py @@ -9,7 +9,7 @@ from .file_link import FileLink class FileNode(BaseModel): """A vault file as a graph node.""" - path: str = Field(default=..., description="Vault-relative file path") + path: str = Field(default=..., description="Path relative to the vault") st_mtime: float = Field(default=..., description="Filesystem mtime (seconds)") links: list[FileLink] = Field(default_factory=list, description="Outgoing wikilinks") chunk_ids: list[str] = Field(default_factory=list, description="Owned FileChunk ids") diff --git a/reme4/schema/request.py b/reme4/schema/request.py index c1d9b2ab..5c744c08 100644 --- a/reme4/schema/request.py +++ b/reme4/schema/request.py @@ -8,4 +8,4 @@ class Request(BaseModel): model_config = ConfigDict(extra="allow") - metadata: dict = Field(default_factory=dict, description="Request metadata for context") + metadata: dict | None = Field(default=None, description="Request metadata for context") diff --git a/reme4/steps/background/index_changes.py b/reme4/steps/background/index_changes.py index 76759555..f6f1d135 100644 --- a/reme4/steps/background/index_changes.py +++ b/reme4/steps/background/index_changes.py @@ -66,7 +66,7 @@ class IndexChangesStep(BaseStep): for path in deleted: p = Path(path).absolute() try: - rel_deleted.append(str(p.relative_to(self.working_path))) + rel_deleted.append(str(p.relative_to(self.vault_path))) except ValueError: rel_deleted.append(str(p)) try: diff --git a/reme4/steps/background/update_store.py b/reme4/steps/background/update_store.py index a72357fc..9ecde745 100644 --- a/reme4/steps/background/update_store.py +++ b/reme4/steps/background/update_store.py @@ -22,10 +22,10 @@ class UpdateStoreStep(BaseStep): raw: list[str] = self.context.get("watch_paths", []) suffixes: list[str] = self.context.get("suffix_filters", ["md"]) - working_path = self.working_path + vault_path = self.vault_path paths = [raw] if isinstance(raw, str) else raw - watch_paths = [working_path / x for x in paths if (working_path / x).exists()] + watch_paths = [vault_path / x for x in paths if (vault_path / x).exists()] existing: dict[str, float] = {} for path in watch_paths: @@ -39,7 +39,7 @@ class UpdateStoreStep(BaseStep): existing[str(abs_p)] = abs_p.stat().st_mtime indexed: dict[str, float] = { - str(Path(n.path) if Path(n.path).is_absolute() else working_path / n.path): n.st_mtime + str(Path(n.path) if Path(n.path).is_absolute() else vault_path / n.path): n.st_mtime for n in await self.file_store.get_nodes() } diff --git a/reme4/steps/background/watch_changes.py b/reme4/steps/background/watch_changes.py index fee3a4c8..68d28764 100644 --- a/reme4/steps/background/watch_changes.py +++ b/reme4/steps/background/watch_changes.py @@ -39,9 +39,9 @@ class WatchChangesStep(BaseStep): raw = self.context.get("watch_paths", []) paths = [raw] if isinstance(raw, str) else raw - valid_paths = [self.working_path / x for x in paths if (self.working_path / x).exists()] + valid_paths = [self.vault_path / x for x in paths if (self.vault_path / x).exists()] if not valid_paths: - raise RuntimeError(f"No valid watch paths under {self.working_path}: {paths}") + raise RuntimeError(f"No valid watch paths under {self.vault_path}: {paths}") self.logger.info(f"Watching: {[str(p) for p in valid_paths]}") async for raw_changes in awatch( diff --git a/reme4/steps/base_step.py b/reme4/steps/base_step.py index b3a50aa6..d60564e1 100644 --- a/reme4/steps/base_step.py +++ b/reme4/steps/base_step.py @@ -84,11 +84,11 @@ class BaseStep(ABC): return result @property - def working_path(self) -> Path: - """Resolved working directory from app context or cwd.""" + def vault_path(self) -> Path: + """Resolved vault root path from app context or cwd.""" if self.app_context is None: return Path.cwd() - return Path(self.app_context.app_config.working_dir).absolute() + return Path(self.app_context.app_config.vault_dir).absolute() def _resolve( self, diff --git a/reme4/steps/crud/_file_io.py b/reme4/steps/crud/_file_io.py index 7dfabccd..8b02c4b4 100644 --- a/reme4/steps/crud/_file_io.py +++ b/reme4/steps/crud/_file_io.py @@ -38,11 +38,11 @@ _STANDARD_TEXT_EXTS = { _NON_STANDARD_EXTS = {".csv", ".bat", ".cmd", ".reg"} -def resolve_path(working_path: Path, raw: str) -> tuple[Path | None, str | None]: - """Resolve a `path=` argument against ``working_path``. +def resolve_path(vault_path: Path, raw: str) -> tuple[Path | None, str | None]: + """Resolve a `path=` argument against ``vault_path``. Rules: - - Relative paths are joined under ``working_path``. + - Relative paths are joined under ``vault_path``. - Absolute paths are accepted and returned as-is; a warning is logged recommending relative paths, but the read still proceeds. Returns ``(abs_path, None)`` on success, or ``(None, error_message)`` on failure @@ -57,7 +57,7 @@ def resolve_path(working_path: Path, raw: str) -> tuple[Path | None, str | None] if p.is_absolute(): logger.info("absolute path detected, recommending relative paths") return p, None - return working_path / p, None + return vault_path / p, None def gate_md(target: Path) -> tuple[Path, bool]: diff --git a/reme4/steps/crud/append.py b/reme4/steps/crud/append.py index 91b493c7..ed4c0913 100644 --- a/reme4/steps/crud/append.py +++ b/reme4/steps/crud/append.py @@ -28,7 +28,7 @@ class AppendStep(BaseStep): content = self.context.get("content") content_str = "" if content is None else str(content) - target, err = resolve_path(self.working_path, raw) + target, err = resolve_path(self.vault_path, raw) if err: self._fail(err) return None diff --git a/reme4/steps/crud/edit.py b/reme4/steps/crud/edit.py index 14352b00..14b51b21 100644 --- a/reme4/steps/crud/edit.py +++ b/reme4/steps/crud/edit.py @@ -37,7 +37,7 @@ class EditStep(BaseStep): old_str = str(old) new_str = str(new) - target, err = resolve_path(self.working_path, raw) + target, err = resolve_path(self.vault_path, raw) if err: self._fail(err) return None diff --git a/reme4/steps/crud/read.py b/reme4/steps/crud/read.py index 387d0725..52cf8c98 100644 --- a/reme4/steps/crud/read.py +++ b/reme4/steps/crud/read.py @@ -1,4 +1,4 @@ -"""Read a markdown file from the vault, with line-range slicing and byte-truncation.""" +"""Read a markdown file from vault_dir, with line-range slicing and byte-truncation.""" from ._file_io import ( NON_MD_WARNING, @@ -28,7 +28,7 @@ class ReadStep(BaseStep): start_line = self.context.get("start_line") end_line = self.context.get("end_line") - target, err = resolve_path(self.working_path, raw) + target, err = resolve_path(self.vault_path, raw) if err: self._fail(err) return None diff --git a/reme4/steps/crud/write.py b/reme4/steps/crud/write.py index 5f60c242..ba0f478f 100644 --- a/reme4/steps/crud/write.py +++ b/reme4/steps/crud/write.py @@ -30,7 +30,7 @@ class WriteStep(BaseStep): content = self.context.get("content") content = "" if content is None else str(content) - target, err = resolve_path(self.working_path, raw) + target, err = resolve_path(self.vault_path, raw) if err: self._fail(err) return None diff --git a/tests4/unittest/test_background_steps.py b/tests4/unittest/test_background_steps.py index 7a0af518..2c6878ad 100644 --- a/tests4/unittest/test_background_steps.py +++ b/tests4/unittest/test_background_steps.py @@ -115,7 +115,7 @@ def test_update_store_initial_all_added(): async def run(): with tempfile.TemporaryDirectory() as tmpdir, temp_chdir(tmpdir): - # Use Path.cwd() as the basis so we match BaseStep.working_path on macOS + # Use Path.cwd() as the basis so we match BaseStep.vault_path on macOS # (where /var resolves to /private/var via a symlink). cwd = Path.cwd() vault = cwd / "vault" diff --git a/tests4/unittest/test_chunked_file_parser.py b/tests4/unittest/test_chunked_file_parser.py index 6529beb0..ec3cc6a7 100644 --- a/tests4/unittest/test_chunked_file_parser.py +++ b/tests4/unittest/test_chunked_file_parser.py @@ -285,7 +285,7 @@ def test_parse_links_in_file(): async def run(): content = ( "---\n" - "title: demo\n" + "name: demo\n" "---\n" "\n" "Intro paragraph with [[alpha]] and [[beta#h2]].\n" @@ -321,7 +321,7 @@ def test_parse_links_empty_when_no_content(): empty_path = f.name # Front-matter-only file with tempfile.NamedTemporaryFile(mode="w", delete=False, suffix=".md") as f: - f.write("---\ntitle: x\n---\n") + f.write("---\nname: x\n---\n") fm_only_path = f.name try: diff --git a/tests4/unittest/test_crud_md_steps.py b/tests4/unittest/test_crud_md_steps.py deleted file mode 100644 index a9ab57e2..00000000 --- a/tests4/unittest/test_crud_md_steps.py +++ /dev/null @@ -1,1071 +0,0 @@ -"""End-to-end tests for reme4 crud_md steps: spawn `reme4 start`, drive via HTTP, -verify responses, then shut down. Each test uses an isolated cwd so the working_dir -(.reme by default) does not collide. - -CLI rule: `path=` is relative-only, rooted at the reme working_dir. A bare path with -no suffix auto-appends `.md`; non-`.md` suffix is rejected. Absolute paths are -rejected. -""" - -import asyncio -import os -import tempfile -import warnings -from pathlib import Path - -from reme4.utils import call_action, call_and_check, mock_reme_server - -warnings.filterwarnings("ignore", category=DeprecationWarning, module="jieba") -warnings.filterwarnings("ignore", category=DeprecationWarning, module="pkg_resources") - - -class _temp_chdir: - """chdir to path for the duration of the block; restore on exit.""" - - def __init__(self, path): - self.path = path - self._old = None - - def __enter__(self): - self._old = os.getcwd() - os.chdir(self.path) - return self - - def __exit__(self, *exc): - os.chdir(self._old) - - -def _run(coro): - """Run an async coroutine on a fresh isolated event loop.""" - asyncio.run(coro) - - -def _seed_md(working_dir: Path, rel: str, body: str) -> Path: - target = working_dir / rel - target.parent.mkdir(parents=True, exist_ok=True) - target.write_text(body, encoding="utf-8") - return target - - -# --------------------------------------------------------------------------- -# Individual job tests -# --------------------------------------------------------------------------- - - -def test_read_relative_path(): - """`reme4 read path=Templates/Recipe.md` returns the file body from .reme/.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - body = "# Recipe\n\nMix flour and water.\n" - _seed_md(working, "Templates/Recipe.md", body) - async with mock_reme_server() as (host, port): - await call_and_check( - "read", - host=host, - port=port, - path="Templates/Recipe.md", - validator=lambda r: ( - isinstance(r, dict) - and r.get("success") is True - and "# Recipe" in str(r.get("answer", "")) - and "flour and water" in str(r.get("answer", "")) - ), - ) - print("✓ test_read_relative_path passed") - - _run(run()) - - -def test_read_no_suffix_autoappends_md(): - """A bare path with no suffix auto-appends `.md`.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "Templates/Recipe.md", "auto-md\n") - async with mock_reme_server() as (host, port): - await call_and_check( - "read", - host=host, - port=port, - path="Templates/Recipe", - validator=lambda r: ( - isinstance(r, dict) and r.get("success") is True and "auto-md" in str(r.get("answer", "")) - ), - ) - print("✓ test_read_no_suffix_autoappends_md passed") - - _run(run()) - - -def test_read_line_range(): - """start_line / end_line slice the file 1-based, inclusive.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "Notes.md", "L1\nL2\nL3\nL4\nL5\n") - async with mock_reme_server() as (host, port): - await call_and_check( - "read", - host=host, - port=port, - path="Notes.md", - start_line=2, - end_line=4, - validator=lambda r: ( - isinstance(r, dict) - and r.get("success") is True - and "L2" in str(r["answer"]) - and "L3" in str(r["answer"]) - and "L4" in str(r["answer"]) - and "L1" not in str(r["answer"]) - and "L5" not in str(r["answer"]) - ), - ) - print("✓ test_read_line_range passed") - - _run(run()) - - -def test_read_absolute_path_accepted(): - """Absolute paths are accepted (a log warning is emitted but the read proceeds).""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - target = _seed_md(working, "Abs.md", "x\n") - async with mock_reme_server() as (host, port): - result = await call_action( - "read", - host=host, - port=port, - path=str(target.resolve()), - ) - if not ( - isinstance(result, dict) and result.get("success") is True and "x" in str(result.get("answer", "")) - ): - raise AssertionError(f"expected absolute-path read to succeed, got {result!r}") - print("✓ test_read_absolute_path_accepted passed") - - _run(run()) - - -def test_read_non_md_degraded(): - """Paths whose suffix is not `.md` are read in compatibility mode.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "data/foo.txt", "plain-text body\n") - async with mock_reme_server() as (host, port): - await call_and_check( - "read", - host=host, - port=port, - path="data/foo.txt", - validator=lambda r: ( - isinstance(r, dict) - and r.get("success") is True - and "plain-text body" in str(r.get("answer", "")) - ), - ) - print("✓ test_read_non_md_degraded passed") - - _run(run()) - - -def test_read_missing_file(): - """Reading a non-existent file should fail with a clear error.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - result = await call_action( - "read", - host=host, - port=port, - path="NotThere.md", - ) - if not ( - isinstance(result, dict) - and result.get("success") is False - and "does not exist" in str(result.get("answer", "")).lower() - ): - raise AssertionError(f"expected missing-file rejection, got {result!r}") - print("✓ test_read_missing_file passed") - - _run(run()) - - -def test_read_start_after_end(): - """start_line > end_line is invalid.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "Range.md", "a\nb\nc\n") - async with mock_reme_server() as (host, port): - result = await call_action( - "read", - host=host, - port=port, - path="Range.md", - start_line=3, - end_line=1, - ) - if not ( - isinstance(result, dict) - and result.get("success") is False - and "start_line" in str(result.get("answer", "")) - ): - raise AssertionError(f"expected start>end rejection, got {result!r}") - print("✓ test_read_start_after_end passed") - - _run(run()) - - -def test_read_start_line_exceeds_total(): - """start_line beyond total line count is invalid.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "Short.md", "only-one-line\n") - async with mock_reme_server() as (host, port): - result = await call_action( - "read", - host=host, - port=port, - path="Short.md", - start_line=99, - ) - if not ( - isinstance(result, dict) - and result.get("success") is False - and "exceeds" in str(result.get("answer", "")).lower() - ): - raise AssertionError(f"expected exceeds-length rejection, got {result!r}") - print("✓ test_read_start_line_exceeds_total passed") - - _run(run()) - - -def test_read_truncation(): - """A file larger than DEFAULT_MAX_BYTES triggers truncation with a continuation notice.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - # Seed > DEFAULT_MAX_BYTES (50 KiB) so the default truncation kicks in. - body = "\n".join(f"line {i}" for i in range(8000)) + "\n" - _seed_md(working, "Big.md", body) - async with mock_reme_server() as (host, port): - await call_and_check( - "read", - host=host, - port=port, - path="Big.md", - validator=lambda r: ( - isinstance(r, dict) - and r.get("success") is True - and "truncated" in str(r["answer"]) - and "start_line=" in str(r["answer"]) - ), - ) - print("✓ test_read_truncation passed") - - _run(run()) - - -def test_read_empty_path_rejected(): - """An empty `path` should be rejected.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - result = await call_action("read", host=host, port=port, path="") - if not ( - isinstance(result, dict) - and result.get("success") is False - and "required" in str(result.get("answer", "")).lower() - ): - raise AssertionError(f"expected `path` required rejection, got {result!r}") - print("✓ test_read_empty_path_rejected passed") - - _run(run()) - - -# --------------------------------------------------------------------------- -# write / edit / append tests -# --------------------------------------------------------------------------- - - -def test_write_basic_with_frontmatter(): - """`reme4 write path=... name=... description=... content=...` writes a YAML front matter block.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - await call_and_check( - "write", - host=host, - port=port, - path="Notes/A.md", - name="Greetings", - description="a friendly hello note", - content="# Hello", - validator=lambda r: ( - isinstance(r, dict) and r.get("success") is True and "Wrote" in str(r.get("answer", "")) - ), - ) - on_disk = (working / "Notes/A.md").read_text(encoding="utf-8") - assert on_disk.startswith("---\n"), on_disk - assert "name: Greetings" in on_disk - assert "description: a friendly hello note" in on_disk - assert "# Hello" in on_disk - print("✓ test_write_basic_with_frontmatter passed") - - _run(run()) - - -def test_write_no_suffix_autoappends_md(): - """`path` with no suffix gets `.md` appended.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - await call_and_check( - "write", - host=host, - port=port, - path="Notes/My", - content="x", - validator=lambda r: r.get("success") is True, - ) - assert (working / "Notes/My.md").exists() - print("✓ test_write_no_suffix_autoappends_md passed") - - _run(run()) - - -def test_write_overwrites_with_notice(): - """Writing into an existing path overwrites the file and surfaces a system notice.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "Existing.md", "old\n") - async with mock_reme_server() as (host, port): - await call_and_check( - "write", - host=host, - port=port, - path="Existing.md", - content="new", - validator=lambda r: ( - isinstance(r, dict) - and r.get("success") is True - and "Wrote" in str(r.get("answer", "")) - and "already existed" in str(r.get("answer", "")) - and "overwritten" in str(r.get("answer", "")) - ), - ) - # File body has been replaced. - on_disk = (working / "Existing.md").read_text(encoding="utf-8") - assert "new" in on_disk and "old" not in on_disk, on_disk - print("✓ test_write_overwrites_with_notice passed") - - _run(run()) - - -def test_write_creates_parent_dirs(): - """Nested-non-existent parents are auto-created.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - await call_and_check( - "write", - host=host, - port=port, - path="a/b/c/D.md", - content="hi", - validator=lambda r: r.get("success") is True, - ) - assert (working / "a/b/c/D.md").exists() - print("✓ test_write_creates_parent_dirs passed") - - _run(run()) - - -def test_write_no_frontmatter_when_all_empty(): - """When both `name` and `description` are empty strings, the file is body-only. - - The CLI schema declares them required, but the step is intentionally lenient - so manual calls without these fields don't fail catastrophically.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - await call_and_check( - "write", - host=host, - port=port, - path="Plain.md", - name="", - description="", - content="# Hello", - validator=lambda r: r.get("success") is True, - ) - on_disk = (working / "Plain.md").read_text(encoding="utf-8") - assert not on_disk.startswith("---"), on_disk - assert "# Hello" in on_disk - print("✓ test_write_no_frontmatter_when_all_empty passed") - - _run(run()) - - -def test_write_ignores_arbitrary_extra_fields(): - """Extra kwargs beyond name/description are silently ignored (schema is strict).""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - await call_and_check( - "write", - host=host, - port=port, - path="Custom.md", - name="My Note", - description="short summary", - content="body", - # Extras below should NOT appear in front matter under the - # hardcoded-fields schema. - title="ignored", - author="ignored", - tags='["x","y"]', - validator=lambda r: r.get("success") is True, - ) - on_disk = (working / "Custom.md").read_text(encoding="utf-8") - assert on_disk.startswith("---\n"), on_disk - assert "name: My Note" in on_disk - assert "description: short summary" in on_disk - assert "title:" not in on_disk - assert "author:" not in on_disk - assert "tags:" not in on_disk - assert "body" in on_disk - print("✓ test_write_ignores_arbitrary_extra_fields passed") - - _run(run()) - - -def test_write_only_description_present(): - """Step is lenient: providing only `description` works; missing `name` is skipped.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - await call_and_check( - "write", - host=host, - port=port, - path="OnlyDesc.md", - description="just a description", - content="body", - validator=lambda r: r.get("success") is True, - ) - on_disk = (working / "OnlyDesc.md").read_text(encoding="utf-8") - assert on_disk.startswith("---\n"), on_disk - assert "description: just a description" in on_disk - assert "name:" not in on_disk - print("✓ test_write_only_description_present passed") - - _run(run()) - - -def test_edit_global_replace(): - """`reme4 edit` replaces every occurrence of `old` with `new`.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "E.md", "foo bar foo\nfoo\n") - async with mock_reme_server() as (host, port): - await call_and_check( - "edit", - host=host, - port=port, - path="E.md", - old="foo", - new="qux", - validator=lambda r: ( - r.get("success") is True and "3" in str(r.get("answer", "")) # 3 replacements - ), - ) - assert (working / "E.md").read_text(encoding="utf-8") == "qux bar qux\nqux\n" - print("✓ test_edit_global_replace passed") - - _run(run()) - - -def test_edit_old_not_found(): - """`old` absent in the file → success=False.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "E.md", "hello world\n") - async with mock_reme_server() as (host, port): - result = await call_action( - "edit", - host=host, - port=port, - path="E.md", - old="absent", - new="x", - ) - if not ( - isinstance(result, dict) - and result.get("success") is False - and "not found" in str(result.get("answer", "")).lower() - ): - raise AssertionError(f"expected not-found rejection, got {result!r}") - # File unchanged. - assert (working / "E.md").read_text(encoding="utf-8") == "hello world\n" - print("✓ test_edit_old_not_found passed") - - _run(run()) - - -def test_edit_missing_file(): - """Editing a non-existent file should fail.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - result = await call_action( - "edit", - host=host, - port=port, - path="NotThere.md", - old="x", - new="y", - ) - if not ( - isinstance(result, dict) - and result.get("success") is False - and "does not exist" in str(result.get("answer", "")).lower() - ): - raise AssertionError(f"expected missing-file rejection, got {result!r}") - print("✓ test_edit_missing_file passed") - - _run(run()) - - -def test_edit_skips_frontmatter(): - """A match present in both front matter and body is replaced only in the body.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - body = ( - "---\n" - "name: alpha\n" - "description: alpha-doc\n" - "---\n" - "intro paragraph mentioning alpha and alpha again.\n" - ) - _seed_md(working, "WithFM.md", body) - async with mock_reme_server() as (host, port): - await call_and_check( - "edit", - host=host, - port=port, - path="WithFM.md", - old="alpha", - new="beta", - validator=lambda r: ( - r.get("success") is True and "2" in str(r.get("answer", "")) # 2 body occurrences only - ), - ) - on_disk = (working / "WithFM.md").read_text(encoding="utf-8") - # Front matter untouched. - assert "name: alpha" in on_disk, on_disk - assert "description: alpha-doc" in on_disk, on_disk - # Body fully rewritten. - assert "beta and beta" in on_disk, on_disk - assert "alpha and alpha" not in on_disk, on_disk - print("✓ test_edit_skips_frontmatter passed") - - _run(run()) - - -def test_edit_match_only_in_frontmatter_fails(): - """If `old` appears ONLY inside front matter, edit reports not-found and writes nothing.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - body = "---\nname: secret\ndescription: nope\n---\nplain body without the keyword.\n" - _seed_md(working, "FMOnly.md", body) - async with mock_reme_server() as (host, port): - result = await call_action( - "edit", - host=host, - port=port, - path="FMOnly.md", - old="secret", - new="leaked", - ) - if not ( - isinstance(result, dict) - and result.get("success") is False - and "not found" in str(result.get("answer", "")).lower() - ): - raise AssertionError(f"expected not-found rejection, got {result!r}") - # File untouched. - assert (working / "FMOnly.md").read_text(encoding="utf-8") == body - print("✓ test_edit_match_only_in_frontmatter_fails passed") - - _run(run()) - - -def test_append_basic(): - """Append adds content to the end of an existing file.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "A.md", "L1\n") - async with mock_reme_server() as (host, port): - await call_and_check( - "append", - host=host, - port=port, - path="A.md", - content="L2\n", - validator=lambda r: r.get("success") is True and "Appended" in r["answer"], - ) - assert (working / "A.md").read_text(encoding="utf-8") == "L1\nL2\n" - print("✓ test_append_basic passed") - - _run(run()) - - -def test_append_concatenates_verbatim(): - """Append concatenates content verbatim — no implicit newline insertion.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "A.md", "abc") # no trailing newline - async with mock_reme_server() as (host, port): - await call_and_check( - "append", - host=host, - port=port, - path="A.md", - content="def", - validator=lambda r: r.get("success") is True, - ) - assert (working / "A.md").read_text(encoding="utf-8") == "abcdef" - print("✓ test_append_concatenates_verbatim passed") - - _run(run()) - - -def test_append_auto_creates_missing_file(): - """Append on a non-existent path creates the file and surfaces a system notice.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - await call_and_check( - "append", - host=host, - port=port, - path="Fresh.md", - content="hello\n", - validator=lambda r: ( - isinstance(r, dict) - and r.get("success") is True - and "Appended" in str(r.get("answer", "")) - and "auto-created" in str(r.get("answer", "")) - ), - ) - assert (working / "Fresh.md").read_text(encoding="utf-8") == "hello\n" - print("✓ test_append_auto_creates_missing_file passed") - - _run(run()) - - -def test_append_empty_content_on_existing_file_is_noop(): - """Appending empty content to an existing file leaves it unchanged.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "A.md", "L1\n") - async with mock_reme_server() as (host, port): - await call_and_check( - "append", - host=host, - port=port, - path="A.md", - content="", - validator=lambda r: r.get("success") is True and "0 bytes" in str(r.get("answer", "")), - ) - assert (working / "A.md").read_text(encoding="utf-8") == "L1\n" - print("✓ test_append_empty_content_on_existing_file_is_noop passed") - - _run(run()) - - -def test_append_empty_content_creates_empty_file(): - """Appending empty content to a missing path creates an empty file (with notice).""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - await call_and_check( - "append", - host=host, - port=port, - path="Empty.md", - content="", - validator=lambda r: (r.get("success") is True and "auto-created" in str(r.get("answer", ""))), - ) - target = working / "Empty.md" - assert target.exists() and target.read_text(encoding="utf-8") == "" - print("✓ test_append_empty_content_creates_empty_file passed") - - _run(run()) - - -# --------------------------------------------------------------------------- -# Non-markdown degraded-mode tests -# --------------------------------------------------------------------------- - - -def test_write_non_md_skips_frontmatter(): - """Writing to a non-md path skips name/description and emits a recommendation notice.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - async with mock_reme_server() as (host, port): - await call_and_check( - "write", - host=host, - port=port, - path="data/notes.txt", - name="Greetings", - description="should be ignored", - content="# Hello", - validator=lambda r: ( - r.get("success") is True - and "Wrote" in str(r.get("answer", "")) - and "non-markdown" in str(r.get("answer", "")).lower() - ), - ) - on_disk = (working / "data/notes.txt").read_text(encoding="utf-8") - assert not on_disk.startswith("---"), on_disk - assert "name: Greetings" not in on_disk - assert on_disk == "# Hello" - print("✓ test_write_non_md_skips_frontmatter passed") - - _run(run()) - - -def test_edit_non_md_full_text(): - """Editing a non-md path operates on the full file body (no frontmatter parsing).""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - # A YAML-looking header that would otherwise be stripped as frontmatter. - body = "---\nname: keep-me\n---\nfoo bar foo\n" - _seed_md(working, "data/code.txt", body) - async with mock_reme_server() as (host, port): - await call_and_check( - "edit", - host=host, - port=port, - path="data/code.txt", - old="keep-me", - new="replaced", - validator=lambda r: ( - r.get("success") is True - and "1" in str(r.get("answer", "")) - and "non-markdown" in str(r.get("answer", "")).lower() - ), - ) - on_disk = (working / "data/code.txt").read_text(encoding="utf-8") - assert "name: replaced" in on_disk, on_disk - assert "foo bar foo" in on_disk, on_disk - print("✓ test_edit_non_md_full_text passed") - - _run(run()) - - -def test_append_non_md_warns(): - """Appending to a non-md file succeeds and surfaces the compatibility notice.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "data/log.txt", "line1\n") - async with mock_reme_server() as (host, port): - await call_and_check( - "append", - host=host, - port=port, - path="data/log.txt", - content="line2\n", - validator=lambda r: ( - r.get("success") is True and "non-markdown" in str(r.get("answer", "")).lower() - ), - ) - assert (working / "data/log.txt").read_text(encoding="utf-8") == "line1\nline2\n" - print("✓ test_append_non_md_warns passed") - - _run(run()) - - -def test_read_non_utf8_encoding(): - """A GBK-encoded legacy file (e.g. CN-Windows CSV) is decoded via the GBK fallback.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - target = working / "data.csv" - target.parent.mkdir(parents=True, exist_ok=True) - text = "姓名,职业\n你好世界,工程师\n" - target.write_bytes(text.encode("gbk")) - async with mock_reme_server() as (host, port): - await call_and_check( - "read", - host=host, - port=port, - path="data.csv", - validator=lambda r: (r.get("success") is True and "你好世界" in str(r.get("answer", ""))), - ) - print("✓ test_read_non_utf8_encoding passed") - - _run(run()) - - -def test_append_preserves_gbk_encoding(): - """Appending to a GBK file re-encodes new content in GBK (no UTF-8 corruption).""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - target = working / "data.csv" - target.parent.mkdir(parents=True, exist_ok=True) - existing = ("姓名,年龄\n张三,30\n" * 20).encode("gbk") - target.write_bytes(existing) - async with mock_reme_server() as (host, port): - await call_and_check( - "append", - host=host, - port=port, - path="data.csv", - content="李四,25\n", - validator=lambda r: r.get("success") is True, - ) - # File must round-trip as GBK; UTF-8 decoding would fail or yield mojibake. - raw = target.read_bytes() - assert raw.endswith("李四,25\n".encode("gbk")), raw[-20:] - decoded = raw.decode("gbk") - assert "张三" in decoded and "李四" in decoded - print("✓ test_append_preserves_gbk_encoding passed") - - _run(run()) - - -def test_edit_preserves_gbk_encoding(): - """Editing a GBK file keeps the file encoded in GBK after the rewrite.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - target = working / "notes.csv" - target.parent.mkdir(parents=True, exist_ok=True) - text = "原始内容,占位\n" * 20 - target.write_bytes(text.encode("gbk")) - async with mock_reme_server() as (host, port): - await call_and_check( - "edit", - host=host, - port=port, - path="notes.csv", - old="原始内容", - new="替换后", - validator=lambda r: r.get("success") is True, - ) - raw = target.read_bytes() - # File still decodes as GBK (would raise if we'd silently converted to UTF-8). - decoded = raw.decode("gbk") - assert "替换后" in decoded and "原始内容" not in decoded - print("✓ test_edit_preserves_gbk_encoding passed") - - _run(run()) - - -def test_read_utf8_bom(): - """Reading a UTF-8 file with BOM strips the BOM transparently.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - target = working / "bom.txt" - target.parent.mkdir(parents=True, exist_ok=True) - target.write_bytes(b"\xef\xbb\xbfhello world\n") - async with mock_reme_server() as (host, port): - await call_and_check( - "read", - host=host, - port=port, - path="bom.txt", - validator=lambda r: ( - r.get("success") is True - and "hello world" in str(r.get("answer", "")) - and "" not in str(r.get("answer", "")) - ), - ) - print("✓ test_read_utf8_bom passed") - - _run(run()) - - -# --------------------------------------------------------------------------- -# Aggregate test: reuse one server instance for all read cases (faster). -# --------------------------------------------------------------------------- - - -def test_all_read_cases_one_server(): - """Run multiple read scenarios against a single shared server for efficiency.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, _temp_chdir(tmp): - working = Path(tmp) / ".reme" - working.mkdir(parents=True, exist_ok=True) - _seed_md(working, "Templates/Recipe.md", "# Recipe\nbody\n") - _seed_md(working, "Notes.md", "L1\nL2\nL3\n") - async with mock_reme_server() as (host, port): - await call_and_check( - "read", - host=host, - port=port, - path="Templates/Recipe.md", - validator=lambda r: r.get("success") is True and "# Recipe" in r["answer"], - ) - await call_and_check( - "read", - host=host, - port=port, - path="Notes", - validator=lambda r: r.get("success") is True and "L1" in r["answer"], - ) - await call_and_check( - "read", - host=host, - port=port, - path="Notes.md", - start_line=2, - end_line=2, - validator=lambda r: r.get("success") is True and r["answer"].strip() == "L2", - ) - print("✓ test_all_read_cases_one_server passed") - - _run(run()) - - -if __name__ == "__main__": - print("\n=== reme4 crud_md (read) E2E tests ===") - test_read_relative_path() - test_read_no_suffix_autoappends_md() - test_read_line_range() - test_read_non_md_degraded() - test_read_missing_file() - test_read_start_after_end() - test_read_start_line_exceeds_total() - test_read_truncation() - test_read_empty_path_rejected() - test_all_read_cases_one_server() - print("\n=== reme4 crud_md (write/edit/append) E2E tests ===") - test_write_basic_with_frontmatter() - test_write_no_suffix_autoappends_md() - test_write_overwrites_with_notice() - test_write_creates_parent_dirs() - test_write_no_frontmatter_when_all_empty() - test_write_ignores_arbitrary_extra_fields() - test_write_only_description_present() - test_edit_global_replace() - test_edit_old_not_found() - test_edit_missing_file() - test_edit_skips_frontmatter() - test_edit_match_only_in_frontmatter_fails() - test_append_basic() - test_append_concatenates_verbatim() - test_append_auto_creates_missing_file() - test_append_empty_content_on_existing_file_is_noop() - test_append_empty_content_creates_empty_file() - print("\n=== reme4 crud_md (non-md degraded mode) E2E tests ===") - test_write_non_md_skips_frontmatter() - test_edit_non_md_full_text() - test_append_non_md_warns() - test_read_non_utf8_encoding() - test_append_preserves_gbk_encoding() - test_edit_preserves_gbk_encoding() - test_read_utf8_bom() - print("\n所有测试通过!") diff --git a/tests4/unittest/test_file_graph.py b/tests4/unittest/test_file_graph.py index 1dc3f6f5..0ac00777 100644 --- a/tests4/unittest/test_file_graph.py +++ b/tests4/unittest/test_file_graph.py @@ -9,6 +9,7 @@ import tempfile import pytest from reme4.components.file_graph import LocalFileGraph, NxFileGraph +from reme4.enumeration import LinkScopeEnum from reme4.schema import FileLink, FileNode @@ -102,8 +103,14 @@ def test_upsert_replaces_old_links(backend_cls): @pytest.mark.parametrize("backend_cls", BACKENDS) -def test_outlinks_skip_virtual_targets(backend_cls): - """get_outlinks only returns links pointing to real (existing) nodes.""" +def test_outlinks_include_virtual_targets(backend_cls): + """get_outlinks(scope=ALL) surfaces edges into virtual targets too; + scope=VIRTUAL isolates them; default scope=REAL hides them. + + A wikilink to a not-yet-indexed file is still real data the source + contains — callers like the day-index aggregator and lint:dangling + need to see it via the opt-in scopes. + """ async def run(): with tempfile.TemporaryDirectory() as tmpdir, temp_chdir(tmpdir): @@ -118,43 +125,68 @@ def test_outlinks_skip_virtual_targets(backend_cls): ], ) - outs = await graph.get_outlinks("a.md") - targets = {lnk.target_path for lnk in outs} - assert targets == {"b.md"} + all_targets = {lnk.target_path for lnk in await graph.get_outlinks("a.md", scope=LinkScopeEnum.ALL)} + assert all_targets == {"b.md", "ghost.md"} + + virtual_targets = {lnk.target_path for lnk in await graph.get_outlinks("a.md", scope=LinkScopeEnum.VIRTUAL)} + assert virtual_targets == {"ghost.md"} + + # Default scope=REAL hides the dangling edge. + real_targets = {lnk.target_path for lnk in await graph.get_outlinks("a.md")} + assert real_targets == {"b.md"} await graph.close() - print(f"✓ test_outlinks_skip_virtual_targets[{backend_cls.__name__}] passed") + print(f"✓ test_outlinks_include_virtual_targets[{backend_cls.__name__}] passed") asyncio.run(run()) @pytest.mark.parametrize("backend_cls", BACKENDS) -def test_inlinks_promotion_after_upsert(backend_cls): - """Edges to virtual targets become real inlinks once the target is upserted.""" +def test_inlinks_visible_for_virtual_target(backend_cls): + """Edges to a virtual target are queryable via get_inlinks(scope=VIRTUAL/ALL). + + The data model stores ``target → {sources}`` regardless of whether the + target is a real node or a placeholder; scope=REAL hides virtual-target + inlinks, VIRTUAL/ALL surface them so callers like graph_retarget_step + can fix dangling references. + """ async def run(): with tempfile.TemporaryDirectory() as tmpdir, temp_chdir(tmpdir): graph = backend_cls() await graph.start() - # b doesn't exist yet — link is pending + # b doesn't exist yet — edge lives against a virtual placeholder. await graph.upsert_nodes([make_node("a.md", [("b.md", None)])]) - assert await graph.get_inlinks("b.md") == [] # b not real yet - - # Now create b — pending edge promotes - await graph.upsert_nodes([make_node("b.md")]) - inlinks = await graph.get_inlinks("b.md") + # Default (real only): b is virtual, so nothing. + assert await graph.get_inlinks("b.md") == [] + # Opt-in virtual scope: surface the dangling inlink. + inlinks = await graph.get_inlinks("b.md", scope=LinkScopeEnum.VIRTUAL) assert {lnk.source_path for lnk in inlinks} == {"a.md"} + # ALL: equivalent here since b is purely virtual. + inlinks_all = await graph.get_inlinks("b.md", scope=LinkScopeEnum.ALL) + assert {lnk.source_path for lnk in inlinks_all} == {"a.md"} + + # Promoting b to real makes the inlink visible by default again, + # and the virtual scope now returns nothing. + await graph.upsert_nodes([make_node("b.md")]) + assert {lnk.source_path for lnk in await graph.get_inlinks("b.md")} == {"a.md"} + assert await graph.get_inlinks("b.md", scope=LinkScopeEnum.VIRTUAL) == [] await graph.close() - print(f"✓ test_inlinks_promotion_after_upsert[{backend_cls.__name__}] passed") + print(f"✓ test_inlinks_visible_for_virtual_target[{backend_cls.__name__}] passed") asyncio.run(run()) @pytest.mark.parametrize("backend_cls", BACKENDS) -def test_delete_node_demotes_inbound(backend_cls): - """Deleting a node makes it virtual; sources still hold the link, but get_inlinks([deleted]) is [].""" +def test_delete_keeps_inbound_view(backend_cls): + """Deleting a node demotes it to virtual; with ``scope=VIRTUAL`` (or + ``ALL``) the inbound view is preserved (sources still hold the link), + and outlinks into the now-virtual target are surfaced — the link + payload is what matters, virtuality is just an indexing artifact. + Default scope (REAL) hides both views once the target is virtual. + """ async def run(): with tempfile.TemporaryDirectory() as tmpdir, temp_chdir(tmpdir): @@ -172,17 +204,22 @@ def test_delete_node_demotes_inbound(backend_cls): await graph.delete_nodes(["b.md"]) # b is no longer a real node assert await graph.get_nodes(["b.md"]) == [] - # inlinks query for a non-real node returns [] + # Default scope=REAL: virtual b has no visible inlinks. assert await graph.get_inlinks("b.md") == [] - # a's outlink to b is hidden because b is virtual + # scope=VIRTUAL surfaces the preserved dangling reference. + virtual_inlinks = await graph.get_inlinks("b.md", scope=LinkScopeEnum.VIRTUAL) + assert {lnk.source_path for lnk in virtual_inlinks} == {"a.md"} + # Default also hides a's outlink into virtual b; scope=ALL surfaces it. assert await graph.get_outlinks("a.md") == [] + all_outlinks = await graph.get_outlinks("a.md", scope=LinkScopeEnum.ALL) + assert {lnk.target_path for lnk in all_outlinks} == {"b.md"} - # Re-upsert b — pending should re-promote + # Re-upsert b — inlinks visible by default again. await graph.upsert_nodes([make_node("b.md")]) assert {lnk.source_path for lnk in await graph.get_inlinks("b.md")} == {"a.md"} await graph.close() - print(f"✓ test_delete_node_demotes_inbound[{backend_cls.__name__}] passed") + print(f"✓ test_delete_keeps_inbound_view[{backend_cls.__name__}] passed") asyncio.run(run()) @@ -322,9 +359,9 @@ if __name__ == "__main__": for backend in BACKENDS: test_upsert_and_get_nodes(backend) test_upsert_replaces_old_links(backend) - test_outlinks_skip_virtual_targets(backend) - test_inlinks_promotion_after_upsert(backend) - test_delete_node_demotes_inbound(backend) + test_outlinks_include_virtual_targets(backend) + test_inlinks_visible_for_virtual_target(backend) + test_delete_keeps_inbound_view(backend) test_delete_outgoing_links_cleared(backend) test_clear(backend) test_rebuild_links_idempotent(backend) diff --git a/tests4/unittest/test_linked_file_parser.py b/tests4/unittest/test_linked_file_parser.py index 034d19ff..ce82427c 100644 --- a/tests4/unittest/test_linked_file_parser.py +++ b/tests4/unittest/test_linked_file_parser.py @@ -1,4 +1,10 @@ -"""Tests for LinkedFileParser (markdown parser + wikilink extraction).""" +"""Tests for LinkedFileParser (markdown parser + wikilink extraction). + +Wikilink convention here is strict: targets are taken literally, no +short-form basename search, no implicit ``.md``, no folder-note +expansion. ``lint:dangling`` handles validation; the parser is just +a markdown-to-FileNode transformer. +""" # pylint: disable=protected-access @@ -6,9 +12,7 @@ import asyncio import os import tempfile -from reme4.components.file_graph import LocalFileGraph from reme4.components.file_parser import LinkedFileParser -from reme4.schema import FileNode class temp_chdir: @@ -28,22 +32,12 @@ class temp_chdir: def _write_md(tmpdir: str, name: str, body: str) -> str: - """Drop a markdown file under tmpdir, return its path.""" - path = os.path.join(tmpdir, name) + """Drop a markdown file under tmpdir, return its relative path (matches cwd).""" if "/" in name: - os.makedirs(os.path.dirname(path), exist_ok=True) - with open(path, "w", encoding="utf-8") as f: + os.makedirs(os.path.join(tmpdir, os.path.dirname(name)), exist_ok=True) + with open(os.path.join(tmpdir, name), "w", encoding="utf-8") as f: f.write(body) - return path - - -async def _make_graph(*nodes: FileNode) -> LocalFileGraph: - """Build a started LocalFileGraph seeded with the given nodes.""" - graph = LocalFileGraph() - await graph.start() - if nodes: - await graph.upsert_nodes(list(nodes)) - return graph + return name def test_parse_empty_file(): @@ -51,28 +45,28 @@ def test_parse_empty_file(): async def run(): with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - path = _write_md(tmp, "empty.md", "") + path = _write_md(tmp, "x.md", "") parser = LinkedFileParser() node, chunks = await parser.parse(path) + assert node.path == "x.md" assert chunks == [] assert node.links == [] - assert node.chunk_ids == [] print("✓ test_parse_empty_file passed") asyncio.run(run()) def test_parse_frontmatter_only(): - """Front-matter without body → FileNode with metadata, no chunks/links.""" + """A file with only frontmatter (no body) → no chunks, no links.""" async def run(): with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - path = _write_md(tmp, "fm.md", "---\ntitle: Demo\ntags: [a, b]\n---\n") + path = _write_md(tmp, "fm.md", "---\nname: t\n---\n") parser = LinkedFileParser() node, chunks = await parser.parse(path) + assert node.front_matter.name == "t" assert chunks == [] - assert node.front_matter.title == "Demo" - assert list(node.front_matter.tags or []) == ["a", "b"] + assert node.links == [] print("✓ test_parse_frontmatter_only passed") asyncio.run(run()) @@ -83,27 +77,27 @@ def test_parse_small_body_one_chunk(): async def run(): with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - path = _write_md(tmp, "small.md", "# Hello\n\nworld") - parser = LinkedFileParser(chunk_chars=2000) - _, chunks = await parser.parse(path) + body = "# Hello\n\nthis is a small body." + path = _write_md(tmp, "small.md", body) + parser = LinkedFileParser(chunk_chars=500) + node, chunks = await parser.parse(path) assert len(chunks) == 1 - assert "world" in chunks[0].text - assert chunks[0].start_line >= 1 - assert chunks[0].end_line >= chunks[0].start_line + assert "this is a small body" in chunks[0].text + assert node.chunk_ids == [chunks[0].id] print("✓ test_parse_small_body_one_chunk passed") asyncio.run(run()) def test_parse_oversized_body_splits(): - """A body that exceeds chunk_chars produces multiple chunks.""" + """A body exceeding chunk_chars triggers multiple chunks.""" async def run(): with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - # 30 paragraphs of 100 chars each ≈ 3000 chars body - body = "# Big\n\n" + "\n\n".join("p" * 100 for _ in range(30)) + paras = "\n\n".join(f"paragraph {i} with some content text here." for i in range(50)) + body = "# H\n\n" + paras path = _write_md(tmp, "big.md", body) - parser = LinkedFileParser(chunk_chars=500, embed_toc=False) + parser = LinkedFileParser(chunk_chars=200) _, chunks = await parser.parse(path) assert len(chunks) > 1 print("✓ test_parse_oversized_body_splits passed") @@ -112,12 +106,14 @@ def test_parse_oversized_body_splits(): def test_parse_chunk_ids_match_node_chunk_ids(): - """FileNode.chunk_ids should match the ids of the chunks returned.""" + """node.chunk_ids is the ordered list of chunk hashes.""" async def run(): with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - path = _write_md(tmp, "ids.md", "# X\n\nbody") - parser = LinkedFileParser() + paras = "\n\n".join(f"para {i} body content here." for i in range(40)) + body = "# H\n\n" + paras + path = _write_md(tmp, "p.md", body) + parser = LinkedFileParser(chunk_chars=200) node, chunks = await parser.parse(path) assert node.chunk_ids == [c.id for c in chunks] print("✓ test_parse_chunk_ids_match_node_chunk_ids passed") @@ -125,41 +121,44 @@ def test_parse_chunk_ids_match_node_chunk_ids(): asyncio.run(run()) -def test_parse_links_empty_when_no_graph(): - """Without an app_context / graph, links stay empty even if the body has wikilinks.""" +def test_parse_links_literal_targets(): + """Wikilink targets are taken verbatim — full path → FileLink.target_path.""" async def run(): with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - path = _write_md(tmp, "x.md", "see [[Alice]] and [[Bob]]") + body = "see [[topics/Alice.md]] and [[topics/Bob.md#sec]]" + path = _write_md(tmp, "note.md", body) parser = LinkedFileParser() node, _ = await parser.parse(path) - assert node.links == [] - print("✓ test_parse_links_empty_when_no_graph passed") - - asyncio.run(run()) - - -def test_parse_links_resolved_via_graph(): - """With a graph that knows the targets, wikilinks become FileLinks with resolved target_path.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - graph = await _make_graph( - FileNode(path="topics/Alice.md", st_mtime=0.0), - FileNode(path="topics/Bob.md", st_mtime=0.0), - ) - path = _write_md(tmp, "note.md", "see [[Alice]] and [[Bob#sec]]") - parser = LinkedFileParser() - parser._resolve_file_graph = lambda: graph - node, _ = await parser.parse(path) triples = {(link.target_path, link.target_anchor, link.predicate) for link in node.links} assert ("topics/Alice.md", None, None) in triples assert ("topics/Bob.md", "sec", None) in triples # source_path always equals the node's own path for link in node.links: assert link.source_path == node.path - await graph.close() - print("✓ test_parse_links_resolved_via_graph passed") + print("✓ test_parse_links_literal_targets passed") + + asyncio.run(run()) + + +def test_parse_links_short_and_no_ext_kept_literally(): + """Short and no-ext forms are NOT resolved — they're stored as-is. + + The parser does no resolution; whether the target exists is a + ``lint:dangling`` concern. ``[[Alice]]`` becomes + ``target_path='Alice'`` and will be flagged dangling unless a node + with literal path 'Alice' actually exists. + """ + + async def run(): + with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): + body = "see [[Alice]] and [[topics/Alice]] but also [[topics/Alice.md]]" + path = _write_md(tmp, "note.md", body) + parser = LinkedFileParser() + node, _ = await parser.parse(path) + targets = {link.target_path for link in node.links} + assert targets == {"Alice", "topics/Alice", "topics/Alice.md"} + print("✓ test_parse_links_short_and_no_ext_kept_literally passed") asyncio.run(run()) @@ -169,75 +168,28 @@ def test_parse_links_predicate_inline_and_line(): async def run(): with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - graph = await _make_graph( - FileNode(path="A.md", st_mtime=0.0), - FileNode(path="B.md", st_mtime=0.0), - ) - body = "extends:: [[A]]\n\nsome [concerns:: [[B]]] inline\n" + body = "extends:: [[A.md]]\n\nsome [concerns:: [[B.md]]] inline\n" path = _write_md(tmp, "note.md", body) parser = LinkedFileParser() - parser._resolve_file_graph = lambda: graph node, _ = await parser.parse(path) pairs = {(link.target_path, link.predicate) for link in node.links} assert ("A.md", "extends") in pairs assert ("B.md", "concerns") in pairs - await graph.close() print("✓ test_parse_links_predicate_inline_and_line passed") asyncio.run(run()) -def test_parse_links_short_path_ambiguity_expands(): - """A short link matching multiple nodes expands to one FileLink per candidate.""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - graph = await _make_graph( - FileNode(path="topics/Bob.md", st_mtime=0.0), - FileNode(path="people/Bob.md", st_mtime=0.0), - ) - path = _write_md(tmp, "note.md", "ref [[Bob]]") - parser = LinkedFileParser() - parser._resolve_file_graph = lambda: graph - node, _ = await parser.parse(path) - targets = sorted(link.target_path for link in node.links) - assert targets == ["people/Bob.md", "topics/Bob.md"] - await graph.close() - print("✓ test_parse_links_short_path_ambiguity_expands passed") - - asyncio.run(run()) - - -def test_parse_links_dangling_dropped(): - """Wikilink to a non-existent target is silently dropped (no graph node).""" - - async def run(): - with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - graph = await _make_graph(FileNode(path="topics/Alice.md", st_mtime=0.0)) - path = _write_md(tmp, "note.md", "[[Alice]] and [[Ghost]]") - parser = LinkedFileParser() - parser._resolve_file_graph = lambda: graph - node, _ = await parser.parse(path) - targets = {link.target_path for link in node.links} - assert targets == {"topics/Alice.md"} - await graph.close() - print("✓ test_parse_links_dangling_dropped passed") - - asyncio.run(run()) - - def test_parse_links_deduped(): """Repeated wikilinks with the same (target, predicate, anchor) emit one FileLink.""" async def run(): with tempfile.TemporaryDirectory() as tmp, temp_chdir(tmp): - graph = await _make_graph(FileNode(path="A.md", st_mtime=0.0)) - path = _write_md(tmp, "note.md", "[[A]] again [[A]] and [[A]]") + body = "[[A.md]] again [[A.md]] and [[A.md]]" + path = _write_md(tmp, "note.md", body) parser = LinkedFileParser() - parser._resolve_file_graph = lambda: graph node, _ = await parser.parse(path) assert len([link for link in node.links if link.target_path == "A.md"]) == 1 - await graph.close() print("✓ test_parse_links_deduped passed") asyncio.run(run()) @@ -273,11 +225,9 @@ if __name__ == "__main__": test_parse_small_body_one_chunk() test_parse_oversized_body_splits() test_parse_chunk_ids_match_node_chunk_ids() - test_parse_links_empty_when_no_graph() - test_parse_links_resolved_via_graph() + test_parse_links_literal_targets() + test_parse_links_short_and_no_ext_kept_literally() test_parse_links_predicate_inline_and_line() - test_parse_links_short_path_ambiguity_expands() - test_parse_links_dangling_dropped() test_parse_links_deduped() test_parse_min_chunk_chars_clamped() test_parse_embed_toc_prefixes_chunk_text() diff --git a/tests4/unittest/test_neo4j_file_graph.py b/tests4/unittest/test_neo4j_file_graph.py index fef9ef4a..dc212126 100644 --- a/tests4/unittest/test_neo4j_file_graph.py +++ b/tests4/unittest/test_neo4j_file_graph.py @@ -280,9 +280,8 @@ def test_node_roundtrip_preserves_frontmatter_and_links(): ], chunk_ids=["chunk-a1", "chunk-a2", "chunk-a3"], front_matter=FileFrontMatter( - title="Alice", + name="Alice", description="a person", - tags=["x", "y"], ), ) await graph.upsert_nodes([node, make_node("topics/Bob.md")]) @@ -291,9 +290,8 @@ def test_node_roundtrip_preserves_frontmatter_and_links(): back = got[0] assert back.path == "topics/Alice.md" assert back.st_mtime == 1234.5 - assert back.front_matter.title == "Alice" + assert back.front_matter.name == "Alice" assert back.front_matter.description == "a person" - assert sorted(back.front_matter.tags or []) == ["x", "y"] assert back.chunk_ids == ["chunk-a1", "chunk-a2", "chunk-a3"] assert len(back.links) == 1 link = back.links[0]