mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-09-30 01:51:20 +00:00
feat(markdown): support Obsidian-style [[wikilink]] syntax
Adds WIKILINK_RE alongside existing LINK_RE to parse [[name]], [[name.md]], [[folder/name]], [[name#heading]], [[name|alias]] forms. Skips ![[image]] embeds and wikilink-shaped text inside fenced or inline code blocks. Resolution tries sibling path first, then repo root, with .md/.mdx auto-suffix when extension is omitted. Shares seenLinks with markdown-link extraction so the same source/ target pair never produces two IMPORTS edges when both syntaxes coexist. Use case: Basic Memory MCP and Obsidian-style notes use [[name]] relations, which previously did not produce graph edges in GitNexus.
This commit is contained in:
parent
99b8c7b03b
commit
79bbc9c76f
2 changed files with 405 additions and 0 deletions
|
|
@ -13,6 +13,22 @@ import { KnowledgeGraph } from '../graph/types.js';
|
|||
|
||||
const HEADING_RE = /^(#{1,6})\s+(.+)$/;
|
||||
const LINK_RE = /\[([^\]]*)\]\(([^)]+)\)/g;
|
||||
// Obsidian / Basic Memory style wikilink:
|
||||
// [[id]] — bare target
|
||||
// [[id.md]] — explicit extension
|
||||
// [[folder/id]] — relative path
|
||||
// [[id#heading]] — heading anchor (stripped before path resolution)
|
||||
// [[id|alias]] — alias (alias text ignored for path resolution)
|
||||
//
|
||||
// The leading `(?<!!)` excludes image-style embeds `![[image.png]]`.
|
||||
// The target is captured up to the first `#`, `|`, or `]`, so trailing
|
||||
// fragments and aliases never bleed into the resolved path.
|
||||
const WIKILINK_RE = /(?<!!)\[\[([^\]|#\r\n]+)(?:#[^\]|\r\n]*)?(?:\|[^\]\r\n]*)?\]\]/g;
|
||||
// Strip fenced code blocks (``` ... ``` and ~~~ ... ~~~) and inline code spans
|
||||
// before scanning for wikilinks so that snippets like `[[fake]]` inside a
|
||||
// code block don't produce spurious IMPORTS edges.
|
||||
const FENCED_CODE_RE = /(^|\n)([ \t]*)(```|~~~)[\s\S]*?\n\2\3[ \t]*(?=\n|$)/g;
|
||||
const INLINE_CODE_RE = /`[^`\r\n]*`/g;
|
||||
const MD_EXTENSIONS = new Set(['.md', '.mdx']);
|
||||
|
||||
interface MdFile {
|
||||
|
|
@ -160,6 +176,74 @@ export const processMarkdown = (
|
|||
totalLinks++;
|
||||
}
|
||||
}
|
||||
|
||||
// --- Extract Obsidian / Basic Memory wikilinks ---
|
||||
// Strip code (fenced + inline) before scanning so `[[x]]` inside code
|
||||
// doesn't produce edges. Replace with same-length whitespace-equivalent
|
||||
// content (newlines preserved) is unnecessary because we no longer use
|
||||
// offsets here; we just need a sanitized scan target.
|
||||
const sanitized = file.content
|
||||
.replace(FENCED_CODE_RE, (m) => m.replace(/[^\n]/g, ' '))
|
||||
.replace(INLINE_CODE_RE, (m) => ' '.repeat(m.length));
|
||||
|
||||
let wm: RegExpExecArray | null;
|
||||
WIKILINK_RE.lastIndex = 0;
|
||||
while ((wm = WIKILINK_RE.exec(sanitized)) !== null) {
|
||||
const rawTarget = wm[1].trim();
|
||||
if (!rawTarget) continue;
|
||||
|
||||
// Defense in depth: the regex already drops `#heading` and `|alias`,
|
||||
// but normalize again in case the target itself contained one.
|
||||
const cleanTarget = rawTarget.split('#')[0].split('|')[0].trim();
|
||||
if (!cleanTarget) continue;
|
||||
|
||||
const hasMdExt = cleanTarget.endsWith('.md') || cleanTarget.endsWith('.mdx');
|
||||
|
||||
// Resolution order: original path, sibling .md/.mdx, then repo-root .md/.mdx.
|
||||
// path.posix.normalize collapses any `..`/redundant separators.
|
||||
const candidates: string[] = [];
|
||||
const push = (p: string) => {
|
||||
if (!p) return;
|
||||
const norm = path.posix.normalize(p);
|
||||
if (!candidates.includes(norm)) candidates.push(norm);
|
||||
};
|
||||
|
||||
// Sibling-first (relative to current file's directory)
|
||||
push(path.posix.join(fileDir, cleanTarget));
|
||||
if (!hasMdExt) {
|
||||
push(path.posix.join(fileDir, `${cleanTarget}.md`));
|
||||
push(path.posix.join(fileDir, `${cleanTarget}.mdx`));
|
||||
}
|
||||
// Repo-root fallback (treats target as a path from repo root)
|
||||
push(cleanTarget);
|
||||
if (!hasMdExt) {
|
||||
push(`${cleanTarget}.md`);
|
||||
push(`${cleanTarget}.mdx`);
|
||||
}
|
||||
|
||||
const resolvedWiki = candidates.find((c) => allPathSet.has(c));
|
||||
if (!resolvedWiki) continue;
|
||||
|
||||
const targetFileId = generateId('File', resolvedWiki);
|
||||
if (!graph.getNode(targetFileId)) continue;
|
||||
|
||||
// Don't create self-loops
|
||||
if (targetFileId === fileNodeId) continue;
|
||||
|
||||
const linkKey = `${fileNodeId}->${targetFileId}`;
|
||||
if (seenLinks.has(linkKey)) continue;
|
||||
seenLinks.add(linkKey);
|
||||
|
||||
graph.addRelationship({
|
||||
id: generateId('IMPORTS', linkKey),
|
||||
type: 'IMPORTS',
|
||||
sourceId: fileNodeId,
|
||||
targetId: targetFileId,
|
||||
confidence: 0.8,
|
||||
reason: 'markdown-wikilink',
|
||||
});
|
||||
totalLinks++;
|
||||
}
|
||||
}
|
||||
|
||||
return { sections: totalSections, links: totalLinks };
|
||||
|
|
|
|||
321
gitnexus/test/integration/markdown-processor-wikilink.test.ts
Normal file
321
gitnexus/test/integration/markdown-processor-wikilink.test.ts
Normal file
|
|
@ -0,0 +1,321 @@
|
|||
/**
|
||||
* Tests for Obsidian / Basic Memory style [[wikilink]] support in the
|
||||
* markdown processor. Wikilinks coexist with the existing [text](path.md)
|
||||
* markdown link syntax and produce the same kind of IMPORTS edge.
|
||||
*
|
||||
* Resolution rules (see markdown-processor.ts WIKILINK_RE block):
|
||||
* - Strip `#heading` and `|alias` before path resolution
|
||||
* - Try sibling path first (relative to source file's directory),
|
||||
* then repo-root, with auto-suffix `.md` and `.mdx` if no extension
|
||||
* - Skip image embeds `![[...]]`
|
||||
* - Skip wikilinks inside fenced or inline code
|
||||
* - Share dedup with markdown-link extraction so the same source/target
|
||||
* pair never produces two IMPORTS edges
|
||||
*/
|
||||
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { processMarkdown } from '../../src/core/ingestion/markdown-processor.js';
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import { generateId } from '../../src/lib/utils.js';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
import type { KnowledgeGraph } from '../../src/core/graph/types.js';
|
||||
|
||||
function setupGraphWithFiles(filePaths: string[]): KnowledgeGraph {
|
||||
const graph = createKnowledgeGraph();
|
||||
for (const filePath of filePaths) {
|
||||
const fileNode: GraphNode = {
|
||||
id: generateId('File', filePath),
|
||||
label: 'File',
|
||||
properties: { name: filePath, filePath },
|
||||
};
|
||||
graph.addNode(fileNode);
|
||||
}
|
||||
return graph;
|
||||
}
|
||||
|
||||
function getImports(graph: KnowledgeGraph): GraphRelationship[] {
|
||||
return [...graph.iterRelationshipsByType('IMPORTS')];
|
||||
}
|
||||
|
||||
function hasImports(graph: KnowledgeGraph, fromPath: string, toPath: string): boolean {
|
||||
const fromId = generateId('File', fromPath);
|
||||
const toId = generateId('File', toPath);
|
||||
return getImports(graph).some((r) => r.sourceId === fromId && r.targetId === toId);
|
||||
}
|
||||
|
||||
describe('markdown-processor markdown-link regression', () => {
|
||||
it('still resolves standard [text](path.md) markdown links', () => {
|
||||
const fromPath = 'a.md';
|
||||
const toPath = 'b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, toPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [B](b.md) for details.' }],
|
||||
new Set([fromPath, toPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(1);
|
||||
expect(hasImports(graph, fromPath, toPath)).toBe(true);
|
||||
const edges = getImports(graph);
|
||||
expect(edges).toHaveLength(1);
|
||||
expect(edges[0].reason).toBe('markdown-link');
|
||||
});
|
||||
});
|
||||
|
||||
describe('markdown-processor wikilinks', () => {
|
||||
it('resolves [[id]] without extension', () => {
|
||||
const fromPath = 'a.md';
|
||||
const toPath = 'b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, toPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [[b]] for details.' }],
|
||||
new Set([fromPath, toPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(1);
|
||||
expect(hasImports(graph, fromPath, toPath)).toBe(true);
|
||||
expect(getImports(graph)[0].reason).toBe('markdown-wikilink');
|
||||
});
|
||||
|
||||
it('resolves [[id.md]] with explicit extension', () => {
|
||||
const fromPath = 'a.md';
|
||||
const toPath = 'b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, toPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [[b.md]].' }],
|
||||
new Set([fromPath, toPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(1);
|
||||
expect(hasImports(graph, fromPath, toPath)).toBe(true);
|
||||
});
|
||||
|
||||
it('resolves [[folder/id]] folder-relative path', () => {
|
||||
const fromPath = 'notes/a.md';
|
||||
const toPath = 'notes/sub/b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, toPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [[sub/b]].' }],
|
||||
new Set([fromPath, toPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(1);
|
||||
expect(hasImports(graph, fromPath, toPath)).toBe(true);
|
||||
});
|
||||
|
||||
it('resolves [[id|alias]] (alias does not affect path)', () => {
|
||||
const fromPath = 'a.md';
|
||||
const toPath = 'b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, toPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [[b|the B note]].' }],
|
||||
new Set([fromPath, toPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(1);
|
||||
expect(hasImports(graph, fromPath, toPath)).toBe(true);
|
||||
});
|
||||
|
||||
it('resolves [[id#heading]] (fragment does not affect path)', () => {
|
||||
const fromPath = 'a.md';
|
||||
const toPath = 'b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, toPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [[b#some-heading]].' }],
|
||||
new Set([fromPath, toPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(1);
|
||||
expect(hasImports(graph, fromPath, toPath)).toBe(true);
|
||||
});
|
||||
|
||||
it('resolves [[id#heading|alias]] (both stripped)', () => {
|
||||
const fromPath = 'a.md';
|
||||
const toPath = 'b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, toPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [[b#section|nice name]].' }],
|
||||
new Set([fromPath, toPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(1);
|
||||
expect(hasImports(graph, fromPath, toPath)).toBe(true);
|
||||
});
|
||||
|
||||
it('does not link [[nonexistent]] when no matching file exists', () => {
|
||||
const fromPath = 'a.md';
|
||||
const graph = setupGraphWithFiles([fromPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [[nonexistent]].' }],
|
||||
new Set([fromPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(0);
|
||||
expect(getImports(graph)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('skips ![[image.png]] image embeds', () => {
|
||||
const fromPath = 'a.md';
|
||||
const imagePath = 'image.png';
|
||||
// image.png is a real file in the repo but it's not markdown.
|
||||
// Even if there were a literal markdown twin, image embed should not link.
|
||||
const graph = setupGraphWithFiles([fromPath, imagePath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See ![[image.png]] embedded.' }],
|
||||
new Set([fromPath, imagePath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(0);
|
||||
expect(getImports(graph)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('does not link wikilink-shaped text inside fenced code blocks', () => {
|
||||
const fromPath = 'a.md';
|
||||
const toPath = 'b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, toPath]);
|
||||
|
||||
const content =
|
||||
'Real link: [[b]]\n' +
|
||||
'\n' +
|
||||
'```markdown\n' +
|
||||
'Inside code: [[b]] should NOT count as a separate edge.\n' +
|
||||
'```\n';
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content }],
|
||||
new Set([fromPath, toPath]),
|
||||
);
|
||||
|
||||
// Only one IMPORTS edge from the real link; the in-code [[b]] should be
|
||||
// ignored. Even if it were not ignored, dedup would still cap us at 1.
|
||||
expect(stats.links).toBe(1);
|
||||
expect(getImports(graph)).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('does not link wikilink-shaped text inside inline code', () => {
|
||||
const fromPath = 'a.md';
|
||||
const graph = setupGraphWithFiles([fromPath, 'b.md']);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[
|
||||
{
|
||||
path: fromPath,
|
||||
content: 'Use the syntax `[[b]]` to create a wikilink.',
|
||||
},
|
||||
],
|
||||
new Set([fromPath, 'b.md']),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(0);
|
||||
expect(getImports(graph)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('dedups when the same target is reached by both link styles', () => {
|
||||
const fromPath = 'a.md';
|
||||
const toPath = 'b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, toPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[
|
||||
{
|
||||
path: fromPath,
|
||||
content: 'See [B](b.md) and also [[b]].',
|
||||
},
|
||||
],
|
||||
new Set([fromPath, toPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(1);
|
||||
const edges = getImports(graph);
|
||||
expect(edges).toHaveLength(1);
|
||||
// First-wins: markdown-link runs before wikilink scanner.
|
||||
expect(edges[0].reason).toBe('markdown-link');
|
||||
});
|
||||
|
||||
it('produces multiple IMPORTS edges for distinct wikilink targets', () => {
|
||||
const fromPath = 'a.md';
|
||||
const graph = setupGraphWithFiles([fromPath, 'b.md', 'c.md']);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[
|
||||
{
|
||||
path: fromPath,
|
||||
content: 'See [[b]], [[c]], and [[b]] again.',
|
||||
},
|
||||
],
|
||||
new Set([fromPath, 'b.md', 'c.md']),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(2);
|
||||
expect(hasImports(graph, fromPath, 'b.md')).toBe(true);
|
||||
expect(hasImports(graph, fromPath, 'c.md')).toBe(true);
|
||||
});
|
||||
|
||||
it('does not create self-loops', () => {
|
||||
const fromPath = 'a.md';
|
||||
const graph = setupGraphWithFiles([fromPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [[a]] (myself).' }],
|
||||
new Set([fromPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(0);
|
||||
expect(getImports(graph)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('falls back to repo-root when sibling path does not match', () => {
|
||||
// a.md lives in notes/, target is at repo root.
|
||||
const fromPath = 'notes/a.md';
|
||||
const toPath = 'b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, toPath]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [[b]].' }],
|
||||
new Set([fromPath, toPath]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(1);
|
||||
expect(hasImports(graph, fromPath, toPath)).toBe(true);
|
||||
});
|
||||
|
||||
it('prefers sibling over repo-root when both exist', () => {
|
||||
const fromPath = 'notes/a.md';
|
||||
const sibling = 'notes/b.md';
|
||||
const repoRoot = 'b.md';
|
||||
const graph = setupGraphWithFiles([fromPath, sibling, repoRoot]);
|
||||
|
||||
const stats = processMarkdown(
|
||||
graph,
|
||||
[{ path: fromPath, content: 'See [[b]].' }],
|
||||
new Set([fromPath, sibling, repoRoot]),
|
||||
);
|
||||
|
||||
expect(stats.links).toBe(1);
|
||||
expect(hasImports(graph, fromPath, sibling)).toBe(true);
|
||||
expect(hasImports(graph, fromPath, repoRoot)).toBe(false);
|
||||
});
|
||||
});
|
||||
Loading…
Add table
Reference in a new issue