Revert v1 incremental indexing (5 commits)

Reverts the v1 design that parsed only closure files into a fresh
graph and tried to hydrate the rest from DB. Real-repo equivalence
test failed: cross-file resolution operates on partial parse data
(closure files only), so CALLS edges that resolve through unchanged
files silently fall off. Diff against full rebuild on the same
edited state: -50 nodes, -425 edges, -5 communities, -48 processes.

Architecture pivot: switch to PR #533-style content-addressed parse
cache. Pipeline parses every file (cache-served when possible),
giving cross-file resolution full data, with DB writeback then
restricted to changed-file rows.

Reverts:
  d4b9de47 fix(incremental): drop invalid --no-renames=false
  f35f7634 feat(analyze): incremental orchestrator branch + meta schema
  bc039686 feat(pipeline): hydrate phase + parse-filter
  98bb893d feat(lbug): loadGraphFromLbug, queryImporters, ...
  aa8d7ae3 feat(incremental): change-detection, surface signatures, closure

Kept:
  d9e340b0 feat(communities): seed Leiden RNG (foundational)
  8235ca36 docs: incremental indexing design spec (will be revised)

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
abhigyanpatwari 2026-05-10 14:31:32 +05:30
parent d4b9de4765
commit 37ef3dda02
16 changed files with 13 additions and 2087 deletions

View file

@ -6,7 +6,6 @@ export type PipelinePhase =
| 'idle'
| 'extracting'
| 'structure'
| 'hydrate'
| 'parsing'
| 'imports'
| 'calls'

View file

@ -1,115 +0,0 @@
/**
* Iterative importer-closure computation for incremental indexing.
*
* Given a set of changed files, walks the IMPORTS graph (queried from the
* existing DB) to determine which other files must also be re-parsed for the
* incremental run to be byte-equivalent to a full rebuild.
*
* Algorithm:
* 1. Start with `closure = changedFiles`.
* 2. For each file `f` newly added to closure: parse it, extract surface.
* 3. If surface differs from `prevSurfaces[f]`, add all importers of `f`
* to closure.
* 4. Repeat until no new files added.
*
* Termination: each iteration either adds files or stops. The universe of
* files is finite (bounded by the repo size), so the loop terminates.
*
* Correctness invariant: a file is added to the closure iff its CALLS or
* IMPORTS edges might resolve differently than they did at `lastCommit`. The
* surface check is the crisp condition: surface unchanged → no resolver
* change → file's edges don't need re-emission.
*/
export interface ClosureInput<TParseResult> {
/** Initial set of files (from git diff: modified ∪ added). */
initialChangedFiles: Set<string>;
/** Previously stored surface signatures (filePath → hash). */
prevSurfaces: Record<string, string>;
/**
* Parse one file and return the parser result. The closure logic only
* cares about producing a stable surface signature, so callers may use
* the same parse worker the main pipeline uses.
*/
parseFile: (filePath: string) => Promise<TParseResult>;
/**
* Compute the surface signature for `filePath` given the parse result.
* Returned signature is hashed and compared against `prevSurfaces`.
*/
surfaceFor: (filePath: string, parsed: TParseResult) => string;
/**
* Query the existing DB for files that import `filePath`. Returns
* repo-relative paths. A missing file (e.g. importer of a brand-new file)
* legitimately returns `[]`.
*/
queryImporters: (filePath: string) => Promise<string[]>;
}
export interface ClosureResult<TParseResult> {
/** Final set of files that must be re-parsed. */
closure: Set<string>;
/**
* Cache of parsed results, keyed by file path. The orchestrator hands
* this to the pipeline so the parse phase doesn't re-parse closure files.
*/
parseCache: Map<string, TParseResult>;
/** Newly computed surface signatures, keyed by file path. */
newSurfaces: Map<string, string>;
/**
* Files added to closure ONLY because an importer chain pulled them in
* (not in the initial changed set). Useful for logging.
*/
expandedFromImporters: Set<string>;
}
/**
* Compute the transitive importer closure of the initial changed files,
* pruned by the surface-change optimization.
*
* The function is generic over `TParseResult` — the orchestrator is free
* to use the existing parse-worker result type, but the closure module
* itself is decoupled from any specific parse representation.
*/
export async function computeImporterClosure<TParseResult>(
input: ClosureInput<TParseResult>,
): Promise<ClosureResult<TParseResult>> {
const { initialChangedFiles, prevSurfaces, parseFile, surfaceFor, queryImporters } = input;
const closure = new Set<string>(initialChangedFiles);
const queue: string[] = [...initialChangedFiles];
const parseCache = new Map<string, TParseResult>();
const newSurfaces = new Map<string, string>();
const expandedFromImporters = new Set<string>();
while (queue.length > 0) {
const f = queue.shift()!;
// Parse this file (if we haven't already in this run).
let parsed = parseCache.get(f);
if (parsed === undefined) {
parsed = await parseFile(f);
parseCache.set(f, parsed);
}
// Compute the new surface signature.
const newSurface = surfaceFor(f, parsed);
newSurfaces.set(f, newSurface);
// If the surface changed, every file importing `f` may have stale
// resolver output and must be re-parsed.
const prev = prevSurfaces[f];
const changed = prev === undefined || prev !== newSurface;
if (changed) {
const importers = await queryImporters(f);
for (const i of importers) {
if (!closure.has(i)) {
closure.add(i);
queue.push(i);
expandedFromImporters.add(i);
}
}
}
}
return { closure, parseCache, newSurfaces, expandedFromImporters };
}

View file

@ -1,61 +0,0 @@
/**
* File-content hashing — v1 "surface signature" for incremental indexing.
*
* v1 trade-off: we use SHA-256 of file content as the surface signature.
* This is conservative — a body-only edit (which doesn't actually change
* any other file's resolution) still triggers 1-hop closure expansion
* because the content hash differs.
*
* v2 will switch to a real surface-only signature (extracted from the
* post-parse graph via `extractSurfaceSignature` in `surface.ts`) so that
* body-only edits stay at closure size 1. The plumbing for that is in
* place — `surface.ts` and the closure module are signature-agnostic —
* but reusing the parse-worker output for one-off surface extraction
* requires more pipeline integration than is needed for v1.
*/
import { createHash } from 'crypto';
import fs from 'fs/promises';
import path from 'path';
/**
* Compute SHA-256 hex digest of a single file. Returns null when the
* file can't be read (deleted between scan and hash, permission error,
* etc.) — caller should treat null as "no signature available, assume
* changed".
*/
export async function computeFileHash(absPath: string): Promise<string | null> {
try {
const buf = await fs.readFile(absPath);
return createHash('sha256').update(buf).digest('hex');
} catch {
return null;
}
}
/**
* Compute SHA-256 hashes for a list of files in `repoPath` (paths are
* repo-relative). Parallel batched I/O bounded at 100 concurrent reads
* to avoid fd exhaustion on huge repos.
*
* Returns a Map<relPath, hash>. Files that fail to read are omitted
* from the result (caller treats them as "no signature").
*/
export async function computeFileHashes(
repoPath: string,
relPaths: readonly string[],
): Promise<Map<string, string>> {
const out = new Map<string, string>();
const BATCH = 100;
for (let i = 0; i < relPaths.length; i += BATCH) {
const batch = relPaths.slice(i, i + BATCH);
const results = await Promise.all(
batch.map(async (rel) => {
const hash = await computeFileHash(path.join(repoPath, rel));
return hash ? ([rel, hash] as const) : null;
}),
);
for (const r of results) if (r) out.set(r[0], r[1]);
}
return out;
}

View file

@ -1,211 +0,0 @@
/**
* Git-based change detection for incremental indexing.
*
* Combines `git diff --name-status <lastCommit> HEAD` (committed changes since
* last index) with `git status --porcelain` (uncommitted/dirty tree changes)
* to produce a precise per-file change set.
*
* Renames (R<sim> in diff output, R in porcelain) are flattened into
* delete(old) + add(new) — downstream consumers don't need to know about
* renames specifically; they just need to know "this path's nodes are stale,
* delete them" and "this path is new, parse it".
*
* Returns a typed result; throws `LastCommitMissingError` when `lastCommit`
* no longer exists in the repository (rebase or shallow clone). Callers
* should fall back to a full rebuild on this error.
*/
import { execFileSync } from 'child_process';
export interface ChangedFiles {
/** Files whose content changed (re-parse needed). */
modified: string[];
/** Files newly introduced since lastCommit (re-parse needed). */
added: string[];
/** Files that existed at lastCommit but no longer do (delete-only, no parse). */
deleted: string[];
}
export class LastCommitMissingError extends Error {
constructor(commit: string) {
super(`lastCommit ${commit} not found in repository — cannot compute incremental diff`);
this.name = 'LastCommitMissingError';
}
}
/**
* Compute the set of files changed in `repoPath` since `lastCommit`.
* Combines committed differences (`git diff` against HEAD) with the
* current dirty working-tree state (`git status`).
*
* @throws LastCommitMissingError when `lastCommit` is not reachable.
*/
export function getChangedFilesSinceCommit(
repoPath: string,
lastCommit: string,
): ChangedFiles {
if (!commitExists(repoPath, lastCommit)) {
throw new LastCommitMissingError(lastCommit);
}
const modified = new Set<string>();
const added = new Set<string>();
const deleted = new Set<string>();
// ── Committed differences: lastCommit..HEAD ────────────────────────────
// Output (NUL-delimited via -z): STATUS\0path1\0[path2\0]
// Status codes: M, A, D, T (type change → treat as modified),
// R<sim>, C<sim> (rename/copy with similarity %).
const diffRaw = runGit(repoPath, [
'diff',
'--name-status',
'-z',
`${lastCommit}`,
'HEAD',
]);
for (const entry of parseNameStatusZ(diffRaw)) {
classify(entry, modified, added, deleted);
}
// ── Working-tree differences: HEAD..disk ───────────────────────────────
// Format (NUL-delimited): XY<space>path[\0orig-path]
// X = staged, Y = unstaged. Either non-' ' counts as a change.
const statusRaw = runGit(repoPath, ['status', '--porcelain', '-z']);
for (const entry of parsePorcelainZ(statusRaw)) {
classify(entry, modified, added, deleted);
}
// Resolve overlaps: a file added in the diff and modified in the status
// is just "new on disk" → keep it in `added`. A file deleted in the diff
// but present in status as added (re-introduced) → modified.
for (const f of added) {
modified.delete(f);
deleted.delete(f);
}
for (const f of deleted) {
modified.delete(f);
}
return {
modified: [...modified].sort(),
added: [...added].sort(),
deleted: [...deleted].sort(),
};
}
interface ParsedEntry {
status: string;
path: string;
origPath?: string;
}
function classify(
entry: ParsedEntry,
modified: Set<string>,
added: Set<string>,
deleted: Set<string>,
): void {
const code = entry.status[0];
switch (code) {
case 'M':
case 'T': // type change (file → symlink, etc.)
modified.add(entry.path);
break;
case 'A':
case '?': // untracked (porcelain '??') treat as added
added.add(entry.path);
break;
case 'D':
deleted.add(entry.path);
break;
case 'R': // rename: flatten to delete(orig) + add(new)
case 'C': // copy: original survives; new file is added
if (entry.origPath && code === 'R') deleted.add(entry.origPath);
added.add(entry.path);
break;
case 'U': // unmerged — surface as modified, caller's problem
modified.add(entry.path);
break;
default:
// Unknown status — be conservative, treat as modified.
modified.add(entry.path);
}
}
/**
* Parse `git diff --name-status -z` output. NUL-delimited, status before each path:
* "M\0a.ts\0A\0b.ts\0R100\0old.ts\0new.ts\0"
*/
function parseNameStatusZ(raw: string): ParsedEntry[] {
const entries: ParsedEntry[] = [];
if (!raw) return entries;
const tokens = raw.split('\0').filter((t) => t.length > 0);
let i = 0;
while (i < tokens.length) {
const status = tokens[i++];
const path = tokens[i++];
if (path === undefined) break;
if (status[0] === 'R' || status[0] === 'C') {
const newPath = tokens[i++];
if (newPath !== undefined) {
entries.push({ status, path: newPath, origPath: path });
}
} else {
entries.push({ status, path });
}
}
return entries;
}
/**
* Parse `git status --porcelain -z` output. NUL-delimited, two-char status
* code + space + path; for renames the original path follows after another NUL:
* "M a.ts\0?? b.ts\0R new.ts\0old.ts\0"
*/
function parsePorcelainZ(raw: string): ParsedEntry[] {
const entries: ParsedEntry[] = [];
if (!raw) return entries;
const tokens = raw.split('\0').filter((t) => t.length > 0);
let i = 0;
while (i < tokens.length) {
const tok = tokens[i++];
// First two chars are the XY status, then a space, then the path.
const xy = tok.slice(0, 2);
const path = tok.slice(3);
// Effective status: prefer staged (X) when not ' ', otherwise unstaged (Y).
const code = xy[0] !== ' ' && xy[0] !== '?' ? xy[0] : xy[1];
if (xy.startsWith('R') || xy[1] === 'R') {
// Rename: next token is the original path.
const orig = tokens[i++];
entries.push({ status: 'R', path, origPath: orig });
} else if (xy === '??') {
entries.push({ status: '?', path });
} else {
entries.push({ status: code, path });
}
}
return entries;
}
function commitExists(repoPath: string, commit: string): boolean {
try {
execFileSync('git', ['cat-file', '-e', `${commit}^{commit}`], {
cwd: repoPath,
stdio: ['ignore', 'ignore', 'ignore'],
});
return true;
} catch {
return false;
}
}
function runGit(repoPath: string, args: string[]): string {
return execFileSync('git', args, {
cwd: repoPath,
stdio: ['ignore', 'pipe', 'ignore'],
encoding: 'utf8',
// Larger buffer for large repos with many changes.
maxBuffer: 100 * 1024 * 1024,
});
}

View file

@ -1,267 +0,0 @@
/**
* Incremental-indexing orchestrator helpers.
*
* High-level helpers used by `run-analyze.ts` to keep the incremental path
* out of the main full-rebuild orchestrator function. Each helper has one
* responsibility:
*
* isIncrementalEligible(...) — should this run go incremental?
* computeIncrementalClosure(...) — git diff → content-hash → closure
* extractIncrementalSubgraph(...) — ctx.graph → "nodes/edges to write"
* commitIncrementalProgress(...) — set the dirty flag in meta.json
*
* See docs/superpowers/specs/2026-05-10-incremental-indexing-design.md.
*/
import { execFileSync } from 'child_process';
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
import { createKnowledgeGraph } from '../graph/graph.js';
import type { KnowledgeGraph } from '../graph/types.js';
import {
type RepoMeta,
INCREMENTAL_SCHEMA_VERSION,
saveMeta,
} from '../../storage/repo-manager.js';
import { hasGitDir } from '../../storage/git.js';
import {
getChangedFilesSinceCommit,
type ChangedFiles,
} from './git-diff.js';
import { computeImporterClosure } from './closure.js';
import { computeFileHashes } from './file-hash.js';
import { queryImporters } from '../lbug/lbug-adapter.js';
/** Decision returned by isIncrementalEligible. */
export interface IncrementalEligibility {
/** True iff this run should attempt the incremental path. */
eligible: boolean;
/** When `eligible === false`, a short reason for logging. */
reason?: string;
/** The previous lastCommit, when eligible. */
lastCommit?: string;
}
/**
* Decide whether a run is eligible for the incremental path.
*
* All conditions must hold:
* - --force not passed
* - existing meta.json present and previously a full rebuild has populated
* surfaceSignatures + schemaVersion (matching CURRENT)
* - repo has .git (non-git repos always do full rebuild)
* - meta.lastCommit still resolvable (not rebased away)
* - no incrementalInProgress flag (would force full rebuild for safety)
*/
export function isIncrementalEligible(
repoPath: string,
existingMeta: RepoMeta | null | undefined,
optionsForce: boolean | undefined,
): IncrementalEligibility {
if (optionsForce) {
return { eligible: false, reason: '--force passed' };
}
if (!existingMeta) {
return { eligible: false, reason: 'no existing index' };
}
if (existingMeta.incrementalInProgress) {
return {
eligible: false,
reason: 'previous incremental run did not complete cleanly',
};
}
if (
existingMeta.schemaVersion === undefined ||
existingMeta.schemaVersion !== INCREMENTAL_SCHEMA_VERSION
) {
return {
eligible: false,
reason: `schemaVersion mismatch (have ${existingMeta.schemaVersion}, want ${INCREMENTAL_SCHEMA_VERSION})`,
};
}
if (!existingMeta.surfaceSignatures || Object.keys(existingMeta.surfaceSignatures).length === 0) {
return {
eligible: false,
reason: 'no prior surfaceSignatures in meta.json',
};
}
if (!hasGitDir(repoPath)) {
return { eligible: false, reason: 'non-git repo' };
}
if (!existingMeta.lastCommit) {
return { eligible: false, reason: 'no lastCommit recorded' };
}
if (!commitExists(repoPath, existingMeta.lastCommit)) {
return {
eligible: false,
reason: `lastCommit ${existingMeta.lastCommit.slice(0, 7)} not in repo`,
};
}
return { eligible: true, lastCommit: existingMeta.lastCommit };
}
function commitExists(repoPath: string, commit: string): boolean {
try {
execFileSync('git', ['cat-file', '-e', `${commit}^{commit}`], {
cwd: repoPath,
stdio: ['ignore', 'ignore', 'ignore'],
});
return true;
} catch {
return false;
}
}
export interface IncrementalSetupResult {
/** Files that must be re-parsed in this run. */
closure: Set<string>;
/** Files deleted on disk since lastCommit (rows must be removed from DB). */
deletedFiles: string[];
/** Per-file content hashes computed during closure expansion. */
newFileHashes: Map<string, string>;
/** ChangedFiles result for diagnostics. */
changes: ChangedFiles;
}
/**
* Compute the closure of files to re-parse for an incremental run.
*
* Algorithm:
* 1. git diff lastCommit HEAD ∪ git status → ChangedFiles
* 2. closure ← modified ∪ added
* 3. For each f in closure: hash(f); if hash differs from previous, query
* DB importers and add them to closure. Iterate to fixpoint.
*
* Throws if `lastCommit` is gone (caller should fall back to full rebuild).
*/
export async function computeIncrementalClosure(
repoPath: string,
lastCommit: string,
prevSurfaces: Record<string, string>,
): Promise<IncrementalSetupResult> {
const changes = getChangedFilesSinceCommit(repoPath, lastCommit);
const initialChangedFiles = new Set<string>([...changes.modified, ...changes.added]);
// Closure module wants generic parseFile + surfaceFor. v1 uses
// file-content hashes (cheap, conservative). The "ParseResult" type
// is just the content hash itself.
const { closure, newSurfaces } = await computeImporterClosure<string>({
initialChangedFiles,
prevSurfaces,
parseFile: async (filePath) => {
const hashes = await computeFileHashes(repoPath, [filePath]);
// Missing files (race with delete) → empty hash; counts as "changed".
return hashes.get(filePath) ?? '';
},
surfaceFor: (_filePath, parsed) => parsed,
queryImporters: async (filePath) => queryImporters(filePath),
});
return {
closure,
deletedFiles: changes.deleted,
newFileHashes: newSurfaces,
changes,
};
}
/**
* Persist the `incrementalInProgress` dirty flag to meta.json BEFORE any
* destructive DB mutation. The flag is cleared on success by overwriting
* meta.json with the final state. If the run crashes between, the next
* run sees the flag and forces a full rebuild.
*/
export async function commitIncrementalProgress(
storagePath: string,
existingMeta: RepoMeta,
closure: Set<string>,
): Promise<void> {
const meta: RepoMeta = {
...existingMeta,
incrementalInProgress: {
closure: [...closure],
startedAt: Date.now(),
},
};
await saveMeta(storagePath, meta);
}
/**
* Build a subgraph of `ctx.graph` containing ONLY the nodes/edges that
* need to be written to LadybugDB in incremental mode:
*
* - All nodes whose filePath is in `closure` (newly parsed in this run).
* - All graph-wide nodes (Community, Process) — they're regenerated by
* the communities/processes phases on every run.
* - All edges where AT LEAST ONE endpoint is in this set. Edges entirely
* between hydrated unchanged-file nodes are NOT included — they're
* already in the DB and re-inserting them would PK-conflict.
*
* This lets us call the existing `loadGraphToLbug` against the filtered
* subgraph: the COPY semantics will write only what's missing, while the
* unchanged-unchanged DB rows we never deleted stay intact.
*/
export function extractIncrementalSubgraph(
fullGraph: KnowledgeGraph,
closure: ReadonlySet<string>,
): KnowledgeGraph {
const sub = createKnowledgeGraph();
const isGraphWide = (label: string): boolean => label === 'Community' || label === 'Process';
// Phase 1: nodes
const writableNodeIds = new Set<string>();
fullGraph.forEachNode((n: GraphNode) => {
const filePath = n.properties?.filePath as string | undefined;
const inClosure = filePath ? closure.has(filePath) : false;
if (inClosure || isGraphWide(n.label)) {
sub.addNode(n);
writableNodeIds.add(n.id);
}
});
// Phase 2: edges where AT LEAST ONE endpoint is in the writable set.
// Edges entirely between hydrated nodes are skipped (already in DB).
fullGraph.forEachRelationship((r: GraphRelationship) => {
if (writableNodeIds.has(r.sourceId) || writableNodeIds.has(r.targetId)) {
sub.addRelationship(r);
}
});
return sub;
}
/**
* Compute the surface signatures for every file in the repo from a graph.
* In v1 this is just the per-file content hash (cheap, conservative). v2
* will switch to true surface-only signatures derived via `surface.ts`.
*
* Used after a full rebuild to populate `meta.json.surfaceSignatures` so
* the next run can run incrementally.
*/
export async function computeAllFileSignatures(
repoPath: string,
filePaths: readonly string[],
): Promise<Record<string, string>> {
const map = await computeFileHashes(repoPath, filePaths);
const out: Record<string, string> = {};
for (const [k, v] of map) out[k] = v;
return out;
}
/**
* Merge an updated set of file hashes into a previous snapshot. Used after
* an incremental run: `prevSurfaces` is the set in meta.json, `newSurfaces`
* is what we computed for closure files this run, and `deletedFiles` are
* files that no longer exist on disk.
*/
export function mergeSurfaceSignatures(
prevSurfaces: Record<string, string>,
newSurfaces: Map<string, string>,
deletedFiles: readonly string[],
): Record<string, string> {
const merged: Record<string, string> = { ...prevSurfaces };
for (const f of deletedFiles) delete merged[f];
for (const [f, h] of newSurfaces) merged[f] = h;
return merged;
}

View file

@ -1,135 +0,0 @@
/**
* Public-surface signature extraction for incremental indexing.
*
* The surface signature of a file is a stable hash of everything that could
* affect callers in *other* files: the names, signatures, and heritage of
* its publicly-visible symbols (functions, classes, methods, interfaces).
*
* If a file's surface hash is unchanged between runs, no other file's
* resolution depends on the changed content — we only need to re-parse the
* file itself, not its importers. This is the closure-scoping optimization
* driven by `closure.ts`.
*
* The hash is content-only and excludes formatting, comments, and ordering
* (we sort the symbol list before hashing). It does NOT include implementation
* bodies — that's the whole point: a body-only edit produces the same surface.
*/
import { createHash } from 'crypto';
import type { GraphNode } from 'gitnexus-shared';
import type { KnowledgeGraph } from '../graph/types.js';
/** Node labels whose presence in a file contributes to its public surface. */
const SURFACE_LABELS = new Set<string>([
'Function',
'Class',
'Method',
'Constructor',
'Interface',
'TypeAlias',
'Enum',
'Struct',
'Trait',
'Namespace',
'Module',
'Const',
'Static',
'Property',
'Record',
'Delegate',
'Annotation',
'Template',
'Union',
'Macro',
'Typedef',
]);
/**
* Extract the surface signature of `filePath` from `graph`.
* Returns a stable hex hash; identical-surface inputs → identical output.
*/
export function extractSurfaceSignature(
graph: KnowledgeGraph,
filePath: string,
): string {
const lines: string[] = [];
// Gather surface-relevant nodes for this file.
const surfaceNodes: GraphNode[] = [];
graph.forEachNode((node) => {
if (
node.properties?.filePath === filePath &&
SURFACE_LABELS.has(node.label)
) {
surfaceNodes.push(node);
}
});
// Sort deterministically by id so reorder doesn't affect the hash.
surfaceNodes.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0));
for (const n of surfaceNodes) {
lines.push(serializeNode(n));
}
// Heritage: edges that fan OUT of this file affect downstream resolution.
// (EXTENDS / IMPLEMENTS targets — if the parent class changes, dependents
// may resolve methods differently.) These edges are co-located with the
// child symbol whose `filePath` matches; we walk them off the source side.
const edgeKeys: string[] = [];
graph.forEachRelationship((rel) => {
if (rel.type !== 'EXTENDS' && rel.type !== 'IMPLEMENTS') return;
const src = graph.getNode(rel.sourceId);
if (!src) return;
if (src.properties?.filePath !== filePath) return;
edgeKeys.push(`H ${rel.type} ${rel.sourceId} -> ${rel.targetId}`);
});
edgeKeys.sort();
for (const k of edgeKeys) lines.push(k);
const h = createHash('sha256');
for (const line of lines) {
h.update(line);
h.update('\n');
}
return h.digest('hex');
}
/** True iff `current` differs from `prev` (or `prev` is undefined). */
export function surfaceChanged(
prev: string | undefined,
current: string,
): boolean {
return prev === undefined || prev !== current;
}
/**
* Serialize a single surface node to a stable string. We include label,
* name, parameter count + types, return type, and visibility-style flags
* — anything that could affect how a caller resolves against this node.
*
* We DO NOT include line numbers, file offsets, or any other position
* data: surface should be invariant under whitespace/formatting changes.
*/
function serializeNode(node: GraphNode): string {
const p = (node.properties ?? {}) as Record<string, unknown>;
const parts: string[] = [
`N`,
node.label,
String(p.name ?? ''),
`id=${node.id}`,
];
if (p.parameterCount !== undefined) parts.push(`pc=${String(p.parameterCount)}`);
if (Array.isArray(p.parameterTypes)) {
parts.push(`pt=${(p.parameterTypes as string[]).join(',')}`);
}
if (p.returnType !== undefined) parts.push(`rt=${String(p.returnType)}`);
if (p.visibility !== undefined) parts.push(`v=${String(p.visibility)}`);
if (p.isStatic) parts.push('static');
if (p.isAbstract) parts.push('abstract');
if (p.isReadonly) parts.push('readonly');
if (p.isAsync) parts.push('async');
// `level` (inheritance depth marker for methods) affects MRO
if (p.level !== undefined) parts.push(`lvl=${String(p.level)}`);
return parts.join('|');
}

View file

@ -1,91 +0,0 @@
/**
* Phase: hydrate
*
* In incremental-indexing mode, populates `ctx.graph` with all nodes and
* relationships belonging to files OUTSIDE the current closure (i.e., files
* that didn't change). This way the parse phase only needs to re-emit nodes
* and edges for the closure files, while downstream phases (mro, communities,
* processes) see a fully-populated graph and produce results equivalent to a
* full rebuild.
*
* No-op in full-rebuild mode: when `ctx.options.filesToParse` is unset,
* the pipeline starts with an empty graph and parses every file (existing
* behavior).
*
* @deps structure
* @reads allPaths (from structure)
* @writes graph (every node/edge belonging to unchanged files)
*/
import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js';
import { getPhaseOutput } from './types.js';
import type { StructureOutput } from './structure.js';
import { loadGraphFromLbug } from '../../lbug/lbug-adapter.js';
import { isDev } from '../utils/env.js';
import { logger } from '../../logger.js';
export interface HydrateOutput {
/** True when this run actually loaded prior state (incremental). */
readonly hydrated: boolean;
/** Number of nodes loaded from DB (0 in full-rebuild mode). */
readonly nodesLoaded: number;
/** Number of relationships loaded from DB (0 in full-rebuild mode). */
readonly edgesLoaded: number;
}
export const hydratePhase: PipelinePhase<HydrateOutput> = {
name: 'hydrate',
deps: ['structure'],
async execute(
ctx: PipelineContext,
deps: ReadonlyMap<string, PhaseResult<unknown>>,
): Promise<HydrateOutput> {
const filesToParse = ctx.options?.filesToParse;
// Full-rebuild mode: nothing to hydrate.
if (!filesToParse) {
return { hydrated: false, nodesLoaded: 0, edgesLoaded: 0 };
}
const { allPaths, totalFiles } = getPhaseOutput<StructureOutput>(deps, 'structure');
// Compute the unchanged complement: every scanned path NOT in the closure.
const unchanged = new Set<string>();
for (const p of allPaths) {
if (!filesToParse.has(p)) unchanged.add(p);
}
ctx.onProgress({
phase: 'hydrate',
percent: 22,
message: `Hydrating ${unchanged.size} unchanged files from index...`,
stats: { filesProcessed: 0, totalFiles, nodesCreated: ctx.graph.nodeCount },
});
const result = await loadGraphFromLbug(ctx.graph, unchanged);
if (isDev) {
logger.info(
`💧 Hydrate: ${result.nodesLoaded} nodes, ${result.edgesLoaded} edges loaded for ${unchanged.size} unchanged files (closure: ${filesToParse.size})`,
);
}
ctx.onProgress({
phase: 'hydrate',
percent: 25,
message: `Hydrated ${result.nodesLoaded} nodes from previous index`,
stats: {
filesProcessed: unchanged.size,
totalFiles,
nodesCreated: ctx.graph.nodeCount,
},
});
return {
hydrated: true,
nodesLoaded: result.nodesLoaded,
edgesLoaded: result.edgesLoaded,
};
},
};

View file

@ -9,7 +9,6 @@
export { scanPhase, type ScanOutput } from './scan.js';
export { structurePhase, type StructureOutput } from './structure.js';
export { hydratePhase, type HydrateOutput } from './hydrate.js';
export { markdownPhase, type MarkdownOutput } from './markdown.js';
export { cobolPhase, type CobolOutput } from './cobol.js';
export { parsePhase, type ParseOutput } from './parse.js';

View file

@ -96,18 +96,9 @@ export const parsePhase: PipelinePhase<ParseOutput> = {
'structure',
);
// Incremental-indexing filter: when `filesToParse` is set, only the
// closure files get parsed in this run. Files outside the closure had
// their nodes/edges pre-loaded by the `hydrate` phase. See
// docs/superpowers/specs/2026-05-10-incremental-indexing-design.md.
const filesToParse = ctx.options?.filesToParse;
const targetScanned = filesToParse
? scannedFiles.filter((f) => filesToParse.has(f.path))
: scannedFiles;
const result = await runChunkedParseAndResolve(
ctx.graph,
targetScanned,
scannedFiles,
allPaths,
totalFiles,
ctx.repoPath,

View file

@ -23,7 +23,6 @@ import {
getPhaseOutput,
scanPhase,
structurePhase,
hydratePhase,
markdownPhase,
cobolPhase,
parsePhase,
@ -56,18 +55,6 @@ export interface PipelineOptions {
minFiles?: number;
minBytes?: number;
};
/**
* Incremental-indexing mode: when set, the parse phase only re-parses
* files in this set, and the new `hydrate` phase pre-loads node/edge
* state for everything else from the existing LadybugDB index.
*
* Unset (the default) → full-rebuild mode: parse phase processes every
* scanned file and hydrate is a no-op. Set by `runFullAnalysis` when it
* detects an eligible incremental run; never set by callers directly.
*
* See `docs/superpowers/specs/2026-05-10-incremental-indexing-design.md`.
*/
filesToParse?: ReadonlySet<string>;
}
// ── Phase registry ─────────────────────────────────────────────────────────
@ -87,7 +74,6 @@ function buildPhaseList(options?: PipelineOptions): PipelinePhase[] {
const phases: PipelinePhase[] = [
scanPhase,
structurePhase,
hydratePhase,
markdownPhase,
cobolPhase,
parsePhase,

View file

@ -1204,245 +1204,6 @@ export const deleteNodesForFile = async (
export const getEmbeddingTableName = (): string => EMBEDDING_TABLE_NAME;
// ============================================================================
// Incremental indexing: DB → KnowledgeGraph hydration
// ============================================================================
/**
* Node tables that have a `filePath` column and are eligible for hydration.
* `Community` and `Process` are graph-wide (no filePath) and are ALWAYS
* regenerated by the communities/processes phases — we never load them back.
*/
const HYDRATABLE_NODE_TABLES: readonly NodeTableName[] = NODE_TABLES.filter(
(t) => t !== 'Community' && t !== 'Process',
);
/** Per-table extra columns to project beyond the base (id, name, filePath). */
const TABLE_EXTRA_COLUMNS: Record<string, string[]> = {
File: ['content'],
Folder: [],
Function: ['startLine', 'endLine', 'isExported', 'content', 'description'],
Class: ['startLine', 'endLine', 'isExported', 'content', 'description'],
Interface: ['startLine', 'endLine', 'isExported', 'content', 'description'],
Method: [
'startLine',
'endLine',
'isExported',
'content',
'description',
'parameterCount',
'returnType',
],
CodeElement: ['startLine', 'endLine', 'isExported', 'content', 'description'],
Section: ['startLine', 'endLine', 'level', 'content', 'description'],
Property: ['startLine', 'endLine', 'content', 'description', 'declaredType'],
Route: ['responseKeys', 'errorKeys', 'middleware'],
Tool: ['description'],
};
/** All other tables share the CODE_ELEMENT_BASE schema (startLine/endLine/content/description). */
const DEFAULT_EXTRA_COLUMNS = ['startLine', 'endLine', 'content', 'description'];
const escapeFilePath = (s: string): string =>
s.replace(/\\/g, '\\\\').replace(/'/g, "''");
/**
* Hydrate `graph` with all nodes (and their relationships) belonging to files
* in `unchangedFilePaths`. Used by the incremental-indexing pipeline to
* pre-populate the in-memory graph with everything that didn't change, so
* downstream phases (mro, communities, processes) see a complete picture.
*
* Skips Community/Process labels and their MEMBER_OF / STEP_IN_PROCESS edges
* — those are graph-wide and will be regenerated from scratch by the
* pipeline's downstream phases.
*
* Returns counts for logging/diagnostics.
*/
export const loadGraphFromLbug = async (
graph: KnowledgeGraph,
unchangedFilePaths: ReadonlySet<string>,
): Promise<{ nodesLoaded: number; edgesLoaded: number }> => {
if (!conn) {
throw new Error('LadybugDB not initialized. Call initLbug first.');
}
if (unchangedFilePaths.size === 0) {
return { nodesLoaded: 0, edgesLoaded: 0 };
}
let nodesLoaded = 0;
let edgesLoaded = 0;
// Track which node IDs we successfully loaded so the relationship pass
// can verify both endpoints are present (cheaper than a DB-side join).
const loadedNodeIds = new Set<string>();
// ── 1. Hydrate nodes per table ─────────────────────────────────────────
// Chunk filePaths to keep query size manageable on huge repos.
const CHUNK = 200;
const filePathArr = [...unchangedFilePaths];
for (const tableName of HYDRATABLE_NODE_TABLES) {
const t = escapeTableName(tableName);
const extras = TABLE_EXTRA_COLUMNS[tableName] ?? DEFAULT_EXTRA_COLUMNS;
// Always include base columns: id, name, filePath
const cols = ['id', 'name', 'filePath', ...extras];
const projection = cols.map((c) => `n.${c} AS ${c}`).join(', ');
for (let i = 0; i < filePathArr.length; i += CHUNK) {
const batch = filePathArr.slice(i, i + CHUNK);
const inList = batch.map((p) => `'${escapeFilePath(p)}'`).join(',');
const cypher = `MATCH (n:${t}) WHERE n.filePath IN [${inList}] RETURN ${projection}`;
try {
const queryResult = await conn.query(cypher);
const result = Array.isArray(queryResult) ? queryResult[0] : queryResult;
const rows = await result.getAll();
for (const row of rows) {
const id = typeof row.id === 'string' ? row.id : String(row.id ?? '');
if (!id) continue;
const properties: Record<string, unknown> = {};
for (const c of cols) {
const v = row[c];
if (v !== undefined && v !== null) properties[c] = v;
}
// Required for downstream phases: ensure name/filePath always set.
if (properties.name === undefined) properties.name = '';
graph.addNode({
id,
label: tableName as unknown as import('gitnexus-shared').NodeLabel,
properties: properties as import('gitnexus-shared').NodeProperties,
});
loadedNodeIds.add(id);
nodesLoaded++;
}
} catch {
// Some tables may not exist in the schema or may be empty — that's fine.
}
}
}
// ── 2. Hydrate relationships ───────────────────────────────────────────
// We pull all CodeRelation rows whose endpoints are nodes we just loaded.
// Using SRC.filePath IN [...] AND TGT.filePath IN [...] is precise but
// requires LadybugDB to traverse with both endpoint constraints. We
// exclude graph-wide edge types (MEMBER_OF, STEP_IN_PROCESS) entirely —
// they'll be regenerated.
//
// Strategy: for each (src filePath chunk × tgt filePath chunk) we'd risk
// O(n²) chunking. Instead, we fetch by source-chunk, then verify the
// target endpoint is in `loadedNodeIds` JS-side. This keeps the DB query
// bounded by source chunks while preserving correctness.
for (let i = 0; i < filePathArr.length; i += CHUNK) {
const batch = filePathArr.slice(i, i + CHUNK);
const inList = batch.map((p) => `'${escapeFilePath(p)}'`).join(',');
const cypher = `
MATCH (a)-[r:${REL_TABLE_NAME}]->(b)
WHERE a.filePath IN [${inList}]
AND r.type <> 'MEMBER_OF'
AND r.type <> 'STEP_IN_PROCESS'
RETURN a.id AS src, b.id AS dst, r.type AS type,
r.confidence AS confidence, r.reason AS reason, r.step AS step
`;
try {
const queryResult = await conn.query(cypher);
const result = Array.isArray(queryResult) ? queryResult[0] : queryResult;
const rows = await result.getAll();
for (const row of rows) {
const src = typeof row.src === 'string' ? row.src : '';
const dst = typeof row.dst === 'string' ? row.dst : '';
if (!src || !dst) continue;
// The other endpoint must be a node we loaded — otherwise the edge
// crosses into a closure file (will be re-emitted) or a dropped
// graph-wide node (Community/Process), and we skip it.
if (!loadedNodeIds.has(dst)) continue;
const relType = typeof row.type === 'string' ? row.type : '';
if (!relType) continue;
const relId = `${src}_${relType}_${dst}`;
graph.addRelationship({
id: relId,
sourceId: src,
targetId: dst,
type: relType as import('gitnexus-shared').RelationshipType,
confidence: typeof row.confidence === 'number' ? row.confidence : 1.0,
reason: typeof row.reason === 'string' ? row.reason : '',
step:
typeof row.step === 'number' && row.step !== 0 ? row.step : undefined,
});
edgesLoaded++;
}
} catch {
// Continue on chunk failure — best-effort hydration.
}
}
return { nodesLoaded, edgesLoaded };
};
/**
* Query the IMPORTS edge table for all files that import `targetFilePath`.
* Used by the incremental-indexing closure-expansion logic to find files
* whose resolution may be stale when `targetFilePath`'s public surface
* changes.
*
* Returns repo-relative paths (i.e. the source file's `filePath`), de-duplicated.
*/
export const queryImporters = async (targetFilePath: string): Promise<string[]> => {
if (!conn) {
throw new Error('LadybugDB not initialized. Call initLbug first.');
}
const escaped = escapeFilePath(targetFilePath);
const cypher = `
MATCH (a)-[r:${REL_TABLE_NAME}]->(b)
WHERE r.type = 'IMPORTS' AND b.filePath = '${escaped}'
RETURN DISTINCT a.filePath AS importer
`;
try {
const queryResult = await conn.query(cypher);
const result = Array.isArray(queryResult) ? queryResult[0] : queryResult;
const rows = await result.getAll();
const out: string[] = [];
for (const row of rows) {
const p = row.importer;
if (typeof p === 'string' && p.length > 0) out.push(p);
}
return out;
} catch {
return [];
}
};
/**
* Delete all Community and Process nodes and their MEMBER_OF /
* STEP_IN_PROCESS edges. Used at the start of an incremental run so the
* communities/processes phases can regenerate them on the merged graph.
*/
export const deleteAllCommunitiesAndProcesses = async (): Promise<{
nodesDeleted: number;
}> => {
if (!conn) {
throw new Error('LadybugDB not initialized. Call initLbug first.');
}
let nodesDeleted = 0;
for (const label of ['Community', 'Process']) {
try {
const countResult = await conn.query(
`MATCH (n:${label}) RETURN count(n) AS cnt`,
);
const result = Array.isArray(countResult) ? countResult[0] : countResult;
const rows = await result.getAll();
const count = Number(rows[0]?.cnt ?? rows[0]?.[0] ?? 0);
if (count > 0) {
await conn.query(`MATCH (n:${label}) DETACH DELETE n`);
nodesDeleted += count;
}
} catch {
// table may not exist yet
}
}
return { nodesDeleted };
};
// ============================================================================
// Full-Text Search (FTS) Functions
// ============================================================================

View file

@ -11,7 +11,6 @@
import path from 'path';
import fs from 'fs/promises';
import { execFileSync } from 'child_process';
import { runPipelineFromRepo } from './ingestion/pipeline.js';
import {
initLbug,
@ -21,8 +20,6 @@ import {
executeWithReusedStatement,
closeLbug,
loadCachedEmbeddings,
deleteNodesForFile,
deleteAllCommunitiesAndProcesses,
} from './lbug/lbug-adapter.js';
import { createSearchFTSIndexes } from './search/fts-indexes.js';
import {
@ -32,7 +29,6 @@ import {
ensureGitNexusIgnored,
registerRepo,
cleanupOldKuzuFiles,
INCREMENTAL_SCHEMA_VERSION,
} from '../storage/repo-manager.js';
import {
getCurrentCommit,
@ -45,14 +41,6 @@ import type { CachedEmbedding } from './embeddings/types.js';
import { generateAIContextFiles } from '../cli/ai-context.js';
import { EMBEDDING_TABLE_NAME } from './lbug/schema.js';
import { STALE_HASH_SENTINEL } from './lbug/schema.js';
import {
isIncrementalEligible,
computeIncrementalClosure,
commitIncrementalProgress,
extractIncrementalSubgraph,
mergeSurfaceSignatures,
computeAllFileSignatures,
} from './incremental/orchestrator.js';
// ---------------------------------------------------------------------------
// Public types
@ -192,65 +180,20 @@ export async function runFullAnalysis(
if (existingMeta && !options.force && existingMeta.lastCommit === currentCommit) {
// Non-git folders have currentCommit = '' — always rebuild since we can't detect changes
if (currentCommit !== '') {
// For git repos, even if HEAD matches lastCommit, the working tree
// may have uncommitted changes. Only short-circuit when the working
// tree is also clean.
const dirty = hasDirtyTree(repoPath);
if (!dirty) {
await ensureGitNexusIgnored(repoPath);
return {
// `resolveRepoIdentityRoot` collapses worktree roots to the
// canonical repo basename (#1259) but leaves arbitrary subdirs
// and `--skip-git` paths unchanged (#1232/#1233 intent preserved).
repoName:
options.registryName ??
getInferredRepoName(repoPath) ??
path.basename(resolveRepoIdentityRoot(repoPath)),
repoPath,
stats: existingMeta.stats ?? {},
alreadyUpToDate: true,
};
}
}
}
// ── Incremental branch ─────────────────────────────────────────────
// Try the incremental path first. Falls through to full rebuild on:
// - --force
// - missing/old meta.json
// - lastCommit gone (rebased)
// - schemaVersion mismatch
// - non-git repo
// - any error during incremental setup
const eligibility = isIncrementalEligible(repoPath, existingMeta, options.force);
if (eligibility.eligible && existingMeta) {
try {
const result = await runIncrementalBranch(
await ensureGitNexusIgnored(repoPath);
return {
// `resolveRepoIdentityRoot` collapses worktree roots to the
// canonical repo basename (#1259) but leaves arbitrary subdirs
// and `--skip-git` paths unchanged (#1232/#1233 intent preserved).
repoName:
options.registryName ??
getInferredRepoName(repoPath) ??
path.basename(resolveRepoIdentityRoot(repoPath)),
repoPath,
storagePath,
lbugPath,
existingMeta,
currentCommit,
options,
callbacks,
);
if (result) return result;
// result === null → falls through to full rebuild (e.g. setup failed)
} catch (err) {
log(
`Incremental run failed (${(err as Error).message}); ` +
`next run will full-rebuild via dirty-flag.`,
);
// Re-throw — the dirty flag is set; next run forces full rebuild.
try {
await closeLbug();
} catch {
/* swallow */
}
throw err;
stats: existingMeta.stats ?? {},
alreadyUpToDate: true,
};
}
} else if (existingMeta) {
log(`Incremental skipped: ${eligibility.reason ?? 'unknown reason'}`);
}
// ── Cache embeddings from existing index before rebuild ────────────
@ -511,32 +454,6 @@ export async function runFullAnalysis(
const effectiveSemanticMode =
semanticMode ??
(runtimeCapabilities.semanticMode === 'vector-index' ? 'vector-index' : 'exact-scan');
// Compute per-file surface signatures so the next run can be incremental.
// v1 uses content-hash (cheap, conservative). v2 will switch to a
// surface-only signature so body-only edits don't expand the closure.
let surfaceSignatures: Record<string, string> | undefined;
if (hasGitDir(repoPath)) {
try {
const allFilePaths: string[] = [];
pipelineResult.graph.forEachNode((n) => {
const fp = n.properties?.filePath as string | undefined;
if (fp && (n.label === 'File' || n.label === 'Folder')) {
// File nodes carry their own path; we want every distinct repo-relative
// file path that participated in indexing. Use File nodes as the
// authoritative source.
if (n.label === 'File') allFilePaths.push(fp);
}
});
if (allFilePaths.length > 0) {
surfaceSignatures = await computeAllFileSignatures(repoPath, allFilePaths);
}
} catch {
/* surface signatures are best-effort; their absence just means the
* next run will fall back to full rebuild. */
}
}
const meta = {
repoPath,
lastCommit: currentCommit,
@ -548,14 +465,6 @@ export async function runFullAnalysis(
// origin remote, which is fine: paths-only repos behave as
// before.
remoteUrl: hasGitDir(repoPath) ? getRemoteUrl(repoPath) : undefined,
// Incremental-indexing fields — populated for git repos only. The
// next analyze run reads these to decide whether to take the
// incremental path. See the Incremental branch above.
schemaVersion: surfaceSignatures ? INCREMENTAL_SCHEMA_VERSION : undefined,
surfaceSignatures,
incrementalInProgress: undefined as
| { closure: string[]; startedAt: number }
| undefined,
stats: {
files: pipelineResult.totalFileCount,
nodes: stats.nodes,
@ -645,242 +554,3 @@ export async function runFullAnalysis(
throw err;
}
}
// ===========================================================================
// Incremental analysis branch — invoked from runFullAnalysis when eligible.
// See docs/superpowers/specs/2026-05-10-incremental-indexing-design.md.
// ===========================================================================
/**
* Cheap check: does the working tree have uncommitted changes? Used to
* decide whether the "lastCommit == HEAD" early-exit is safe to take.
*/
function hasDirtyTree(repoPath: string): boolean {
try {
const out = execFileSync('git', ['status', '--porcelain'], {
cwd: repoPath,
stdio: ['ignore', 'pipe', 'ignore'],
encoding: 'utf8',
});
return out.trim().length > 0;
} catch {
// If git status fails for any reason, conservatively assume dirty so
// we don't accidentally short-circuit a real change.
return true;
}
}
/**
* Execute the incremental branch of `runFullAnalysis`. Returns null when
* setup determines a full rebuild is needed instead (caller falls through).
*
* Throws if the run fails after the dirty flag is set — caller is
* responsible for surfacing the error; the next run will detect the dirty
* flag and force a full rebuild.
*/
async function runIncrementalBranch(
repoPath: string,
storagePath: string,
lbugPath: string,
existingMeta: import('../storage/repo-manager.js').RepoMeta,
currentCommit: string,
options: AnalyzeOptions,
callbacks: AnalyzeCallbacks,
): Promise<AnalyzeResult | null> {
const log = (msg: string) => callbacks.onLog?.(msg);
const progress = (phase: string, percent: number, message: string) =>
callbacks.onProgress(phase, percent, message);
log(
`Incremental: probing for changes since ${existingMeta.lastCommit.slice(0, 7)}...`,
);
// 1. Compute closure (parses each file's content hash, walks DB importers).
let setup: Awaited<ReturnType<typeof computeIncrementalClosure>>;
try {
// Open the existing DB so closure expansion can query the IMPORTS edges.
await initLbug(lbugPath);
setup = await computeIncrementalClosure(
repoPath,
existingMeta.lastCommit,
existingMeta.surfaceSignatures ?? {},
);
} catch (e) {
try {
await closeLbug();
} catch {
/* swallow */
}
log(
`Incremental setup failed: ${(e as Error).message}. Falling back to full rebuild.`,
);
return null;
}
// 2. No changes? Update lastCommit and return early.
if (setup.closure.size === 0 && setup.deletedFiles.length === 0) {
log('Incremental: no file changes detected — refreshing meta only.');
try {
await closeLbug();
} catch {
/* swallow */
}
const meta: import('../storage/repo-manager.js').RepoMeta = {
...existingMeta,
lastCommit: currentCommit,
indexedAt: new Date().toISOString(),
};
await saveMeta(storagePath, meta);
await ensureGitNexusIgnored(repoPath);
return {
repoName:
options.registryName ??
getInferredRepoName(repoPath) ??
path.basename(resolveRepoIdentityRoot(repoPath)),
repoPath,
stats: existingMeta.stats ?? {},
alreadyUpToDate: true,
};
}
log(
`Incremental: closure=${setup.closure.size} (changed=${setup.changes.modified.length} ` +
`+ added=${setup.changes.added.length} + importers=${setup.closure.size - setup.changes.modified.length - setup.changes.added.length}), ` +
`deleted=${setup.deletedFiles.length}`,
);
// 3. Mark dirty BEFORE any DB mutation. Closes lbug temporarily to
// release the connection while saveMeta writes.
await commitIncrementalProgress(storagePath, existingMeta, setup.closure);
// 4. Delete stale rows from the DB.
progress('lbug', 5, 'Removing stale rows for changed files...');
for (const file of setup.closure) {
try {
await deleteNodesForFile(file);
} catch {
/* file may not have been indexed yet — fine */
}
}
for (const file of setup.deletedFiles) {
try {
await deleteNodesForFile(file);
} catch {
/* fine */
}
}
// Always wipe Community / Process — they're regenerated by downstream
// pipeline phases and must come from the merged graph for correctness
// (Leiden runs on the FULL hydrated + parsed graph).
await deleteAllCommunitiesAndProcesses();
// 5. Run the pipeline with filesToParse set so:
// - hydrate phase fills ctx.graph with unchanged-file nodes from DB
// - parse phase only parses closure files
// - downstream phases (mro, communities, processes) see the full graph
const pipelineResult = await runPipelineFromRepo(
repoPath,
(p) => {
const phaseLabel = PHASE_LABELS[p.phase] || p.phase;
const scaled = 10 + Math.round(p.percent * 0.55); // 10–65%
const message = p.detail ? `${p.message || phaseLabel} (${p.detail})` : p.message || phaseLabel;
progress(p.phase, scaled, message);
},
{ filesToParse: setup.closure },
);
// 6. Extract the subgraph that needs to be written: closure-file nodes
// + graph-wide nodes + edges incident to them. Hydrated unchanged-file
// nodes are NOT in this subgraph — their rows are still in DB.
progress('lbug', 70, 'Writing incremental updates to LadybugDB...');
const subgraph = extractIncrementalSubgraph(pipelineResult.graph, setup.closure);
await loadGraphToLbug(subgraph, repoPath, storagePath, (msg) => {
progress('lbug', 80, msg);
});
// 7. Recreate FTS indexes (cheap; full rebuild over the merged DB state).
progress('fts', 90, 'Refreshing search indexes...');
try {
await createSearchFTSIndexes();
} catch {
/* FTS is best-effort; log only */
}
// 8. Compute final stats from the live DB state.
const stats = await getLbugStats();
// 9. Compute new meta (merge surface signatures, clear dirty flag).
const mergedSurfaces = mergeSurfaceSignatures(
existingMeta.surfaceSignatures ?? {},
setup.newFileHashes,
setup.deletedFiles,
);
const newMeta: import('../storage/repo-manager.js').RepoMeta = {
...existingMeta,
lastCommit: currentCommit,
indexedAt: new Date().toISOString(),
remoteUrl: getRemoteUrl(repoPath) ?? existingMeta.remoteUrl,
schemaVersion: INCREMENTAL_SCHEMA_VERSION,
surfaceSignatures: mergedSurfaces,
incrementalInProgress: undefined, // explicit clear
stats: {
...(existingMeta.stats ?? {}),
files: pipelineResult.totalFileCount,
nodes: stats.nodes,
edges: stats.edges,
communities: pipelineResult.communityResult?.stats.totalCommunities,
processes: pipelineResult.processResult?.stats.totalProcesses,
},
};
// 10. Persist meta + register repo.
await saveMeta(storagePath, newMeta);
const projectName = await registerRepo(repoPath, newMeta, {
name: options.registryName,
allowDuplicateName: options.allowDuplicateName,
});
// 11. Best-effort AI-context regeneration so AGENTS.md/CLAUDE.md stay
// in sync with the post-incremental graph state.
let aggregatedClusterCount = 0;
if (pipelineResult.communityResult?.communities) {
const groups = new Map<string, number>();
for (const c of pipelineResult.communityResult.communities) {
const label = c.heuristicLabel || c.label || 'Unknown';
groups.set(label, (groups.get(label) || 0) + c.symbolCount);
}
aggregatedClusterCount = Array.from(groups.values()).filter((cnt) => cnt >= 5).length;
}
try {
await generateAIContextFiles(
repoPath,
storagePath,
projectName,
{
files: pipelineResult.totalFileCount,
nodes: stats.nodes,
edges: stats.edges,
communities: pipelineResult.communityResult?.stats.totalCommunities,
clusters: aggregatedClusterCount,
processes: pipelineResult.processResult?.stats.totalProcesses,
},
undefined,
{ skipAgentsMd: options.skipAgentsMd, noStats: options.noStats },
);
} catch {
/* best-effort */
}
await ensureGitNexusIgnored(repoPath);
await closeLbug();
progress('done', 100, 'Incremental complete');
return {
repoName: projectName,
repoPath,
stats: newMeta.stats ?? {},
pipelineResult,
};
}

View file

@ -71,48 +71,8 @@ export interface RepoMeta {
processes?: number;
embeddings?: number;
};
/**
* Bumped whenever incremental-indexing invariants change in an
* incompatible way (schema bump, closure-algorithm fix, etc.).
* Mismatch → run-analyze forces a full rebuild.
* See docs/superpowers/specs/2026-05-10-incremental-indexing-design.md.
*/
schemaVersion?: number;
/**
* Per-file content/surface signatures used to drive incremental closure
* expansion. v1: SHA-256 of file content (cheap, conservative — body-only
* edits trigger 1-hop closure expansion). v2 will switch to a real
* surface-only signature so body-only edits stay at closure size 1.
*
* Map keys are repo-relative file paths; values are hex digests. Files
* not in this map are treated as "no previous signature" → first-time
* processing.
*/
surfaceSignatures?: Record<string, string>;
/**
* Crash-recovery dirty flag. Written by run-analyze BEFORE any DB
* mutation in an incremental run; cleared by overwrite on success. Its
* presence at the start of the next run forces a full rebuild — the
* cheapest path back to a known-good index after a crashed/cancelled
* incremental run. Consumed by run-analyze.ts.
*/
incrementalInProgress?: {
closure: string[];
startedAt: number;
};
}
/**
* Bumped whenever incremental-indexing invariants change in an
* incompatible way. v1 is `1`. Increment when:
* - The closure-expansion algorithm changes in a way that prior
* surfaceSignatures cannot be trusted.
* - The hydrate phase's DB schema assumptions change.
* - The surface-signature derivation changes (so prior hashes are not
* comparable to current ones).
*/
export const INCREMENTAL_SCHEMA_VERSION = 1;
export interface IndexedRepo {
repoPath: string;
storagePath: string;

View file

@ -1,212 +0,0 @@
import { describe, it, expect, vi } from 'vitest';
import { computeImporterClosure } from '../../src/core/incremental/closure.js';
/**
* Mini test harness: a fixture import graph + per-file surface signatures.
*
* `imports[X] = [a, b]` means files `a` and `b` import file `X`. So when we
* ask for "importers of X" the answer is `[a, b]`.
*/
interface Fixture {
files: string[];
/** target → list of importers */
imports: Record<string, string[]>;
/** previous surfaces */
prevSurfaces: Record<string, string>;
/** new surfaces (what parseFile + surfaceFor will produce) */
newSurfaces: Record<string, string>;
}
function makeHarness(fx: Fixture) {
const parseFile = vi.fn(async (f: string) => ({ filePath: f }));
const surfaceFor = vi.fn((f: string) => fx.newSurfaces[f] ?? '');
const queryImporters = vi.fn(async (f: string) => fx.imports[f] ?? []);
return { parseFile, surfaceFor, queryImporters };
}
describe('computeImporterClosure', () => {
it('empty input → empty closure', async () => {
const { parseFile, surfaceFor, queryImporters } = makeHarness({
files: [],
imports: {},
prevSurfaces: {},
newSurfaces: {},
});
const r = await computeImporterClosure({
initialChangedFiles: new Set(),
prevSurfaces: {},
parseFile,
surfaceFor,
queryImporters,
});
expect(r.closure.size).toBe(0);
expect(parseFile).not.toHaveBeenCalled();
});
it('single file with unchanged surface → closure size 1, no expansion', async () => {
const { parseFile, surfaceFor, queryImporters } = makeHarness({
files: ['a.ts'],
imports: { 'a.ts': ['b.ts', 'c.ts'] },
prevSurfaces: { 'a.ts': 'sig-a-v1' },
newSurfaces: { 'a.ts': 'sig-a-v1' }, // same
});
const r = await computeImporterClosure({
initialChangedFiles: new Set(['a.ts']),
prevSurfaces: { 'a.ts': 'sig-a-v1' },
parseFile,
surfaceFor,
queryImporters,
});
expect([...r.closure].sort()).toEqual(['a.ts']);
// queryImporters should not have been called: surface unchanged.
expect(queryImporters).not.toHaveBeenCalled();
expect(r.expandedFromImporters.size).toBe(0);
});
it('single file with changed surface → expands to direct importers', async () => {
const { parseFile, surfaceFor, queryImporters } = makeHarness({
files: ['a.ts', 'b.ts', 'c.ts'],
imports: {
'a.ts': ['b.ts', 'c.ts'],
'b.ts': [],
'c.ts': [],
},
prevSurfaces: { 'a.ts': 'sig-old', 'b.ts': 'sig-b', 'c.ts': 'sig-c' },
newSurfaces: { 'a.ts': 'sig-new', 'b.ts': 'sig-b', 'c.ts': 'sig-c' },
});
const r = await computeImporterClosure({
initialChangedFiles: new Set(['a.ts']),
prevSurfaces: { 'a.ts': 'sig-old', 'b.ts': 'sig-b', 'c.ts': 'sig-c' },
parseFile,
surfaceFor,
queryImporters,
});
expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts', 'c.ts']);
expect([...r.expandedFromImporters].sort()).toEqual(['b.ts', 'c.ts']);
});
it('multi-hop cascade (A surface change → B in closure → B surface change → C in closure)', async () => {
const { parseFile, surfaceFor, queryImporters } = makeHarness({
files: ['a.ts', 'b.ts', 'c.ts'],
imports: {
'a.ts': ['b.ts'], // b imports a
'b.ts': ['c.ts'], // c imports b
'c.ts': [],
},
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old', 'c.ts': 'c-stable' },
newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-new', 'c.ts': 'c-stable' },
});
const r = await computeImporterClosure({
initialChangedFiles: new Set(['a.ts']),
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old', 'c.ts': 'c-stable' },
parseFile,
surfaceFor,
queryImporters,
});
expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts', 'c.ts']);
});
it('cascade halts when surface stops changing mid-chain', async () => {
const { parseFile, surfaceFor, queryImporters } = makeHarness({
files: ['a.ts', 'b.ts', 'c.ts'],
imports: {
'a.ts': ['b.ts'],
'b.ts': ['c.ts'],
},
// a's surface changed, b is in closure but b's surface unchanged → c stays out
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-stable', 'c.ts': 'c-stable' },
newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-stable', 'c.ts': 'c-stable' },
});
const r = await computeImporterClosure({
initialChangedFiles: new Set(['a.ts']),
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-stable', 'c.ts': 'c-stable' },
parseFile,
surfaceFor,
queryImporters,
});
expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts']);
expect([...r.expandedFromImporters].sort()).toEqual(['b.ts']);
});
it('terminates on cycles in the import graph', async () => {
// a ↔ b cycle. a's surface changes, b is added; b's surface also "changed"
// (relative to undefined prev), so its importers are queried — which
// includes a, already in closure. Loop terminates because closure
// membership prevents re-add.
const { parseFile, surfaceFor, queryImporters } = makeHarness({
files: ['a.ts', 'b.ts'],
imports: { 'a.ts': ['b.ts'], 'b.ts': ['a.ts'] },
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' },
newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-new' },
});
const r = await computeImporterClosure({
initialChangedFiles: new Set(['a.ts']),
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' },
parseFile,
surfaceFor,
queryImporters,
});
expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts']);
// Each file parsed exactly once even though they import each other.
expect(parseFile).toHaveBeenCalledTimes(2);
});
it('newly-added file (no prev surface) triggers expansion', async () => {
// For a brand-new file, prevSurfaces[f] is undefined → surfaceChanged
// returns true → its importers are queried. (For a truly *new* file,
// there should be no importers yet, so closure stays at {f}.)
const { parseFile, surfaceFor, queryImporters } = makeHarness({
files: ['new.ts'],
imports: {},
prevSurfaces: {},
newSurfaces: { 'new.ts': 'sig-new' },
});
const r = await computeImporterClosure({
initialChangedFiles: new Set(['new.ts']),
prevSurfaces: {},
parseFile,
surfaceFor,
queryImporters,
});
expect([...r.closure]).toEqual(['new.ts']);
expect(queryImporters).toHaveBeenCalledOnce();
});
it('parses each file exactly once and caches the result', async () => {
const { parseFile, surfaceFor, queryImporters } = makeHarness({
files: ['a.ts', 'b.ts'],
imports: { 'a.ts': ['b.ts'] },
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b' },
newSurfaces: { 'a.ts': 'new', 'b.ts': 'b' },
});
const r = await computeImporterClosure({
initialChangedFiles: new Set(['a.ts']),
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b' },
parseFile,
surfaceFor,
queryImporters,
});
expect(parseFile).toHaveBeenCalledTimes(2);
expect(r.parseCache.size).toBe(2);
expect(r.parseCache.has('a.ts')).toBe(true);
expect(r.parseCache.has('b.ts')).toBe(true);
});
it('records new surfaces for every parsed file', async () => {
const { parseFile, surfaceFor, queryImporters } = makeHarness({
files: ['a.ts', 'b.ts'],
imports: { 'a.ts': ['b.ts'] },
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' },
newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-new' },
});
const r = await computeImporterClosure({
initialChangedFiles: new Set(['a.ts']),
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' },
parseFile,
surfaceFor,
queryImporters,
});
expect(r.newSurfaces.get('a.ts')).toBe('new');
expect(r.newSurfaces.get('b.ts')).toBe('b-new');
});
});

View file

@ -1,163 +0,0 @@
import { describe, it, expect, vi, beforeEach } from 'vitest';
import { execFileSync } from 'child_process';
import {
getChangedFilesSinceCommit,
LastCommitMissingError,
} from '../../src/core/incremental/git-diff.js';
vi.mock('child_process', () => ({
execFileSync: vi.fn(),
}));
const mockExec = vi.mocked(execFileSync);
/**
* Mock helper: queue a sequence of execFileSync return values.
* Order matters — the first call returns the first value, etc.
*
* Calls in `getChangedFilesSinceCommit`:
* 1. cat-file -e <commit> (commitExists check)
* 2. diff --name-status -z (committed changes)
* 3. status --porcelain -z (dirty tree)
*/
function mockGitSequence(
catFileSucceeds: boolean,
diffOutput: string,
statusOutput: string,
) {
mockExec.mockReset();
if (catFileSucceeds) {
mockExec.mockImplementationOnce(() => Buffer.from(''));
} else {
mockExec.mockImplementationOnce(() => {
throw new Error('not in repo');
});
}
mockExec.mockImplementationOnce(() => diffOutput);
mockExec.mockImplementationOnce(() => statusOutput);
}
describe('getChangedFilesSinceCommit', () => {
beforeEach(() => {
vi.clearAllMocks();
});
it('throws LastCommitMissingError when commit not in repo', () => {
mockGitSequence(false, '', '');
expect(() => getChangedFilesSinceCommit('/repo', 'deadbeef')).toThrow(
LastCommitMissingError,
);
});
it('returns empty arrays for clean repo and no committed changes', () => {
mockGitSequence(true, '', '');
const r = getChangedFilesSinceCommit('/repo', 'abc');
expect(r).toEqual({ modified: [], added: [], deleted: [] });
});
it('parses committed M/A/D from diff --name-status -z', () => {
mockGitSequence(
true,
'M\0src/a.ts\0A\0src/b.ts\0D\0src/c.ts\0',
'',
);
const r = getChangedFilesSinceCommit('/repo', 'abc');
expect(r).toEqual({
modified: ['src/a.ts'],
added: ['src/b.ts'],
deleted: ['src/c.ts'],
});
});
it('flattens R<sim> renames into delete(orig) + add(new)', () => {
mockGitSequence(
true,
'R100\0src/old.ts\0src/new.ts\0',
'',
);
const r = getChangedFilesSinceCommit('/repo', 'abc');
expect(r).toEqual({
modified: [],
added: ['src/new.ts'],
deleted: ['src/old.ts'],
});
});
it('treats T (type change) as modified', () => {
mockGitSequence(true, 'T\0src/link.ts\0', '');
const r = getChangedFilesSinceCommit('/repo', 'abc');
expect(r.modified).toContain('src/link.ts');
});
it('parses dirty tree from status --porcelain -z', () => {
// 'M a.ts' = staged-modified. ' M b.ts' = unstaged-modified.
// '?? c.ts' = untracked. ' D d.ts' = unstaged-deleted.
mockGitSequence(
true,
'',
'M a.ts\0 M b.ts\0?? c.ts\0 D d.ts\0',
);
const r = getChangedFilesSinceCommit('/repo', 'abc');
expect(r.modified.sort()).toEqual(['a.ts', 'b.ts']);
expect(r.added).toEqual(['c.ts']);
expect(r.deleted).toEqual(['d.ts']);
});
it('unions committed + dirty changes', () => {
mockGitSequence(
true,
'M\0a.ts\0', // committed: a.ts modified
' M b.ts\0?? c.ts\0', // dirty: b.ts modified, c.ts untracked
);
const r = getChangedFilesSinceCommit('/repo', 'abc');
expect(r.modified.sort()).toEqual(['a.ts', 'b.ts']);
expect(r.added).toEqual(['c.ts']);
expect(r.deleted).toEqual([]);
});
it('resolves overlap: file added in diff and modified in status → added', () => {
mockGitSequence(
true,
'A\0newfile.ts\0',
' M newfile.ts\0',
);
const r = getChangedFilesSinceCommit('/repo', 'abc');
expect(r).toEqual({
modified: [],
added: ['newfile.ts'],
deleted: [],
});
});
it('handles porcelain rename ("R new\\0old")', () => {
mockGitSequence(
true,
'',
'R new.ts\0old.ts\0',
);
const r = getChangedFilesSinceCommit('/repo', 'abc');
expect(r.added).toEqual(['new.ts']);
// Porcelain rename: `R` in status doesn't pre-flatten old into deleted
// because git already resolved the rename — but our parser does flatten
// when the diff layer flags it. For status-only, the original is the
// rename source and we don't claim to know it was deleted from index.
});
it('returns sorted output for stable comparison', () => {
mockGitSequence(
true,
'M\0z.ts\0M\0a.ts\0M\0m.ts\0',
'',
);
const r = getChangedFilesSinceCommit('/repo', 'abc');
expect(r.modified).toEqual(['a.ts', 'm.ts', 'z.ts']);
});
it('passes correct cwd to git', () => {
mockGitSequence(true, '', '');
getChangedFilesSinceCommit('/some/path', 'abc');
for (const call of mockExec.mock.calls) {
expect((call[2] as { cwd: string }).cwd).toBe('/some/path');
}
});
});

View file

@ -1,185 +0,0 @@
import { describe, it, expect } from 'vitest';
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
import {
extractSurfaceSignature,
surfaceChanged,
} from '../../src/core/incremental/surface.js';
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
function fn(
id: string,
filePath: string,
name: string,
extra: Record<string, unknown> = {},
): GraphNode {
return {
id,
label: 'Function',
properties: { name, filePath, ...extra },
};
}
function method(
id: string,
filePath: string,
name: string,
extra: Record<string, unknown> = {},
): GraphNode {
return {
id,
label: 'Method',
properties: { name, filePath, ...extra },
};
}
function cls(
id: string,
filePath: string,
name: string,
extra: Record<string, unknown> = {},
): GraphNode {
return {
id,
label: 'Class',
properties: { name, filePath, ...extra },
};
}
function rel(
id: string,
type: GraphRelationship['type'],
src: string,
dst: string,
): GraphRelationship {
return { id, type, sourceId: src, targetId: dst, confidence: 1, reason: 't' };
}
describe('extractSurfaceSignature', () => {
it('returns a stable hash for an empty file', () => {
const g = createKnowledgeGraph();
const h1 = extractSurfaceSignature(g, 'a.ts');
const h2 = extractSurfaceSignature(g, 'a.ts');
expect(h1).toBe(h2);
expect(h1).toMatch(/^[a-f0-9]{64}$/);
});
it('produces the same hash for the same surface', () => {
const g1 = createKnowledgeGraph();
g1.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
const g2 = createKnowledgeGraph();
g2.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
expect(extractSurfaceSignature(g1, 'a.ts')).toBe(
extractSurfaceSignature(g2, 'a.ts'),
);
});
it('is invariant under node insertion order', () => {
const g1 = createKnowledgeGraph();
g1.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
g1.addNode(fn('Function:a.ts:bar#1', 'a.ts', 'bar', { parameterCount: 1 }));
const g2 = createKnowledgeGraph();
g2.addNode(fn('Function:a.ts:bar#1', 'a.ts', 'bar', { parameterCount: 1 }));
g2.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
expect(extractSurfaceSignature(g1, 'a.ts')).toBe(
extractSurfaceSignature(g2, 'a.ts'),
);
});
it('changes when a function is renamed', () => {
const g1 = createKnowledgeGraph();
g1.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo'));
const g2 = createKnowledgeGraph();
g2.addNode(fn('Function:a.ts:bar#0', 'a.ts', 'bar'));
expect(extractSurfaceSignature(g1, 'a.ts')).not.toBe(
extractSurfaceSignature(g2, 'a.ts'),
);
});
it('changes when a function signature changes', () => {
const g1 = createKnowledgeGraph();
g1.addNode(
fn('Function:a.ts:foo#1', 'a.ts', 'foo', {
parameterCount: 1,
parameterTypes: ['number'],
returnType: 'string',
}),
);
const g2 = createKnowledgeGraph();
g2.addNode(
fn('Function:a.ts:foo#1', 'a.ts', 'foo', {
parameterCount: 1,
parameterTypes: ['string'],
returnType: 'string',
}),
);
expect(extractSurfaceSignature(g1, 'a.ts')).not.toBe(
extractSurfaceSignature(g2, 'a.ts'),
);
});
it('does NOT change for body-only edits (no surface mutation)', () => {
// Body-only edits don't add/remove/rename surface nodes — same hash.
const g = createKnowledgeGraph();
g.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
const h1 = extractSurfaceSignature(g, 'a.ts');
// Re-build with same surface (simulating a re-parse of a body-only edit).
const g2 = createKnowledgeGraph();
g2.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
const h2 = extractSurfaceSignature(g2, 'a.ts');
expect(h1).toBe(h2);
});
it('only considers nodes whose filePath matches', () => {
const g = createKnowledgeGraph();
g.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo'));
g.addNode(fn('Function:b.ts:bar#0', 'b.ts', 'bar'));
const ha = extractSurfaceSignature(g, 'a.ts');
const hb = extractSurfaceSignature(g, 'b.ts');
expect(ha).not.toBe(hb);
// Adding a node in a OTHER file should not change a's signature.
g.addNode(fn('Function:c.ts:baz#0', 'c.ts', 'baz'));
expect(extractSurfaceSignature(g, 'a.ts')).toBe(ha);
});
it('changes when EXTENDS target changes', () => {
const g1 = createKnowledgeGraph();
g1.addNode(cls('Class:a.ts:Child#0', 'a.ts', 'Child'));
g1.addNode(cls('Class:b.ts:ParentA#0', 'b.ts', 'ParentA'));
g1.addRelationship(
rel('r1', 'EXTENDS', 'Class:a.ts:Child#0', 'Class:b.ts:ParentA#0'),
);
const g2 = createKnowledgeGraph();
g2.addNode(cls('Class:a.ts:Child#0', 'a.ts', 'Child'));
g2.addNode(cls('Class:b.ts:ParentB#0', 'b.ts', 'ParentB'));
g2.addRelationship(
rel('r1', 'EXTENDS', 'Class:a.ts:Child#0', 'Class:b.ts:ParentB#0'),
);
expect(extractSurfaceSignature(g1, 'a.ts')).not.toBe(
extractSurfaceSignature(g2, 'a.ts'),
);
});
it('includes Method/Class/Interface in the surface', () => {
const g = createKnowledgeGraph();
g.addNode(cls('Class:a.ts:C#0', 'a.ts', 'C'));
g.addNode(method('Method:a.ts:C.m#0', 'a.ts', 'm'));
const h = extractSurfaceSignature(g, 'a.ts');
expect(h).toMatch(/^[a-f0-9]{64}$/);
});
});
describe('surfaceChanged', () => {
it('returns true when prev is undefined (first index of file)', () => {
expect(surfaceChanged(undefined, 'abc')).toBe(true);
});
it('returns false when prev equals current', () => {
expect(surfaceChanged('abc', 'abc')).toBe(false);
});
it('returns true when prev differs from current', () => {
expect(surfaceChanged('abc', 'def')).toBe(true);
});
});