diff --git a/gitnexus-shared/src/pipeline.ts b/gitnexus-shared/src/pipeline.ts index 70b8645da..5f7e61c57 100644 --- a/gitnexus-shared/src/pipeline.ts +++ b/gitnexus-shared/src/pipeline.ts @@ -6,7 +6,6 @@ export type PipelinePhase = | 'idle' | 'extracting' | 'structure' - | 'hydrate' | 'parsing' | 'imports' | 'calls' diff --git a/gitnexus/src/core/incremental/closure.ts b/gitnexus/src/core/incremental/closure.ts deleted file mode 100644 index 6e07d23f1..000000000 --- a/gitnexus/src/core/incremental/closure.ts +++ /dev/null @@ -1,115 +0,0 @@ -/** - * Iterative importer-closure computation for incremental indexing. - * - * Given a set of changed files, walks the IMPORTS graph (queried from the - * existing DB) to determine which other files must also be re-parsed for the - * incremental run to be byte-equivalent to a full rebuild. - * - * Algorithm: - * 1. Start with `closure = changedFiles`. - * 2. For each file `f` newly added to closure: parse it, extract surface. - * 3. If surface differs from `prevSurfaces[f]`, add all importers of `f` - * to closure. - * 4. Repeat until no new files added. - * - * Termination: each iteration either adds files or stops. The universe of - * files is finite (bounded by the repo size), so the loop terminates. - * - * Correctness invariant: a file is added to the closure iff its CALLS or - * IMPORTS edges might resolve differently than they did at `lastCommit`. The - * surface check is the crisp condition: surface unchanged → no resolver - * change → file's edges don't need re-emission. - */ - -export interface ClosureInput { - /** Initial set of files (from git diff: modified ∪ added). */ - initialChangedFiles: Set; - /** Previously stored surface signatures (filePath → hash). */ - prevSurfaces: Record; - /** - * Parse one file and return the parser result. The closure logic only - * cares about producing a stable surface signature, so callers may use - * the same parse worker the main pipeline uses. - */ - parseFile: (filePath: string) => Promise; - /** - * Compute the surface signature for `filePath` given the parse result. - * Returned signature is hashed and compared against `prevSurfaces`. - */ - surfaceFor: (filePath: string, parsed: TParseResult) => string; - /** - * Query the existing DB for files that import `filePath`. Returns - * repo-relative paths. A missing file (e.g. importer of a brand-new file) - * legitimately returns `[]`. - */ - queryImporters: (filePath: string) => Promise; -} - -export interface ClosureResult { - /** Final set of files that must be re-parsed. */ - closure: Set; - /** - * Cache of parsed results, keyed by file path. The orchestrator hands - * this to the pipeline so the parse phase doesn't re-parse closure files. - */ - parseCache: Map; - /** Newly computed surface signatures, keyed by file path. */ - newSurfaces: Map; - /** - * Files added to closure ONLY because an importer chain pulled them in - * (not in the initial changed set). Useful for logging. - */ - expandedFromImporters: Set; -} - -/** - * Compute the transitive importer closure of the initial changed files, - * pruned by the surface-change optimization. - * - * The function is generic over `TParseResult` — the orchestrator is free - * to use the existing parse-worker result type, but the closure module - * itself is decoupled from any specific parse representation. - */ -export async function computeImporterClosure( - input: ClosureInput, -): Promise> { - const { initialChangedFiles, prevSurfaces, parseFile, surfaceFor, queryImporters } = input; - - const closure = new Set(initialChangedFiles); - const queue: string[] = [...initialChangedFiles]; - const parseCache = new Map(); - const newSurfaces = new Map(); - const expandedFromImporters = new Set(); - - while (queue.length > 0) { - const f = queue.shift()!; - - // Parse this file (if we haven't already in this run). - let parsed = parseCache.get(f); - if (parsed === undefined) { - parsed = await parseFile(f); - parseCache.set(f, parsed); - } - - // Compute the new surface signature. - const newSurface = surfaceFor(f, parsed); - newSurfaces.set(f, newSurface); - - // If the surface changed, every file importing `f` may have stale - // resolver output and must be re-parsed. - const prev = prevSurfaces[f]; - const changed = prev === undefined || prev !== newSurface; - if (changed) { - const importers = await queryImporters(f); - for (const i of importers) { - if (!closure.has(i)) { - closure.add(i); - queue.push(i); - expandedFromImporters.add(i); - } - } - } - } - - return { closure, parseCache, newSurfaces, expandedFromImporters }; -} diff --git a/gitnexus/src/core/incremental/file-hash.ts b/gitnexus/src/core/incremental/file-hash.ts deleted file mode 100644 index 673e655af..000000000 --- a/gitnexus/src/core/incremental/file-hash.ts +++ /dev/null @@ -1,61 +0,0 @@ -/** - * File-content hashing — v1 "surface signature" for incremental indexing. - * - * v1 trade-off: we use SHA-256 of file content as the surface signature. - * This is conservative — a body-only edit (which doesn't actually change - * any other file's resolution) still triggers 1-hop closure expansion - * because the content hash differs. - * - * v2 will switch to a real surface-only signature (extracted from the - * post-parse graph via `extractSurfaceSignature` in `surface.ts`) so that - * body-only edits stay at closure size 1. The plumbing for that is in - * place — `surface.ts` and the closure module are signature-agnostic — - * but reusing the parse-worker output for one-off surface extraction - * requires more pipeline integration than is needed for v1. - */ - -import { createHash } from 'crypto'; -import fs from 'fs/promises'; -import path from 'path'; - -/** - * Compute SHA-256 hex digest of a single file. Returns null when the - * file can't be read (deleted between scan and hash, permission error, - * etc.) — caller should treat null as "no signature available, assume - * changed". - */ -export async function computeFileHash(absPath: string): Promise { - try { - const buf = await fs.readFile(absPath); - return createHash('sha256').update(buf).digest('hex'); - } catch { - return null; - } -} - -/** - * Compute SHA-256 hashes for a list of files in `repoPath` (paths are - * repo-relative). Parallel batched I/O bounded at 100 concurrent reads - * to avoid fd exhaustion on huge repos. - * - * Returns a Map. Files that fail to read are omitted - * from the result (caller treats them as "no signature"). - */ -export async function computeFileHashes( - repoPath: string, - relPaths: readonly string[], -): Promise> { - const out = new Map(); - const BATCH = 100; - for (let i = 0; i < relPaths.length; i += BATCH) { - const batch = relPaths.slice(i, i + BATCH); - const results = await Promise.all( - batch.map(async (rel) => { - const hash = await computeFileHash(path.join(repoPath, rel)); - return hash ? ([rel, hash] as const) : null; - }), - ); - for (const r of results) if (r) out.set(r[0], r[1]); - } - return out; -} diff --git a/gitnexus/src/core/incremental/git-diff.ts b/gitnexus/src/core/incremental/git-diff.ts deleted file mode 100644 index 04a7c75de..000000000 --- a/gitnexus/src/core/incremental/git-diff.ts +++ /dev/null @@ -1,211 +0,0 @@ -/** - * Git-based change detection for incremental indexing. - * - * Combines `git diff --name-status HEAD` (committed changes since - * last index) with `git status --porcelain` (uncommitted/dirty tree changes) - * to produce a precise per-file change set. - * - * Renames (R in diff output, R in porcelain) are flattened into - * delete(old) + add(new) — downstream consumers don't need to know about - * renames specifically; they just need to know "this path's nodes are stale, - * delete them" and "this path is new, parse it". - * - * Returns a typed result; throws `LastCommitMissingError` when `lastCommit` - * no longer exists in the repository (rebase or shallow clone). Callers - * should fall back to a full rebuild on this error. - */ - -import { execFileSync } from 'child_process'; - -export interface ChangedFiles { - /** Files whose content changed (re-parse needed). */ - modified: string[]; - /** Files newly introduced since lastCommit (re-parse needed). */ - added: string[]; - /** Files that existed at lastCommit but no longer do (delete-only, no parse). */ - deleted: string[]; -} - -export class LastCommitMissingError extends Error { - constructor(commit: string) { - super(`lastCommit ${commit} not found in repository — cannot compute incremental diff`); - this.name = 'LastCommitMissingError'; - } -} - -/** - * Compute the set of files changed in `repoPath` since `lastCommit`. - * Combines committed differences (`git diff` against HEAD) with the - * current dirty working-tree state (`git status`). - * - * @throws LastCommitMissingError when `lastCommit` is not reachable. - */ -export function getChangedFilesSinceCommit( - repoPath: string, - lastCommit: string, -): ChangedFiles { - if (!commitExists(repoPath, lastCommit)) { - throw new LastCommitMissingError(lastCommit); - } - - const modified = new Set(); - const added = new Set(); - const deleted = new Set(); - - // ── Committed differences: lastCommit..HEAD ──────────────────────────── - // Output (NUL-delimited via -z): STATUS\0path1\0[path2\0] - // Status codes: M, A, D, T (type change → treat as modified), - // R, C (rename/copy with similarity %). - const diffRaw = runGit(repoPath, [ - 'diff', - '--name-status', - '-z', - `${lastCommit}`, - 'HEAD', - ]); - - for (const entry of parseNameStatusZ(diffRaw)) { - classify(entry, modified, added, deleted); - } - - // ── Working-tree differences: HEAD..disk ─────────────────────────────── - // Format (NUL-delimited): XYpath[\0orig-path] - // X = staged, Y = unstaged. Either non-' ' counts as a change. - const statusRaw = runGit(repoPath, ['status', '--porcelain', '-z']); - for (const entry of parsePorcelainZ(statusRaw)) { - classify(entry, modified, added, deleted); - } - - // Resolve overlaps: a file added in the diff and modified in the status - // is just "new on disk" → keep it in `added`. A file deleted in the diff - // but present in status as added (re-introduced) → modified. - for (const f of added) { - modified.delete(f); - deleted.delete(f); - } - for (const f of deleted) { - modified.delete(f); - } - - return { - modified: [...modified].sort(), - added: [...added].sort(), - deleted: [...deleted].sort(), - }; -} - -interface ParsedEntry { - status: string; - path: string; - origPath?: string; -} - -function classify( - entry: ParsedEntry, - modified: Set, - added: Set, - deleted: Set, -): void { - const code = entry.status[0]; - switch (code) { - case 'M': - case 'T': // type change (file → symlink, etc.) - modified.add(entry.path); - break; - case 'A': - case '?': // untracked (porcelain '??') treat as added - added.add(entry.path); - break; - case 'D': - deleted.add(entry.path); - break; - case 'R': // rename: flatten to delete(orig) + add(new) - case 'C': // copy: original survives; new file is added - if (entry.origPath && code === 'R') deleted.add(entry.origPath); - added.add(entry.path); - break; - case 'U': // unmerged — surface as modified, caller's problem - modified.add(entry.path); - break; - default: - // Unknown status — be conservative, treat as modified. - modified.add(entry.path); - } -} - -/** - * Parse `git diff --name-status -z` output. NUL-delimited, status before each path: - * "M\0a.ts\0A\0b.ts\0R100\0old.ts\0new.ts\0" - */ -function parseNameStatusZ(raw: string): ParsedEntry[] { - const entries: ParsedEntry[] = []; - if (!raw) return entries; - const tokens = raw.split('\0').filter((t) => t.length > 0); - let i = 0; - while (i < tokens.length) { - const status = tokens[i++]; - const path = tokens[i++]; - if (path === undefined) break; - if (status[0] === 'R' || status[0] === 'C') { - const newPath = tokens[i++]; - if (newPath !== undefined) { - entries.push({ status, path: newPath, origPath: path }); - } - } else { - entries.push({ status, path }); - } - } - return entries; -} - -/** - * Parse `git status --porcelain -z` output. NUL-delimited, two-char status - * code + space + path; for renames the original path follows after another NUL: - * "M a.ts\0?? b.ts\0R new.ts\0old.ts\0" - */ -function parsePorcelainZ(raw: string): ParsedEntry[] { - const entries: ParsedEntry[] = []; - if (!raw) return entries; - const tokens = raw.split('\0').filter((t) => t.length > 0); - let i = 0; - while (i < tokens.length) { - const tok = tokens[i++]; - // First two chars are the XY status, then a space, then the path. - const xy = tok.slice(0, 2); - const path = tok.slice(3); - // Effective status: prefer staged (X) when not ' ', otherwise unstaged (Y). - const code = xy[0] !== ' ' && xy[0] !== '?' ? xy[0] : xy[1]; - if (xy.startsWith('R') || xy[1] === 'R') { - // Rename: next token is the original path. - const orig = tokens[i++]; - entries.push({ status: 'R', path, origPath: orig }); - } else if (xy === '??') { - entries.push({ status: '?', path }); - } else { - entries.push({ status: code, path }); - } - } - return entries; -} - -function commitExists(repoPath: string, commit: string): boolean { - try { - execFileSync('git', ['cat-file', '-e', `${commit}^{commit}`], { - cwd: repoPath, - stdio: ['ignore', 'ignore', 'ignore'], - }); - return true; - } catch { - return false; - } -} - -function runGit(repoPath: string, args: string[]): string { - return execFileSync('git', args, { - cwd: repoPath, - stdio: ['ignore', 'pipe', 'ignore'], - encoding: 'utf8', - // Larger buffer for large repos with many changes. - maxBuffer: 100 * 1024 * 1024, - }); -} diff --git a/gitnexus/src/core/incremental/orchestrator.ts b/gitnexus/src/core/incremental/orchestrator.ts deleted file mode 100644 index 69324da3a..000000000 --- a/gitnexus/src/core/incremental/orchestrator.ts +++ /dev/null @@ -1,267 +0,0 @@ -/** - * Incremental-indexing orchestrator helpers. - * - * High-level helpers used by `run-analyze.ts` to keep the incremental path - * out of the main full-rebuild orchestrator function. Each helper has one - * responsibility: - * - * isIncrementalEligible(...) — should this run go incremental? - * computeIncrementalClosure(...) — git diff → content-hash → closure - * extractIncrementalSubgraph(...) — ctx.graph → "nodes/edges to write" - * commitIncrementalProgress(...) — set the dirty flag in meta.json - * - * See docs/superpowers/specs/2026-05-10-incremental-indexing-design.md. - */ - -import { execFileSync } from 'child_process'; -import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; -import { createKnowledgeGraph } from '../graph/graph.js'; -import type { KnowledgeGraph } from '../graph/types.js'; -import { - type RepoMeta, - INCREMENTAL_SCHEMA_VERSION, - saveMeta, -} from '../../storage/repo-manager.js'; -import { hasGitDir } from '../../storage/git.js'; -import { - getChangedFilesSinceCommit, - type ChangedFiles, -} from './git-diff.js'; -import { computeImporterClosure } from './closure.js'; -import { computeFileHashes } from './file-hash.js'; -import { queryImporters } from '../lbug/lbug-adapter.js'; - -/** Decision returned by isIncrementalEligible. */ -export interface IncrementalEligibility { - /** True iff this run should attempt the incremental path. */ - eligible: boolean; - /** When `eligible === false`, a short reason for logging. */ - reason?: string; - /** The previous lastCommit, when eligible. */ - lastCommit?: string; -} - -/** - * Decide whether a run is eligible for the incremental path. - * - * All conditions must hold: - * - --force not passed - * - existing meta.json present and previously a full rebuild has populated - * surfaceSignatures + schemaVersion (matching CURRENT) - * - repo has .git (non-git repos always do full rebuild) - * - meta.lastCommit still resolvable (not rebased away) - * - no incrementalInProgress flag (would force full rebuild for safety) - */ -export function isIncrementalEligible( - repoPath: string, - existingMeta: RepoMeta | null | undefined, - optionsForce: boolean | undefined, -): IncrementalEligibility { - if (optionsForce) { - return { eligible: false, reason: '--force passed' }; - } - if (!existingMeta) { - return { eligible: false, reason: 'no existing index' }; - } - if (existingMeta.incrementalInProgress) { - return { - eligible: false, - reason: 'previous incremental run did not complete cleanly', - }; - } - if ( - existingMeta.schemaVersion === undefined || - existingMeta.schemaVersion !== INCREMENTAL_SCHEMA_VERSION - ) { - return { - eligible: false, - reason: `schemaVersion mismatch (have ${existingMeta.schemaVersion}, want ${INCREMENTAL_SCHEMA_VERSION})`, - }; - } - if (!existingMeta.surfaceSignatures || Object.keys(existingMeta.surfaceSignatures).length === 0) { - return { - eligible: false, - reason: 'no prior surfaceSignatures in meta.json', - }; - } - if (!hasGitDir(repoPath)) { - return { eligible: false, reason: 'non-git repo' }; - } - if (!existingMeta.lastCommit) { - return { eligible: false, reason: 'no lastCommit recorded' }; - } - if (!commitExists(repoPath, existingMeta.lastCommit)) { - return { - eligible: false, - reason: `lastCommit ${existingMeta.lastCommit.slice(0, 7)} not in repo`, - }; - } - return { eligible: true, lastCommit: existingMeta.lastCommit }; -} - -function commitExists(repoPath: string, commit: string): boolean { - try { - execFileSync('git', ['cat-file', '-e', `${commit}^{commit}`], { - cwd: repoPath, - stdio: ['ignore', 'ignore', 'ignore'], - }); - return true; - } catch { - return false; - } -} - -export interface IncrementalSetupResult { - /** Files that must be re-parsed in this run. */ - closure: Set; - /** Files deleted on disk since lastCommit (rows must be removed from DB). */ - deletedFiles: string[]; - /** Per-file content hashes computed during closure expansion. */ - newFileHashes: Map; - /** ChangedFiles result for diagnostics. */ - changes: ChangedFiles; -} - -/** - * Compute the closure of files to re-parse for an incremental run. - * - * Algorithm: - * 1. git diff lastCommit HEAD ∪ git status → ChangedFiles - * 2. closure ← modified ∪ added - * 3. For each f in closure: hash(f); if hash differs from previous, query - * DB importers and add them to closure. Iterate to fixpoint. - * - * Throws if `lastCommit` is gone (caller should fall back to full rebuild). - */ -export async function computeIncrementalClosure( - repoPath: string, - lastCommit: string, - prevSurfaces: Record, -): Promise { - const changes = getChangedFilesSinceCommit(repoPath, lastCommit); - - const initialChangedFiles = new Set([...changes.modified, ...changes.added]); - - // Closure module wants generic parseFile + surfaceFor. v1 uses - // file-content hashes (cheap, conservative). The "ParseResult" type - // is just the content hash itself. - const { closure, newSurfaces } = await computeImporterClosure({ - initialChangedFiles, - prevSurfaces, - parseFile: async (filePath) => { - const hashes = await computeFileHashes(repoPath, [filePath]); - // Missing files (race with delete) → empty hash; counts as "changed". - return hashes.get(filePath) ?? ''; - }, - surfaceFor: (_filePath, parsed) => parsed, - queryImporters: async (filePath) => queryImporters(filePath), - }); - - return { - closure, - deletedFiles: changes.deleted, - newFileHashes: newSurfaces, - changes, - }; -} - -/** - * Persist the `incrementalInProgress` dirty flag to meta.json BEFORE any - * destructive DB mutation. The flag is cleared on success by overwriting - * meta.json with the final state. If the run crashes between, the next - * run sees the flag and forces a full rebuild. - */ -export async function commitIncrementalProgress( - storagePath: string, - existingMeta: RepoMeta, - closure: Set, -): Promise { - const meta: RepoMeta = { - ...existingMeta, - incrementalInProgress: { - closure: [...closure], - startedAt: Date.now(), - }, - }; - await saveMeta(storagePath, meta); -} - -/** - * Build a subgraph of `ctx.graph` containing ONLY the nodes/edges that - * need to be written to LadybugDB in incremental mode: - * - * - All nodes whose filePath is in `closure` (newly parsed in this run). - * - All graph-wide nodes (Community, Process) — they're regenerated by - * the communities/processes phases on every run. - * - All edges where AT LEAST ONE endpoint is in this set. Edges entirely - * between hydrated unchanged-file nodes are NOT included — they're - * already in the DB and re-inserting them would PK-conflict. - * - * This lets us call the existing `loadGraphToLbug` against the filtered - * subgraph: the COPY semantics will write only what's missing, while the - * unchanged-unchanged DB rows we never deleted stay intact. - */ -export function extractIncrementalSubgraph( - fullGraph: KnowledgeGraph, - closure: ReadonlySet, -): KnowledgeGraph { - const sub = createKnowledgeGraph(); - - const isGraphWide = (label: string): boolean => label === 'Community' || label === 'Process'; - - // Phase 1: nodes - const writableNodeIds = new Set(); - fullGraph.forEachNode((n: GraphNode) => { - const filePath = n.properties?.filePath as string | undefined; - const inClosure = filePath ? closure.has(filePath) : false; - if (inClosure || isGraphWide(n.label)) { - sub.addNode(n); - writableNodeIds.add(n.id); - } - }); - - // Phase 2: edges where AT LEAST ONE endpoint is in the writable set. - // Edges entirely between hydrated nodes are skipped (already in DB). - fullGraph.forEachRelationship((r: GraphRelationship) => { - if (writableNodeIds.has(r.sourceId) || writableNodeIds.has(r.targetId)) { - sub.addRelationship(r); - } - }); - - return sub; -} - -/** - * Compute the surface signatures for every file in the repo from a graph. - * In v1 this is just the per-file content hash (cheap, conservative). v2 - * will switch to true surface-only signatures derived via `surface.ts`. - * - * Used after a full rebuild to populate `meta.json.surfaceSignatures` so - * the next run can run incrementally. - */ -export async function computeAllFileSignatures( - repoPath: string, - filePaths: readonly string[], -): Promise> { - const map = await computeFileHashes(repoPath, filePaths); - const out: Record = {}; - for (const [k, v] of map) out[k] = v; - return out; -} - -/** - * Merge an updated set of file hashes into a previous snapshot. Used after - * an incremental run: `prevSurfaces` is the set in meta.json, `newSurfaces` - * is what we computed for closure files this run, and `deletedFiles` are - * files that no longer exist on disk. - */ -export function mergeSurfaceSignatures( - prevSurfaces: Record, - newSurfaces: Map, - deletedFiles: readonly string[], -): Record { - const merged: Record = { ...prevSurfaces }; - for (const f of deletedFiles) delete merged[f]; - for (const [f, h] of newSurfaces) merged[f] = h; - return merged; -} diff --git a/gitnexus/src/core/incremental/surface.ts b/gitnexus/src/core/incremental/surface.ts deleted file mode 100644 index 7cf3e8eb9..000000000 --- a/gitnexus/src/core/incremental/surface.ts +++ /dev/null @@ -1,135 +0,0 @@ -/** - * Public-surface signature extraction for incremental indexing. - * - * The surface signature of a file is a stable hash of everything that could - * affect callers in *other* files: the names, signatures, and heritage of - * its publicly-visible symbols (functions, classes, methods, interfaces). - * - * If a file's surface hash is unchanged between runs, no other file's - * resolution depends on the changed content — we only need to re-parse the - * file itself, not its importers. This is the closure-scoping optimization - * driven by `closure.ts`. - * - * The hash is content-only and excludes formatting, comments, and ordering - * (we sort the symbol list before hashing). It does NOT include implementation - * bodies — that's the whole point: a body-only edit produces the same surface. - */ - -import { createHash } from 'crypto'; -import type { GraphNode } from 'gitnexus-shared'; -import type { KnowledgeGraph } from '../graph/types.js'; - -/** Node labels whose presence in a file contributes to its public surface. */ -const SURFACE_LABELS = new Set([ - 'Function', - 'Class', - 'Method', - 'Constructor', - 'Interface', - 'TypeAlias', - 'Enum', - 'Struct', - 'Trait', - 'Namespace', - 'Module', - 'Const', - 'Static', - 'Property', - 'Record', - 'Delegate', - 'Annotation', - 'Template', - 'Union', - 'Macro', - 'Typedef', -]); - -/** - * Extract the surface signature of `filePath` from `graph`. - * Returns a stable hex hash; identical-surface inputs → identical output. - */ -export function extractSurfaceSignature( - graph: KnowledgeGraph, - filePath: string, -): string { - const lines: string[] = []; - - // Gather surface-relevant nodes for this file. - const surfaceNodes: GraphNode[] = []; - graph.forEachNode((node) => { - if ( - node.properties?.filePath === filePath && - SURFACE_LABELS.has(node.label) - ) { - surfaceNodes.push(node); - } - }); - - // Sort deterministically by id so reorder doesn't affect the hash. - surfaceNodes.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0)); - - for (const n of surfaceNodes) { - lines.push(serializeNode(n)); - } - - // Heritage: edges that fan OUT of this file affect downstream resolution. - // (EXTENDS / IMPLEMENTS targets — if the parent class changes, dependents - // may resolve methods differently.) These edges are co-located with the - // child symbol whose `filePath` matches; we walk them off the source side. - const edgeKeys: string[] = []; - graph.forEachRelationship((rel) => { - if (rel.type !== 'EXTENDS' && rel.type !== 'IMPLEMENTS') return; - const src = graph.getNode(rel.sourceId); - if (!src) return; - if (src.properties?.filePath !== filePath) return; - edgeKeys.push(`H ${rel.type} ${rel.sourceId} -> ${rel.targetId}`); - }); - edgeKeys.sort(); - for (const k of edgeKeys) lines.push(k); - - const h = createHash('sha256'); - for (const line of lines) { - h.update(line); - h.update('\n'); - } - return h.digest('hex'); -} - -/** True iff `current` differs from `prev` (or `prev` is undefined). */ -export function surfaceChanged( - prev: string | undefined, - current: string, -): boolean { - return prev === undefined || prev !== current; -} - -/** - * Serialize a single surface node to a stable string. We include label, - * name, parameter count + types, return type, and visibility-style flags - * — anything that could affect how a caller resolves against this node. - * - * We DO NOT include line numbers, file offsets, or any other position - * data: surface should be invariant under whitespace/formatting changes. - */ -function serializeNode(node: GraphNode): string { - const p = (node.properties ?? {}) as Record; - const parts: string[] = [ - `N`, - node.label, - String(p.name ?? ''), - `id=${node.id}`, - ]; - if (p.parameterCount !== undefined) parts.push(`pc=${String(p.parameterCount)}`); - if (Array.isArray(p.parameterTypes)) { - parts.push(`pt=${(p.parameterTypes as string[]).join(',')}`); - } - if (p.returnType !== undefined) parts.push(`rt=${String(p.returnType)}`); - if (p.visibility !== undefined) parts.push(`v=${String(p.visibility)}`); - if (p.isStatic) parts.push('static'); - if (p.isAbstract) parts.push('abstract'); - if (p.isReadonly) parts.push('readonly'); - if (p.isAsync) parts.push('async'); - // `level` (inheritance depth marker for methods) affects MRO - if (p.level !== undefined) parts.push(`lvl=${String(p.level)}`); - return parts.join('|'); -} diff --git a/gitnexus/src/core/ingestion/pipeline-phases/hydrate.ts b/gitnexus/src/core/ingestion/pipeline-phases/hydrate.ts deleted file mode 100644 index a0c00c2ff..000000000 --- a/gitnexus/src/core/ingestion/pipeline-phases/hydrate.ts +++ /dev/null @@ -1,91 +0,0 @@ -/** - * Phase: hydrate - * - * In incremental-indexing mode, populates `ctx.graph` with all nodes and - * relationships belonging to files OUTSIDE the current closure (i.e., files - * that didn't change). This way the parse phase only needs to re-emit nodes - * and edges for the closure files, while downstream phases (mro, communities, - * processes) see a fully-populated graph and produce results equivalent to a - * full rebuild. - * - * No-op in full-rebuild mode: when `ctx.options.filesToParse` is unset, - * the pipeline starts with an empty graph and parses every file (existing - * behavior). - * - * @deps structure - * @reads allPaths (from structure) - * @writes graph (every node/edge belonging to unchanged files) - */ - -import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js'; -import { getPhaseOutput } from './types.js'; -import type { StructureOutput } from './structure.js'; -import { loadGraphFromLbug } from '../../lbug/lbug-adapter.js'; -import { isDev } from '../utils/env.js'; -import { logger } from '../../logger.js'; - -export interface HydrateOutput { - /** True when this run actually loaded prior state (incremental). */ - readonly hydrated: boolean; - /** Number of nodes loaded from DB (0 in full-rebuild mode). */ - readonly nodesLoaded: number; - /** Number of relationships loaded from DB (0 in full-rebuild mode). */ - readonly edgesLoaded: number; -} - -export const hydratePhase: PipelinePhase = { - name: 'hydrate', - deps: ['structure'], - - async execute( - ctx: PipelineContext, - deps: ReadonlyMap>, - ): Promise { - const filesToParse = ctx.options?.filesToParse; - - // Full-rebuild mode: nothing to hydrate. - if (!filesToParse) { - return { hydrated: false, nodesLoaded: 0, edgesLoaded: 0 }; - } - - const { allPaths, totalFiles } = getPhaseOutput(deps, 'structure'); - - // Compute the unchanged complement: every scanned path NOT in the closure. - const unchanged = new Set(); - for (const p of allPaths) { - if (!filesToParse.has(p)) unchanged.add(p); - } - - ctx.onProgress({ - phase: 'hydrate', - percent: 22, - message: `Hydrating ${unchanged.size} unchanged files from index...`, - stats: { filesProcessed: 0, totalFiles, nodesCreated: ctx.graph.nodeCount }, - }); - - const result = await loadGraphFromLbug(ctx.graph, unchanged); - - if (isDev) { - logger.info( - `💧 Hydrate: ${result.nodesLoaded} nodes, ${result.edgesLoaded} edges loaded for ${unchanged.size} unchanged files (closure: ${filesToParse.size})`, - ); - } - - ctx.onProgress({ - phase: 'hydrate', - percent: 25, - message: `Hydrated ${result.nodesLoaded} nodes from previous index`, - stats: { - filesProcessed: unchanged.size, - totalFiles, - nodesCreated: ctx.graph.nodeCount, - }, - }); - - return { - hydrated: true, - nodesLoaded: result.nodesLoaded, - edgesLoaded: result.edgesLoaded, - }; - }, -}; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/index.ts b/gitnexus/src/core/ingestion/pipeline-phases/index.ts index 908d6c0a7..b1dcf9082 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/index.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/index.ts @@ -9,7 +9,6 @@ export { scanPhase, type ScanOutput } from './scan.js'; export { structurePhase, type StructureOutput } from './structure.js'; -export { hydratePhase, type HydrateOutput } from './hydrate.js'; export { markdownPhase, type MarkdownOutput } from './markdown.js'; export { cobolPhase, type CobolOutput } from './cobol.js'; export { parsePhase, type ParseOutput } from './parse.js'; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse.ts index d0217d033..a20d1e4b0 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse.ts @@ -96,18 +96,9 @@ export const parsePhase: PipelinePhase = { 'structure', ); - // Incremental-indexing filter: when `filesToParse` is set, only the - // closure files get parsed in this run. Files outside the closure had - // their nodes/edges pre-loaded by the `hydrate` phase. See - // docs/superpowers/specs/2026-05-10-incremental-indexing-design.md. - const filesToParse = ctx.options?.filesToParse; - const targetScanned = filesToParse - ? scannedFiles.filter((f) => filesToParse.has(f.path)) - : scannedFiles; - const result = await runChunkedParseAndResolve( ctx.graph, - targetScanned, + scannedFiles, allPaths, totalFiles, ctx.repoPath, diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index e93444352..c220ea224 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -23,7 +23,6 @@ import { getPhaseOutput, scanPhase, structurePhase, - hydratePhase, markdownPhase, cobolPhase, parsePhase, @@ -56,18 +55,6 @@ export interface PipelineOptions { minFiles?: number; minBytes?: number; }; - /** - * Incremental-indexing mode: when set, the parse phase only re-parses - * files in this set, and the new `hydrate` phase pre-loads node/edge - * state for everything else from the existing LadybugDB index. - * - * Unset (the default) → full-rebuild mode: parse phase processes every - * scanned file and hydrate is a no-op. Set by `runFullAnalysis` when it - * detects an eligible incremental run; never set by callers directly. - * - * See `docs/superpowers/specs/2026-05-10-incremental-indexing-design.md`. - */ - filesToParse?: ReadonlySet; } // ── Phase registry ───────────────────────────────────────────────────────── @@ -87,7 +74,6 @@ function buildPhaseList(options?: PipelineOptions): PipelinePhase[] { const phases: PipelinePhase[] = [ scanPhase, structurePhase, - hydratePhase, markdownPhase, cobolPhase, parsePhase, diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index f47d05e1b..cb88986ca 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -1204,245 +1204,6 @@ export const deleteNodesForFile = async ( export const getEmbeddingTableName = (): string => EMBEDDING_TABLE_NAME; -// ============================================================================ -// Incremental indexing: DB → KnowledgeGraph hydration -// ============================================================================ - -/** - * Node tables that have a `filePath` column and are eligible for hydration. - * `Community` and `Process` are graph-wide (no filePath) and are ALWAYS - * regenerated by the communities/processes phases — we never load them back. - */ -const HYDRATABLE_NODE_TABLES: readonly NodeTableName[] = NODE_TABLES.filter( - (t) => t !== 'Community' && t !== 'Process', -); - -/** Per-table extra columns to project beyond the base (id, name, filePath). */ -const TABLE_EXTRA_COLUMNS: Record = { - File: ['content'], - Folder: [], - Function: ['startLine', 'endLine', 'isExported', 'content', 'description'], - Class: ['startLine', 'endLine', 'isExported', 'content', 'description'], - Interface: ['startLine', 'endLine', 'isExported', 'content', 'description'], - Method: [ - 'startLine', - 'endLine', - 'isExported', - 'content', - 'description', - 'parameterCount', - 'returnType', - ], - CodeElement: ['startLine', 'endLine', 'isExported', 'content', 'description'], - Section: ['startLine', 'endLine', 'level', 'content', 'description'], - Property: ['startLine', 'endLine', 'content', 'description', 'declaredType'], - Route: ['responseKeys', 'errorKeys', 'middleware'], - Tool: ['description'], -}; - -/** All other tables share the CODE_ELEMENT_BASE schema (startLine/endLine/content/description). */ -const DEFAULT_EXTRA_COLUMNS = ['startLine', 'endLine', 'content', 'description']; - -const escapeFilePath = (s: string): string => - s.replace(/\\/g, '\\\\').replace(/'/g, "''"); - -/** - * Hydrate `graph` with all nodes (and their relationships) belonging to files - * in `unchangedFilePaths`. Used by the incremental-indexing pipeline to - * pre-populate the in-memory graph with everything that didn't change, so - * downstream phases (mro, communities, processes) see a complete picture. - * - * Skips Community/Process labels and their MEMBER_OF / STEP_IN_PROCESS edges - * — those are graph-wide and will be regenerated from scratch by the - * pipeline's downstream phases. - * - * Returns counts for logging/diagnostics. - */ -export const loadGraphFromLbug = async ( - graph: KnowledgeGraph, - unchangedFilePaths: ReadonlySet, -): Promise<{ nodesLoaded: number; edgesLoaded: number }> => { - if (!conn) { - throw new Error('LadybugDB not initialized. Call initLbug first.'); - } - if (unchangedFilePaths.size === 0) { - return { nodesLoaded: 0, edgesLoaded: 0 }; - } - - let nodesLoaded = 0; - let edgesLoaded = 0; - - // Track which node IDs we successfully loaded so the relationship pass - // can verify both endpoints are present (cheaper than a DB-side join). - const loadedNodeIds = new Set(); - - // ── 1. Hydrate nodes per table ───────────────────────────────────────── - // Chunk filePaths to keep query size manageable on huge repos. - const CHUNK = 200; - const filePathArr = [...unchangedFilePaths]; - - for (const tableName of HYDRATABLE_NODE_TABLES) { - const t = escapeTableName(tableName); - const extras = TABLE_EXTRA_COLUMNS[tableName] ?? DEFAULT_EXTRA_COLUMNS; - // Always include base columns: id, name, filePath - const cols = ['id', 'name', 'filePath', ...extras]; - const projection = cols.map((c) => `n.${c} AS ${c}`).join(', '); - - for (let i = 0; i < filePathArr.length; i += CHUNK) { - const batch = filePathArr.slice(i, i + CHUNK); - const inList = batch.map((p) => `'${escapeFilePath(p)}'`).join(','); - const cypher = `MATCH (n:${t}) WHERE n.filePath IN [${inList}] RETURN ${projection}`; - - try { - const queryResult = await conn.query(cypher); - const result = Array.isArray(queryResult) ? queryResult[0] : queryResult; - const rows = await result.getAll(); - for (const row of rows) { - const id = typeof row.id === 'string' ? row.id : String(row.id ?? ''); - if (!id) continue; - const properties: Record = {}; - for (const c of cols) { - const v = row[c]; - if (v !== undefined && v !== null) properties[c] = v; - } - // Required for downstream phases: ensure name/filePath always set. - if (properties.name === undefined) properties.name = ''; - graph.addNode({ - id, - label: tableName as unknown as import('gitnexus-shared').NodeLabel, - properties: properties as import('gitnexus-shared').NodeProperties, - }); - loadedNodeIds.add(id); - nodesLoaded++; - } - } catch { - // Some tables may not exist in the schema or may be empty — that's fine. - } - } - } - - // ── 2. Hydrate relationships ─────────────────────────────────────────── - // We pull all CodeRelation rows whose endpoints are nodes we just loaded. - // Using SRC.filePath IN [...] AND TGT.filePath IN [...] is precise but - // requires LadybugDB to traverse with both endpoint constraints. We - // exclude graph-wide edge types (MEMBER_OF, STEP_IN_PROCESS) entirely — - // they'll be regenerated. - // - // Strategy: for each (src filePath chunk × tgt filePath chunk) we'd risk - // O(n²) chunking. Instead, we fetch by source-chunk, then verify the - // target endpoint is in `loadedNodeIds` JS-side. This keeps the DB query - // bounded by source chunks while preserving correctness. - for (let i = 0; i < filePathArr.length; i += CHUNK) { - const batch = filePathArr.slice(i, i + CHUNK); - const inList = batch.map((p) => `'${escapeFilePath(p)}'`).join(','); - const cypher = ` - MATCH (a)-[r:${REL_TABLE_NAME}]->(b) - WHERE a.filePath IN [${inList}] - AND r.type <> 'MEMBER_OF' - AND r.type <> 'STEP_IN_PROCESS' - RETURN a.id AS src, b.id AS dst, r.type AS type, - r.confidence AS confidence, r.reason AS reason, r.step AS step - `; - - try { - const queryResult = await conn.query(cypher); - const result = Array.isArray(queryResult) ? queryResult[0] : queryResult; - const rows = await result.getAll(); - for (const row of rows) { - const src = typeof row.src === 'string' ? row.src : ''; - const dst = typeof row.dst === 'string' ? row.dst : ''; - if (!src || !dst) continue; - // The other endpoint must be a node we loaded — otherwise the edge - // crosses into a closure file (will be re-emitted) or a dropped - // graph-wide node (Community/Process), and we skip it. - if (!loadedNodeIds.has(dst)) continue; - const relType = typeof row.type === 'string' ? row.type : ''; - if (!relType) continue; - const relId = `${src}_${relType}_${dst}`; - graph.addRelationship({ - id: relId, - sourceId: src, - targetId: dst, - type: relType as import('gitnexus-shared').RelationshipType, - confidence: typeof row.confidence === 'number' ? row.confidence : 1.0, - reason: typeof row.reason === 'string' ? row.reason : '', - step: - typeof row.step === 'number' && row.step !== 0 ? row.step : undefined, - }); - edgesLoaded++; - } - } catch { - // Continue on chunk failure — best-effort hydration. - } - } - - return { nodesLoaded, edgesLoaded }; -}; - -/** - * Query the IMPORTS edge table for all files that import `targetFilePath`. - * Used by the incremental-indexing closure-expansion logic to find files - * whose resolution may be stale when `targetFilePath`'s public surface - * changes. - * - * Returns repo-relative paths (i.e. the source file's `filePath`), de-duplicated. - */ -export const queryImporters = async (targetFilePath: string): Promise => { - if (!conn) { - throw new Error('LadybugDB not initialized. Call initLbug first.'); - } - const escaped = escapeFilePath(targetFilePath); - const cypher = ` - MATCH (a)-[r:${REL_TABLE_NAME}]->(b) - WHERE r.type = 'IMPORTS' AND b.filePath = '${escaped}' - RETURN DISTINCT a.filePath AS importer - `; - try { - const queryResult = await conn.query(cypher); - const result = Array.isArray(queryResult) ? queryResult[0] : queryResult; - const rows = await result.getAll(); - const out: string[] = []; - for (const row of rows) { - const p = row.importer; - if (typeof p === 'string' && p.length > 0) out.push(p); - } - return out; - } catch { - return []; - } -}; - -/** - * Delete all Community and Process nodes and their MEMBER_OF / - * STEP_IN_PROCESS edges. Used at the start of an incremental run so the - * communities/processes phases can regenerate them on the merged graph. - */ -export const deleteAllCommunitiesAndProcesses = async (): Promise<{ - nodesDeleted: number; -}> => { - if (!conn) { - throw new Error('LadybugDB not initialized. Call initLbug first.'); - } - let nodesDeleted = 0; - for (const label of ['Community', 'Process']) { - try { - const countResult = await conn.query( - `MATCH (n:${label}) RETURN count(n) AS cnt`, - ); - const result = Array.isArray(countResult) ? countResult[0] : countResult; - const rows = await result.getAll(); - const count = Number(rows[0]?.cnt ?? rows[0]?.[0] ?? 0); - if (count > 0) { - await conn.query(`MATCH (n:${label}) DETACH DELETE n`); - nodesDeleted += count; - } - } catch { - // table may not exist yet - } - } - return { nodesDeleted }; -}; - // ============================================================================ // Full-Text Search (FTS) Functions // ============================================================================ diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index b15394019..7227fd945 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -11,7 +11,6 @@ import path from 'path'; import fs from 'fs/promises'; -import { execFileSync } from 'child_process'; import { runPipelineFromRepo } from './ingestion/pipeline.js'; import { initLbug, @@ -21,8 +20,6 @@ import { executeWithReusedStatement, closeLbug, loadCachedEmbeddings, - deleteNodesForFile, - deleteAllCommunitiesAndProcesses, } from './lbug/lbug-adapter.js'; import { createSearchFTSIndexes } from './search/fts-indexes.js'; import { @@ -32,7 +29,6 @@ import { ensureGitNexusIgnored, registerRepo, cleanupOldKuzuFiles, - INCREMENTAL_SCHEMA_VERSION, } from '../storage/repo-manager.js'; import { getCurrentCommit, @@ -45,14 +41,6 @@ import type { CachedEmbedding } from './embeddings/types.js'; import { generateAIContextFiles } from '../cli/ai-context.js'; import { EMBEDDING_TABLE_NAME } from './lbug/schema.js'; import { STALE_HASH_SENTINEL } from './lbug/schema.js'; -import { - isIncrementalEligible, - computeIncrementalClosure, - commitIncrementalProgress, - extractIncrementalSubgraph, - mergeSurfaceSignatures, - computeAllFileSignatures, -} from './incremental/orchestrator.js'; // --------------------------------------------------------------------------- // Public types @@ -192,65 +180,20 @@ export async function runFullAnalysis( if (existingMeta && !options.force && existingMeta.lastCommit === currentCommit) { // Non-git folders have currentCommit = '' — always rebuild since we can't detect changes if (currentCommit !== '') { - // For git repos, even if HEAD matches lastCommit, the working tree - // may have uncommitted changes. Only short-circuit when the working - // tree is also clean. - const dirty = hasDirtyTree(repoPath); - if (!dirty) { - await ensureGitNexusIgnored(repoPath); - return { - // `resolveRepoIdentityRoot` collapses worktree roots to the - // canonical repo basename (#1259) but leaves arbitrary subdirs - // and `--skip-git` paths unchanged (#1232/#1233 intent preserved). - repoName: - options.registryName ?? - getInferredRepoName(repoPath) ?? - path.basename(resolveRepoIdentityRoot(repoPath)), - repoPath, - stats: existingMeta.stats ?? {}, - alreadyUpToDate: true, - }; - } - } - } - - // ── Incremental branch ───────────────────────────────────────────── - // Try the incremental path first. Falls through to full rebuild on: - // - --force - // - missing/old meta.json - // - lastCommit gone (rebased) - // - schemaVersion mismatch - // - non-git repo - // - any error during incremental setup - const eligibility = isIncrementalEligible(repoPath, existingMeta, options.force); - if (eligibility.eligible && existingMeta) { - try { - const result = await runIncrementalBranch( + await ensureGitNexusIgnored(repoPath); + return { + // `resolveRepoIdentityRoot` collapses worktree roots to the + // canonical repo basename (#1259) but leaves arbitrary subdirs + // and `--skip-git` paths unchanged (#1232/#1233 intent preserved). + repoName: + options.registryName ?? + getInferredRepoName(repoPath) ?? + path.basename(resolveRepoIdentityRoot(repoPath)), repoPath, - storagePath, - lbugPath, - existingMeta, - currentCommit, - options, - callbacks, - ); - if (result) return result; - // result === null → falls through to full rebuild (e.g. setup failed) - } catch (err) { - log( - `Incremental run failed (${(err as Error).message}); ` + - `next run will full-rebuild via dirty-flag.`, - ); - // Re-throw — the dirty flag is set; next run forces full rebuild. - try { - await closeLbug(); - } catch { - /* swallow */ - } - throw err; + stats: existingMeta.stats ?? {}, + alreadyUpToDate: true, + }; } - } else if (existingMeta) { - log(`Incremental skipped: ${eligibility.reason ?? 'unknown reason'}`); } // ── Cache embeddings from existing index before rebuild ──────────── @@ -511,32 +454,6 @@ export async function runFullAnalysis( const effectiveSemanticMode = semanticMode ?? (runtimeCapabilities.semanticMode === 'vector-index' ? 'vector-index' : 'exact-scan'); - - // Compute per-file surface signatures so the next run can be incremental. - // v1 uses content-hash (cheap, conservative). v2 will switch to a - // surface-only signature so body-only edits don't expand the closure. - let surfaceSignatures: Record | undefined; - if (hasGitDir(repoPath)) { - try { - const allFilePaths: string[] = []; - pipelineResult.graph.forEachNode((n) => { - const fp = n.properties?.filePath as string | undefined; - if (fp && (n.label === 'File' || n.label === 'Folder')) { - // File nodes carry their own path; we want every distinct repo-relative - // file path that participated in indexing. Use File nodes as the - // authoritative source. - if (n.label === 'File') allFilePaths.push(fp); - } - }); - if (allFilePaths.length > 0) { - surfaceSignatures = await computeAllFileSignatures(repoPath, allFilePaths); - } - } catch { - /* surface signatures are best-effort; their absence just means the - * next run will fall back to full rebuild. */ - } - } - const meta = { repoPath, lastCommit: currentCommit, @@ -548,14 +465,6 @@ export async function runFullAnalysis( // origin remote, which is fine: paths-only repos behave as // before. remoteUrl: hasGitDir(repoPath) ? getRemoteUrl(repoPath) : undefined, - // Incremental-indexing fields — populated for git repos only. The - // next analyze run reads these to decide whether to take the - // incremental path. See the Incremental branch above. - schemaVersion: surfaceSignatures ? INCREMENTAL_SCHEMA_VERSION : undefined, - surfaceSignatures, - incrementalInProgress: undefined as - | { closure: string[]; startedAt: number } - | undefined, stats: { files: pipelineResult.totalFileCount, nodes: stats.nodes, @@ -645,242 +554,3 @@ export async function runFullAnalysis( throw err; } } - -// =========================================================================== -// Incremental analysis branch — invoked from runFullAnalysis when eligible. -// See docs/superpowers/specs/2026-05-10-incremental-indexing-design.md. -// =========================================================================== - -/** - * Cheap check: does the working tree have uncommitted changes? Used to - * decide whether the "lastCommit == HEAD" early-exit is safe to take. - */ -function hasDirtyTree(repoPath: string): boolean { - try { - const out = execFileSync('git', ['status', '--porcelain'], { - cwd: repoPath, - stdio: ['ignore', 'pipe', 'ignore'], - encoding: 'utf8', - }); - return out.trim().length > 0; - } catch { - // If git status fails for any reason, conservatively assume dirty so - // we don't accidentally short-circuit a real change. - return true; - } -} - -/** - * Execute the incremental branch of `runFullAnalysis`. Returns null when - * setup determines a full rebuild is needed instead (caller falls through). - * - * Throws if the run fails after the dirty flag is set — caller is - * responsible for surfacing the error; the next run will detect the dirty - * flag and force a full rebuild. - */ -async function runIncrementalBranch( - repoPath: string, - storagePath: string, - lbugPath: string, - existingMeta: import('../storage/repo-manager.js').RepoMeta, - currentCommit: string, - options: AnalyzeOptions, - callbacks: AnalyzeCallbacks, -): Promise { - const log = (msg: string) => callbacks.onLog?.(msg); - const progress = (phase: string, percent: number, message: string) => - callbacks.onProgress(phase, percent, message); - - log( - `Incremental: probing for changes since ${existingMeta.lastCommit.slice(0, 7)}...`, - ); - - // 1. Compute closure (parses each file's content hash, walks DB importers). - let setup: Awaited>; - try { - // Open the existing DB so closure expansion can query the IMPORTS edges. - await initLbug(lbugPath); - setup = await computeIncrementalClosure( - repoPath, - existingMeta.lastCommit, - existingMeta.surfaceSignatures ?? {}, - ); - } catch (e) { - try { - await closeLbug(); - } catch { - /* swallow */ - } - log( - `Incremental setup failed: ${(e as Error).message}. Falling back to full rebuild.`, - ); - return null; - } - - // 2. No changes? Update lastCommit and return early. - if (setup.closure.size === 0 && setup.deletedFiles.length === 0) { - log('Incremental: no file changes detected — refreshing meta only.'); - try { - await closeLbug(); - } catch { - /* swallow */ - } - const meta: import('../storage/repo-manager.js').RepoMeta = { - ...existingMeta, - lastCommit: currentCommit, - indexedAt: new Date().toISOString(), - }; - await saveMeta(storagePath, meta); - await ensureGitNexusIgnored(repoPath); - return { - repoName: - options.registryName ?? - getInferredRepoName(repoPath) ?? - path.basename(resolveRepoIdentityRoot(repoPath)), - repoPath, - stats: existingMeta.stats ?? {}, - alreadyUpToDate: true, - }; - } - - log( - `Incremental: closure=${setup.closure.size} (changed=${setup.changes.modified.length} ` + - `+ added=${setup.changes.added.length} + importers=${setup.closure.size - setup.changes.modified.length - setup.changes.added.length}), ` + - `deleted=${setup.deletedFiles.length}`, - ); - - // 3. Mark dirty BEFORE any DB mutation. Closes lbug temporarily to - // release the connection while saveMeta writes. - await commitIncrementalProgress(storagePath, existingMeta, setup.closure); - - // 4. Delete stale rows from the DB. - progress('lbug', 5, 'Removing stale rows for changed files...'); - for (const file of setup.closure) { - try { - await deleteNodesForFile(file); - } catch { - /* file may not have been indexed yet — fine */ - } - } - for (const file of setup.deletedFiles) { - try { - await deleteNodesForFile(file); - } catch { - /* fine */ - } - } - // Always wipe Community / Process — they're regenerated by downstream - // pipeline phases and must come from the merged graph for correctness - // (Leiden runs on the FULL hydrated + parsed graph). - await deleteAllCommunitiesAndProcesses(); - - // 5. Run the pipeline with filesToParse set so: - // - hydrate phase fills ctx.graph with unchanged-file nodes from DB - // - parse phase only parses closure files - // - downstream phases (mro, communities, processes) see the full graph - const pipelineResult = await runPipelineFromRepo( - repoPath, - (p) => { - const phaseLabel = PHASE_LABELS[p.phase] || p.phase; - const scaled = 10 + Math.round(p.percent * 0.55); // 10–65% - const message = p.detail ? `${p.message || phaseLabel} (${p.detail})` : p.message || phaseLabel; - progress(p.phase, scaled, message); - }, - { filesToParse: setup.closure }, - ); - - // 6. Extract the subgraph that needs to be written: closure-file nodes - // + graph-wide nodes + edges incident to them. Hydrated unchanged-file - // nodes are NOT in this subgraph — their rows are still in DB. - progress('lbug', 70, 'Writing incremental updates to LadybugDB...'); - const subgraph = extractIncrementalSubgraph(pipelineResult.graph, setup.closure); - await loadGraphToLbug(subgraph, repoPath, storagePath, (msg) => { - progress('lbug', 80, msg); - }); - - // 7. Recreate FTS indexes (cheap; full rebuild over the merged DB state). - progress('fts', 90, 'Refreshing search indexes...'); - try { - await createSearchFTSIndexes(); - } catch { - /* FTS is best-effort; log only */ - } - - // 8. Compute final stats from the live DB state. - const stats = await getLbugStats(); - - // 9. Compute new meta (merge surface signatures, clear dirty flag). - const mergedSurfaces = mergeSurfaceSignatures( - existingMeta.surfaceSignatures ?? {}, - setup.newFileHashes, - setup.deletedFiles, - ); - - const newMeta: import('../storage/repo-manager.js').RepoMeta = { - ...existingMeta, - lastCommit: currentCommit, - indexedAt: new Date().toISOString(), - remoteUrl: getRemoteUrl(repoPath) ?? existingMeta.remoteUrl, - schemaVersion: INCREMENTAL_SCHEMA_VERSION, - surfaceSignatures: mergedSurfaces, - incrementalInProgress: undefined, // explicit clear - stats: { - ...(existingMeta.stats ?? {}), - files: pipelineResult.totalFileCount, - nodes: stats.nodes, - edges: stats.edges, - communities: pipelineResult.communityResult?.stats.totalCommunities, - processes: pipelineResult.processResult?.stats.totalProcesses, - }, - }; - - // 10. Persist meta + register repo. - await saveMeta(storagePath, newMeta); - const projectName = await registerRepo(repoPath, newMeta, { - name: options.registryName, - allowDuplicateName: options.allowDuplicateName, - }); - - // 11. Best-effort AI-context regeneration so AGENTS.md/CLAUDE.md stay - // in sync with the post-incremental graph state. - let aggregatedClusterCount = 0; - if (pipelineResult.communityResult?.communities) { - const groups = new Map(); - for (const c of pipelineResult.communityResult.communities) { - const label = c.heuristicLabel || c.label || 'Unknown'; - groups.set(label, (groups.get(label) || 0) + c.symbolCount); - } - aggregatedClusterCount = Array.from(groups.values()).filter((cnt) => cnt >= 5).length; - } - try { - await generateAIContextFiles( - repoPath, - storagePath, - projectName, - { - files: pipelineResult.totalFileCount, - nodes: stats.nodes, - edges: stats.edges, - communities: pipelineResult.communityResult?.stats.totalCommunities, - clusters: aggregatedClusterCount, - processes: pipelineResult.processResult?.stats.totalProcesses, - }, - undefined, - { skipAgentsMd: options.skipAgentsMd, noStats: options.noStats }, - ); - } catch { - /* best-effort */ - } - - await ensureGitNexusIgnored(repoPath); - await closeLbug(); - - progress('done', 100, 'Incremental complete'); - - return { - repoName: projectName, - repoPath, - stats: newMeta.stats ?? {}, - pipelineResult, - }; -} diff --git a/gitnexus/src/storage/repo-manager.ts b/gitnexus/src/storage/repo-manager.ts index 088f3262b..8c0bda95f 100644 --- a/gitnexus/src/storage/repo-manager.ts +++ b/gitnexus/src/storage/repo-manager.ts @@ -71,48 +71,8 @@ export interface RepoMeta { processes?: number; embeddings?: number; }; - /** - * Bumped whenever incremental-indexing invariants change in an - * incompatible way (schema bump, closure-algorithm fix, etc.). - * Mismatch → run-analyze forces a full rebuild. - * See docs/superpowers/specs/2026-05-10-incremental-indexing-design.md. - */ - schemaVersion?: number; - /** - * Per-file content/surface signatures used to drive incremental closure - * expansion. v1: SHA-256 of file content (cheap, conservative — body-only - * edits trigger 1-hop closure expansion). v2 will switch to a real - * surface-only signature so body-only edits stay at closure size 1. - * - * Map keys are repo-relative file paths; values are hex digests. Files - * not in this map are treated as "no previous signature" → first-time - * processing. - */ - surfaceSignatures?: Record; - /** - * Crash-recovery dirty flag. Written by run-analyze BEFORE any DB - * mutation in an incremental run; cleared by overwrite on success. Its - * presence at the start of the next run forces a full rebuild — the - * cheapest path back to a known-good index after a crashed/cancelled - * incremental run. Consumed by run-analyze.ts. - */ - incrementalInProgress?: { - closure: string[]; - startedAt: number; - }; } -/** - * Bumped whenever incremental-indexing invariants change in an - * incompatible way. v1 is `1`. Increment when: - * - The closure-expansion algorithm changes in a way that prior - * surfaceSignatures cannot be trusted. - * - The hydrate phase's DB schema assumptions change. - * - The surface-signature derivation changes (so prior hashes are not - * comparable to current ones). - */ -export const INCREMENTAL_SCHEMA_VERSION = 1; - export interface IndexedRepo { repoPath: string; storagePath: string; diff --git a/gitnexus/test/unit/incremental-closure.test.ts b/gitnexus/test/unit/incremental-closure.test.ts deleted file mode 100644 index 697056e49..000000000 --- a/gitnexus/test/unit/incremental-closure.test.ts +++ /dev/null @@ -1,212 +0,0 @@ -import { describe, it, expect, vi } from 'vitest'; -import { computeImporterClosure } from '../../src/core/incremental/closure.js'; - -/** - * Mini test harness: a fixture import graph + per-file surface signatures. - * - * `imports[X] = [a, b]` means files `a` and `b` import file `X`. So when we - * ask for "importers of X" the answer is `[a, b]`. - */ -interface Fixture { - files: string[]; - /** target → list of importers */ - imports: Record; - /** previous surfaces */ - prevSurfaces: Record; - /** new surfaces (what parseFile + surfaceFor will produce) */ - newSurfaces: Record; -} - -function makeHarness(fx: Fixture) { - const parseFile = vi.fn(async (f: string) => ({ filePath: f })); - const surfaceFor = vi.fn((f: string) => fx.newSurfaces[f] ?? ''); - const queryImporters = vi.fn(async (f: string) => fx.imports[f] ?? []); - return { parseFile, surfaceFor, queryImporters }; -} - -describe('computeImporterClosure', () => { - it('empty input → empty closure', async () => { - const { parseFile, surfaceFor, queryImporters } = makeHarness({ - files: [], - imports: {}, - prevSurfaces: {}, - newSurfaces: {}, - }); - const r = await computeImporterClosure({ - initialChangedFiles: new Set(), - prevSurfaces: {}, - parseFile, - surfaceFor, - queryImporters, - }); - expect(r.closure.size).toBe(0); - expect(parseFile).not.toHaveBeenCalled(); - }); - - it('single file with unchanged surface → closure size 1, no expansion', async () => { - const { parseFile, surfaceFor, queryImporters } = makeHarness({ - files: ['a.ts'], - imports: { 'a.ts': ['b.ts', 'c.ts'] }, - prevSurfaces: { 'a.ts': 'sig-a-v1' }, - newSurfaces: { 'a.ts': 'sig-a-v1' }, // same - }); - const r = await computeImporterClosure({ - initialChangedFiles: new Set(['a.ts']), - prevSurfaces: { 'a.ts': 'sig-a-v1' }, - parseFile, - surfaceFor, - queryImporters, - }); - expect([...r.closure].sort()).toEqual(['a.ts']); - // queryImporters should not have been called: surface unchanged. - expect(queryImporters).not.toHaveBeenCalled(); - expect(r.expandedFromImporters.size).toBe(0); - }); - - it('single file with changed surface → expands to direct importers', async () => { - const { parseFile, surfaceFor, queryImporters } = makeHarness({ - files: ['a.ts', 'b.ts', 'c.ts'], - imports: { - 'a.ts': ['b.ts', 'c.ts'], - 'b.ts': [], - 'c.ts': [], - }, - prevSurfaces: { 'a.ts': 'sig-old', 'b.ts': 'sig-b', 'c.ts': 'sig-c' }, - newSurfaces: { 'a.ts': 'sig-new', 'b.ts': 'sig-b', 'c.ts': 'sig-c' }, - }); - const r = await computeImporterClosure({ - initialChangedFiles: new Set(['a.ts']), - prevSurfaces: { 'a.ts': 'sig-old', 'b.ts': 'sig-b', 'c.ts': 'sig-c' }, - parseFile, - surfaceFor, - queryImporters, - }); - expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts', 'c.ts']); - expect([...r.expandedFromImporters].sort()).toEqual(['b.ts', 'c.ts']); - }); - - it('multi-hop cascade (A surface change → B in closure → B surface change → C in closure)', async () => { - const { parseFile, surfaceFor, queryImporters } = makeHarness({ - files: ['a.ts', 'b.ts', 'c.ts'], - imports: { - 'a.ts': ['b.ts'], // b imports a - 'b.ts': ['c.ts'], // c imports b - 'c.ts': [], - }, - prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old', 'c.ts': 'c-stable' }, - newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-new', 'c.ts': 'c-stable' }, - }); - const r = await computeImporterClosure({ - initialChangedFiles: new Set(['a.ts']), - prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old', 'c.ts': 'c-stable' }, - parseFile, - surfaceFor, - queryImporters, - }); - expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts', 'c.ts']); - }); - - it('cascade halts when surface stops changing mid-chain', async () => { - const { parseFile, surfaceFor, queryImporters } = makeHarness({ - files: ['a.ts', 'b.ts', 'c.ts'], - imports: { - 'a.ts': ['b.ts'], - 'b.ts': ['c.ts'], - }, - // a's surface changed, b is in closure but b's surface unchanged → c stays out - prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-stable', 'c.ts': 'c-stable' }, - newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-stable', 'c.ts': 'c-stable' }, - }); - const r = await computeImporterClosure({ - initialChangedFiles: new Set(['a.ts']), - prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-stable', 'c.ts': 'c-stable' }, - parseFile, - surfaceFor, - queryImporters, - }); - expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts']); - expect([...r.expandedFromImporters].sort()).toEqual(['b.ts']); - }); - - it('terminates on cycles in the import graph', async () => { - // a ↔ b cycle. a's surface changes, b is added; b's surface also "changed" - // (relative to undefined prev), so its importers are queried — which - // includes a, already in closure. Loop terminates because closure - // membership prevents re-add. - const { parseFile, surfaceFor, queryImporters } = makeHarness({ - files: ['a.ts', 'b.ts'], - imports: { 'a.ts': ['b.ts'], 'b.ts': ['a.ts'] }, - prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' }, - newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-new' }, - }); - const r = await computeImporterClosure({ - initialChangedFiles: new Set(['a.ts']), - prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' }, - parseFile, - surfaceFor, - queryImporters, - }); - expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts']); - // Each file parsed exactly once even though they import each other. - expect(parseFile).toHaveBeenCalledTimes(2); - }); - - it('newly-added file (no prev surface) triggers expansion', async () => { - // For a brand-new file, prevSurfaces[f] is undefined → surfaceChanged - // returns true → its importers are queried. (For a truly *new* file, - // there should be no importers yet, so closure stays at {f}.) - const { parseFile, surfaceFor, queryImporters } = makeHarness({ - files: ['new.ts'], - imports: {}, - prevSurfaces: {}, - newSurfaces: { 'new.ts': 'sig-new' }, - }); - const r = await computeImporterClosure({ - initialChangedFiles: new Set(['new.ts']), - prevSurfaces: {}, - parseFile, - surfaceFor, - queryImporters, - }); - expect([...r.closure]).toEqual(['new.ts']); - expect(queryImporters).toHaveBeenCalledOnce(); - }); - - it('parses each file exactly once and caches the result', async () => { - const { parseFile, surfaceFor, queryImporters } = makeHarness({ - files: ['a.ts', 'b.ts'], - imports: { 'a.ts': ['b.ts'] }, - prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b' }, - newSurfaces: { 'a.ts': 'new', 'b.ts': 'b' }, - }); - const r = await computeImporterClosure({ - initialChangedFiles: new Set(['a.ts']), - prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b' }, - parseFile, - surfaceFor, - queryImporters, - }); - expect(parseFile).toHaveBeenCalledTimes(2); - expect(r.parseCache.size).toBe(2); - expect(r.parseCache.has('a.ts')).toBe(true); - expect(r.parseCache.has('b.ts')).toBe(true); - }); - - it('records new surfaces for every parsed file', async () => { - const { parseFile, surfaceFor, queryImporters } = makeHarness({ - files: ['a.ts', 'b.ts'], - imports: { 'a.ts': ['b.ts'] }, - prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' }, - newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-new' }, - }); - const r = await computeImporterClosure({ - initialChangedFiles: new Set(['a.ts']), - prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' }, - parseFile, - surfaceFor, - queryImporters, - }); - expect(r.newSurfaces.get('a.ts')).toBe('new'); - expect(r.newSurfaces.get('b.ts')).toBe('b-new'); - }); -}); diff --git a/gitnexus/test/unit/incremental-git-diff.test.ts b/gitnexus/test/unit/incremental-git-diff.test.ts deleted file mode 100644 index 85efe8d36..000000000 --- a/gitnexus/test/unit/incremental-git-diff.test.ts +++ /dev/null @@ -1,163 +0,0 @@ -import { describe, it, expect, vi, beforeEach } from 'vitest'; -import { execFileSync } from 'child_process'; -import { - getChangedFilesSinceCommit, - LastCommitMissingError, -} from '../../src/core/incremental/git-diff.js'; - -vi.mock('child_process', () => ({ - execFileSync: vi.fn(), -})); - -const mockExec = vi.mocked(execFileSync); - -/** - * Mock helper: queue a sequence of execFileSync return values. - * Order matters — the first call returns the first value, etc. - * - * Calls in `getChangedFilesSinceCommit`: - * 1. cat-file -e (commitExists check) - * 2. diff --name-status -z (committed changes) - * 3. status --porcelain -z (dirty tree) - */ -function mockGitSequence( - catFileSucceeds: boolean, - diffOutput: string, - statusOutput: string, -) { - mockExec.mockReset(); - if (catFileSucceeds) { - mockExec.mockImplementationOnce(() => Buffer.from('')); - } else { - mockExec.mockImplementationOnce(() => { - throw new Error('not in repo'); - }); - } - mockExec.mockImplementationOnce(() => diffOutput); - mockExec.mockImplementationOnce(() => statusOutput); -} - -describe('getChangedFilesSinceCommit', () => { - beforeEach(() => { - vi.clearAllMocks(); - }); - - it('throws LastCommitMissingError when commit not in repo', () => { - mockGitSequence(false, '', ''); - expect(() => getChangedFilesSinceCommit('/repo', 'deadbeef')).toThrow( - LastCommitMissingError, - ); - }); - - it('returns empty arrays for clean repo and no committed changes', () => { - mockGitSequence(true, '', ''); - const r = getChangedFilesSinceCommit('/repo', 'abc'); - expect(r).toEqual({ modified: [], added: [], deleted: [] }); - }); - - it('parses committed M/A/D from diff --name-status -z', () => { - mockGitSequence( - true, - 'M\0src/a.ts\0A\0src/b.ts\0D\0src/c.ts\0', - '', - ); - const r = getChangedFilesSinceCommit('/repo', 'abc'); - expect(r).toEqual({ - modified: ['src/a.ts'], - added: ['src/b.ts'], - deleted: ['src/c.ts'], - }); - }); - - it('flattens R renames into delete(orig) + add(new)', () => { - mockGitSequence( - true, - 'R100\0src/old.ts\0src/new.ts\0', - '', - ); - const r = getChangedFilesSinceCommit('/repo', 'abc'); - expect(r).toEqual({ - modified: [], - added: ['src/new.ts'], - deleted: ['src/old.ts'], - }); - }); - - it('treats T (type change) as modified', () => { - mockGitSequence(true, 'T\0src/link.ts\0', ''); - const r = getChangedFilesSinceCommit('/repo', 'abc'); - expect(r.modified).toContain('src/link.ts'); - }); - - it('parses dirty tree from status --porcelain -z', () => { - // 'M a.ts' = staged-modified. ' M b.ts' = unstaged-modified. - // '?? c.ts' = untracked. ' D d.ts' = unstaged-deleted. - mockGitSequence( - true, - '', - 'M a.ts\0 M b.ts\0?? c.ts\0 D d.ts\0', - ); - const r = getChangedFilesSinceCommit('/repo', 'abc'); - expect(r.modified.sort()).toEqual(['a.ts', 'b.ts']); - expect(r.added).toEqual(['c.ts']); - expect(r.deleted).toEqual(['d.ts']); - }); - - it('unions committed + dirty changes', () => { - mockGitSequence( - true, - 'M\0a.ts\0', // committed: a.ts modified - ' M b.ts\0?? c.ts\0', // dirty: b.ts modified, c.ts untracked - ); - const r = getChangedFilesSinceCommit('/repo', 'abc'); - expect(r.modified.sort()).toEqual(['a.ts', 'b.ts']); - expect(r.added).toEqual(['c.ts']); - expect(r.deleted).toEqual([]); - }); - - it('resolves overlap: file added in diff and modified in status → added', () => { - mockGitSequence( - true, - 'A\0newfile.ts\0', - ' M newfile.ts\0', - ); - const r = getChangedFilesSinceCommit('/repo', 'abc'); - expect(r).toEqual({ - modified: [], - added: ['newfile.ts'], - deleted: [], - }); - }); - - it('handles porcelain rename ("R new\\0old")', () => { - mockGitSequence( - true, - '', - 'R new.ts\0old.ts\0', - ); - const r = getChangedFilesSinceCommit('/repo', 'abc'); - expect(r.added).toEqual(['new.ts']); - // Porcelain rename: `R` in status doesn't pre-flatten old into deleted - // because git already resolved the rename — but our parser does flatten - // when the diff layer flags it. For status-only, the original is the - // rename source and we don't claim to know it was deleted from index. - }); - - it('returns sorted output for stable comparison', () => { - mockGitSequence( - true, - 'M\0z.ts\0M\0a.ts\0M\0m.ts\0', - '', - ); - const r = getChangedFilesSinceCommit('/repo', 'abc'); - expect(r.modified).toEqual(['a.ts', 'm.ts', 'z.ts']); - }); - - it('passes correct cwd to git', () => { - mockGitSequence(true, '', ''); - getChangedFilesSinceCommit('/some/path', 'abc'); - for (const call of mockExec.mock.calls) { - expect((call[2] as { cwd: string }).cwd).toBe('/some/path'); - } - }); -}); diff --git a/gitnexus/test/unit/incremental-surface.test.ts b/gitnexus/test/unit/incremental-surface.test.ts deleted file mode 100644 index 5ecd6346e..000000000 --- a/gitnexus/test/unit/incremental-surface.test.ts +++ /dev/null @@ -1,185 +0,0 @@ -import { describe, it, expect } from 'vitest'; -import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; -import { - extractSurfaceSignature, - surfaceChanged, -} from '../../src/core/incremental/surface.js'; -import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; - -function fn( - id: string, - filePath: string, - name: string, - extra: Record = {}, -): GraphNode { - return { - id, - label: 'Function', - properties: { name, filePath, ...extra }, - }; -} - -function method( - id: string, - filePath: string, - name: string, - extra: Record = {}, -): GraphNode { - return { - id, - label: 'Method', - properties: { name, filePath, ...extra }, - }; -} - -function cls( - id: string, - filePath: string, - name: string, - extra: Record = {}, -): GraphNode { - return { - id, - label: 'Class', - properties: { name, filePath, ...extra }, - }; -} - -function rel( - id: string, - type: GraphRelationship['type'], - src: string, - dst: string, -): GraphRelationship { - return { id, type, sourceId: src, targetId: dst, confidence: 1, reason: 't' }; -} - -describe('extractSurfaceSignature', () => { - it('returns a stable hash for an empty file', () => { - const g = createKnowledgeGraph(); - const h1 = extractSurfaceSignature(g, 'a.ts'); - const h2 = extractSurfaceSignature(g, 'a.ts'); - expect(h1).toBe(h2); - expect(h1).toMatch(/^[a-f0-9]{64}$/); - }); - - it('produces the same hash for the same surface', () => { - const g1 = createKnowledgeGraph(); - g1.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 })); - const g2 = createKnowledgeGraph(); - g2.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 })); - expect(extractSurfaceSignature(g1, 'a.ts')).toBe( - extractSurfaceSignature(g2, 'a.ts'), - ); - }); - - it('is invariant under node insertion order', () => { - const g1 = createKnowledgeGraph(); - g1.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 })); - g1.addNode(fn('Function:a.ts:bar#1', 'a.ts', 'bar', { parameterCount: 1 })); - const g2 = createKnowledgeGraph(); - g2.addNode(fn('Function:a.ts:bar#1', 'a.ts', 'bar', { parameterCount: 1 })); - g2.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 })); - expect(extractSurfaceSignature(g1, 'a.ts')).toBe( - extractSurfaceSignature(g2, 'a.ts'), - ); - }); - - it('changes when a function is renamed', () => { - const g1 = createKnowledgeGraph(); - g1.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo')); - const g2 = createKnowledgeGraph(); - g2.addNode(fn('Function:a.ts:bar#0', 'a.ts', 'bar')); - expect(extractSurfaceSignature(g1, 'a.ts')).not.toBe( - extractSurfaceSignature(g2, 'a.ts'), - ); - }); - - it('changes when a function signature changes', () => { - const g1 = createKnowledgeGraph(); - g1.addNode( - fn('Function:a.ts:foo#1', 'a.ts', 'foo', { - parameterCount: 1, - parameterTypes: ['number'], - returnType: 'string', - }), - ); - const g2 = createKnowledgeGraph(); - g2.addNode( - fn('Function:a.ts:foo#1', 'a.ts', 'foo', { - parameterCount: 1, - parameterTypes: ['string'], - returnType: 'string', - }), - ); - expect(extractSurfaceSignature(g1, 'a.ts')).not.toBe( - extractSurfaceSignature(g2, 'a.ts'), - ); - }); - - it('does NOT change for body-only edits (no surface mutation)', () => { - // Body-only edits don't add/remove/rename surface nodes — same hash. - const g = createKnowledgeGraph(); - g.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 })); - const h1 = extractSurfaceSignature(g, 'a.ts'); - // Re-build with same surface (simulating a re-parse of a body-only edit). - const g2 = createKnowledgeGraph(); - g2.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 })); - const h2 = extractSurfaceSignature(g2, 'a.ts'); - expect(h1).toBe(h2); - }); - - it('only considers nodes whose filePath matches', () => { - const g = createKnowledgeGraph(); - g.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo')); - g.addNode(fn('Function:b.ts:bar#0', 'b.ts', 'bar')); - const ha = extractSurfaceSignature(g, 'a.ts'); - const hb = extractSurfaceSignature(g, 'b.ts'); - expect(ha).not.toBe(hb); - // Adding a node in a OTHER file should not change a's signature. - g.addNode(fn('Function:c.ts:baz#0', 'c.ts', 'baz')); - expect(extractSurfaceSignature(g, 'a.ts')).toBe(ha); - }); - - it('changes when EXTENDS target changes', () => { - const g1 = createKnowledgeGraph(); - g1.addNode(cls('Class:a.ts:Child#0', 'a.ts', 'Child')); - g1.addNode(cls('Class:b.ts:ParentA#0', 'b.ts', 'ParentA')); - g1.addRelationship( - rel('r1', 'EXTENDS', 'Class:a.ts:Child#0', 'Class:b.ts:ParentA#0'), - ); - - const g2 = createKnowledgeGraph(); - g2.addNode(cls('Class:a.ts:Child#0', 'a.ts', 'Child')); - g2.addNode(cls('Class:b.ts:ParentB#0', 'b.ts', 'ParentB')); - g2.addRelationship( - rel('r1', 'EXTENDS', 'Class:a.ts:Child#0', 'Class:b.ts:ParentB#0'), - ); - - expect(extractSurfaceSignature(g1, 'a.ts')).not.toBe( - extractSurfaceSignature(g2, 'a.ts'), - ); - }); - - it('includes Method/Class/Interface in the surface', () => { - const g = createKnowledgeGraph(); - g.addNode(cls('Class:a.ts:C#0', 'a.ts', 'C')); - g.addNode(method('Method:a.ts:C.m#0', 'a.ts', 'm')); - const h = extractSurfaceSignature(g, 'a.ts'); - expect(h).toMatch(/^[a-f0-9]{64}$/); - }); -}); - -describe('surfaceChanged', () => { - it('returns true when prev is undefined (first index of file)', () => { - expect(surfaceChanged(undefined, 'abc')).toBe(true); - }); - - it('returns false when prev equals current', () => { - expect(surfaceChanged('abc', 'abc')).toBe(false); - }); - - it('returns true when prev differs from current', () => { - expect(surfaceChanged('abc', 'def')).toBe(true); - }); -});