mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-07 02:58:02 +00:00
Revert v1 incremental indexing (5 commits)
Reverts the v1 design that parsed only closure files into a fresh graph and tried to hydrate the rest from DB. Real-repo equivalence test failed: cross-file resolution operates on partial parse data (closure files only), so CALLS edges that resolve through unchanged files silently fall off. Diff against full rebuild on the same edited state: -50 nodes, -425 edges, -5 communities, -48 processes. Architecture pivot: switch to PR #533-style content-addressed parse cache. Pipeline parses every file (cache-served when possible), giving cross-file resolution full data, with DB writeback then restricted to changed-file rows. Reverts:d4b9de47fix(incremental): drop invalid --no-renames=falsef35f7634feat(analyze): incremental orchestrator branch + meta schemabc039686feat(pipeline): hydrate phase + parse-filter98bb893dfeat(lbug): loadGraphFromLbug, queryImporters, ...aa8d7ae3feat(incremental): change-detection, surface signatures, closure Kept:d9e340b0feat(communities): seed Leiden RNG (foundational)8235ca36docs: incremental indexing design spec (will be revised) Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
d4b9de4765
commit
37ef3dda02
16 changed files with 13 additions and 2087 deletions
|
|
@ -6,7 +6,6 @@ export type PipelinePhase =
|
|||
| 'idle'
|
||||
| 'extracting'
|
||||
| 'structure'
|
||||
| 'hydrate'
|
||||
| 'parsing'
|
||||
| 'imports'
|
||||
| 'calls'
|
||||
|
|
|
|||
|
|
@ -1,115 +0,0 @@
|
|||
/**
|
||||
* Iterative importer-closure computation for incremental indexing.
|
||||
*
|
||||
* Given a set of changed files, walks the IMPORTS graph (queried from the
|
||||
* existing DB) to determine which other files must also be re-parsed for the
|
||||
* incremental run to be byte-equivalent to a full rebuild.
|
||||
*
|
||||
* Algorithm:
|
||||
* 1. Start with `closure = changedFiles`.
|
||||
* 2. For each file `f` newly added to closure: parse it, extract surface.
|
||||
* 3. If surface differs from `prevSurfaces[f]`, add all importers of `f`
|
||||
* to closure.
|
||||
* 4. Repeat until no new files added.
|
||||
*
|
||||
* Termination: each iteration either adds files or stops. The universe of
|
||||
* files is finite (bounded by the repo size), so the loop terminates.
|
||||
*
|
||||
* Correctness invariant: a file is added to the closure iff its CALLS or
|
||||
* IMPORTS edges might resolve differently than they did at `lastCommit`. The
|
||||
* surface check is the crisp condition: surface unchanged → no resolver
|
||||
* change → file's edges don't need re-emission.
|
||||
*/
|
||||
|
||||
export interface ClosureInput<TParseResult> {
|
||||
/** Initial set of files (from git diff: modified ∪ added). */
|
||||
initialChangedFiles: Set<string>;
|
||||
/** Previously stored surface signatures (filePath → hash). */
|
||||
prevSurfaces: Record<string, string>;
|
||||
/**
|
||||
* Parse one file and return the parser result. The closure logic only
|
||||
* cares about producing a stable surface signature, so callers may use
|
||||
* the same parse worker the main pipeline uses.
|
||||
*/
|
||||
parseFile: (filePath: string) => Promise<TParseResult>;
|
||||
/**
|
||||
* Compute the surface signature for `filePath` given the parse result.
|
||||
* Returned signature is hashed and compared against `prevSurfaces`.
|
||||
*/
|
||||
surfaceFor: (filePath: string, parsed: TParseResult) => string;
|
||||
/**
|
||||
* Query the existing DB for files that import `filePath`. Returns
|
||||
* repo-relative paths. A missing file (e.g. importer of a brand-new file)
|
||||
* legitimately returns `[]`.
|
||||
*/
|
||||
queryImporters: (filePath: string) => Promise<string[]>;
|
||||
}
|
||||
|
||||
export interface ClosureResult<TParseResult> {
|
||||
/** Final set of files that must be re-parsed. */
|
||||
closure: Set<string>;
|
||||
/**
|
||||
* Cache of parsed results, keyed by file path. The orchestrator hands
|
||||
* this to the pipeline so the parse phase doesn't re-parse closure files.
|
||||
*/
|
||||
parseCache: Map<string, TParseResult>;
|
||||
/** Newly computed surface signatures, keyed by file path. */
|
||||
newSurfaces: Map<string, string>;
|
||||
/**
|
||||
* Files added to closure ONLY because an importer chain pulled them in
|
||||
* (not in the initial changed set). Useful for logging.
|
||||
*/
|
||||
expandedFromImporters: Set<string>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute the transitive importer closure of the initial changed files,
|
||||
* pruned by the surface-change optimization.
|
||||
*
|
||||
* The function is generic over `TParseResult` — the orchestrator is free
|
||||
* to use the existing parse-worker result type, but the closure module
|
||||
* itself is decoupled from any specific parse representation.
|
||||
*/
|
||||
export async function computeImporterClosure<TParseResult>(
|
||||
input: ClosureInput<TParseResult>,
|
||||
): Promise<ClosureResult<TParseResult>> {
|
||||
const { initialChangedFiles, prevSurfaces, parseFile, surfaceFor, queryImporters } = input;
|
||||
|
||||
const closure = new Set<string>(initialChangedFiles);
|
||||
const queue: string[] = [...initialChangedFiles];
|
||||
const parseCache = new Map<string, TParseResult>();
|
||||
const newSurfaces = new Map<string, string>();
|
||||
const expandedFromImporters = new Set<string>();
|
||||
|
||||
while (queue.length > 0) {
|
||||
const f = queue.shift()!;
|
||||
|
||||
// Parse this file (if we haven't already in this run).
|
||||
let parsed = parseCache.get(f);
|
||||
if (parsed === undefined) {
|
||||
parsed = await parseFile(f);
|
||||
parseCache.set(f, parsed);
|
||||
}
|
||||
|
||||
// Compute the new surface signature.
|
||||
const newSurface = surfaceFor(f, parsed);
|
||||
newSurfaces.set(f, newSurface);
|
||||
|
||||
// If the surface changed, every file importing `f` may have stale
|
||||
// resolver output and must be re-parsed.
|
||||
const prev = prevSurfaces[f];
|
||||
const changed = prev === undefined || prev !== newSurface;
|
||||
if (changed) {
|
||||
const importers = await queryImporters(f);
|
||||
for (const i of importers) {
|
||||
if (!closure.has(i)) {
|
||||
closure.add(i);
|
||||
queue.push(i);
|
||||
expandedFromImporters.add(i);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return { closure, parseCache, newSurfaces, expandedFromImporters };
|
||||
}
|
||||
|
|
@ -1,61 +0,0 @@
|
|||
/**
|
||||
* File-content hashing — v1 "surface signature" for incremental indexing.
|
||||
*
|
||||
* v1 trade-off: we use SHA-256 of file content as the surface signature.
|
||||
* This is conservative — a body-only edit (which doesn't actually change
|
||||
* any other file's resolution) still triggers 1-hop closure expansion
|
||||
* because the content hash differs.
|
||||
*
|
||||
* v2 will switch to a real surface-only signature (extracted from the
|
||||
* post-parse graph via `extractSurfaceSignature` in `surface.ts`) so that
|
||||
* body-only edits stay at closure size 1. The plumbing for that is in
|
||||
* place — `surface.ts` and the closure module are signature-agnostic —
|
||||
* but reusing the parse-worker output for one-off surface extraction
|
||||
* requires more pipeline integration than is needed for v1.
|
||||
*/
|
||||
|
||||
import { createHash } from 'crypto';
|
||||
import fs from 'fs/promises';
|
||||
import path from 'path';
|
||||
|
||||
/**
|
||||
* Compute SHA-256 hex digest of a single file. Returns null when the
|
||||
* file can't be read (deleted between scan and hash, permission error,
|
||||
* etc.) — caller should treat null as "no signature available, assume
|
||||
* changed".
|
||||
*/
|
||||
export async function computeFileHash(absPath: string): Promise<string | null> {
|
||||
try {
|
||||
const buf = await fs.readFile(absPath);
|
||||
return createHash('sha256').update(buf).digest('hex');
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute SHA-256 hashes for a list of files in `repoPath` (paths are
|
||||
* repo-relative). Parallel batched I/O bounded at 100 concurrent reads
|
||||
* to avoid fd exhaustion on huge repos.
|
||||
*
|
||||
* Returns a Map<relPath, hash>. Files that fail to read are omitted
|
||||
* from the result (caller treats them as "no signature").
|
||||
*/
|
||||
export async function computeFileHashes(
|
||||
repoPath: string,
|
||||
relPaths: readonly string[],
|
||||
): Promise<Map<string, string>> {
|
||||
const out = new Map<string, string>();
|
||||
const BATCH = 100;
|
||||
for (let i = 0; i < relPaths.length; i += BATCH) {
|
||||
const batch = relPaths.slice(i, i + BATCH);
|
||||
const results = await Promise.all(
|
||||
batch.map(async (rel) => {
|
||||
const hash = await computeFileHash(path.join(repoPath, rel));
|
||||
return hash ? ([rel, hash] as const) : null;
|
||||
}),
|
||||
);
|
||||
for (const r of results) if (r) out.set(r[0], r[1]);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
|
@ -1,211 +0,0 @@
|
|||
/**
|
||||
* Git-based change detection for incremental indexing.
|
||||
*
|
||||
* Combines `git diff --name-status <lastCommit> HEAD` (committed changes since
|
||||
* last index) with `git status --porcelain` (uncommitted/dirty tree changes)
|
||||
* to produce a precise per-file change set.
|
||||
*
|
||||
* Renames (R<sim> in diff output, R in porcelain) are flattened into
|
||||
* delete(old) + add(new) — downstream consumers don't need to know about
|
||||
* renames specifically; they just need to know "this path's nodes are stale,
|
||||
* delete them" and "this path is new, parse it".
|
||||
*
|
||||
* Returns a typed result; throws `LastCommitMissingError` when `lastCommit`
|
||||
* no longer exists in the repository (rebase or shallow clone). Callers
|
||||
* should fall back to a full rebuild on this error.
|
||||
*/
|
||||
|
||||
import { execFileSync } from 'child_process';
|
||||
|
||||
export interface ChangedFiles {
|
||||
/** Files whose content changed (re-parse needed). */
|
||||
modified: string[];
|
||||
/** Files newly introduced since lastCommit (re-parse needed). */
|
||||
added: string[];
|
||||
/** Files that existed at lastCommit but no longer do (delete-only, no parse). */
|
||||
deleted: string[];
|
||||
}
|
||||
|
||||
export class LastCommitMissingError extends Error {
|
||||
constructor(commit: string) {
|
||||
super(`lastCommit ${commit} not found in repository — cannot compute incremental diff`);
|
||||
this.name = 'LastCommitMissingError';
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute the set of files changed in `repoPath` since `lastCommit`.
|
||||
* Combines committed differences (`git diff` against HEAD) with the
|
||||
* current dirty working-tree state (`git status`).
|
||||
*
|
||||
* @throws LastCommitMissingError when `lastCommit` is not reachable.
|
||||
*/
|
||||
export function getChangedFilesSinceCommit(
|
||||
repoPath: string,
|
||||
lastCommit: string,
|
||||
): ChangedFiles {
|
||||
if (!commitExists(repoPath, lastCommit)) {
|
||||
throw new LastCommitMissingError(lastCommit);
|
||||
}
|
||||
|
||||
const modified = new Set<string>();
|
||||
const added = new Set<string>();
|
||||
const deleted = new Set<string>();
|
||||
|
||||
// ── Committed differences: lastCommit..HEAD ────────────────────────────
|
||||
// Output (NUL-delimited via -z): STATUS\0path1\0[path2\0]
|
||||
// Status codes: M, A, D, T (type change → treat as modified),
|
||||
// R<sim>, C<sim> (rename/copy with similarity %).
|
||||
const diffRaw = runGit(repoPath, [
|
||||
'diff',
|
||||
'--name-status',
|
||||
'-z',
|
||||
`${lastCommit}`,
|
||||
'HEAD',
|
||||
]);
|
||||
|
||||
for (const entry of parseNameStatusZ(diffRaw)) {
|
||||
classify(entry, modified, added, deleted);
|
||||
}
|
||||
|
||||
// ── Working-tree differences: HEAD..disk ───────────────────────────────
|
||||
// Format (NUL-delimited): XY<space>path[\0orig-path]
|
||||
// X = staged, Y = unstaged. Either non-' ' counts as a change.
|
||||
const statusRaw = runGit(repoPath, ['status', '--porcelain', '-z']);
|
||||
for (const entry of parsePorcelainZ(statusRaw)) {
|
||||
classify(entry, modified, added, deleted);
|
||||
}
|
||||
|
||||
// Resolve overlaps: a file added in the diff and modified in the status
|
||||
// is just "new on disk" → keep it in `added`. A file deleted in the diff
|
||||
// but present in status as added (re-introduced) → modified.
|
||||
for (const f of added) {
|
||||
modified.delete(f);
|
||||
deleted.delete(f);
|
||||
}
|
||||
for (const f of deleted) {
|
||||
modified.delete(f);
|
||||
}
|
||||
|
||||
return {
|
||||
modified: [...modified].sort(),
|
||||
added: [...added].sort(),
|
||||
deleted: [...deleted].sort(),
|
||||
};
|
||||
}
|
||||
|
||||
interface ParsedEntry {
|
||||
status: string;
|
||||
path: string;
|
||||
origPath?: string;
|
||||
}
|
||||
|
||||
function classify(
|
||||
entry: ParsedEntry,
|
||||
modified: Set<string>,
|
||||
added: Set<string>,
|
||||
deleted: Set<string>,
|
||||
): void {
|
||||
const code = entry.status[0];
|
||||
switch (code) {
|
||||
case 'M':
|
||||
case 'T': // type change (file → symlink, etc.)
|
||||
modified.add(entry.path);
|
||||
break;
|
||||
case 'A':
|
||||
case '?': // untracked (porcelain '??') treat as added
|
||||
added.add(entry.path);
|
||||
break;
|
||||
case 'D':
|
||||
deleted.add(entry.path);
|
||||
break;
|
||||
case 'R': // rename: flatten to delete(orig) + add(new)
|
||||
case 'C': // copy: original survives; new file is added
|
||||
if (entry.origPath && code === 'R') deleted.add(entry.origPath);
|
||||
added.add(entry.path);
|
||||
break;
|
||||
case 'U': // unmerged — surface as modified, caller's problem
|
||||
modified.add(entry.path);
|
||||
break;
|
||||
default:
|
||||
// Unknown status — be conservative, treat as modified.
|
||||
modified.add(entry.path);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse `git diff --name-status -z` output. NUL-delimited, status before each path:
|
||||
* "M\0a.ts\0A\0b.ts\0R100\0old.ts\0new.ts\0"
|
||||
*/
|
||||
function parseNameStatusZ(raw: string): ParsedEntry[] {
|
||||
const entries: ParsedEntry[] = [];
|
||||
if (!raw) return entries;
|
||||
const tokens = raw.split('\0').filter((t) => t.length > 0);
|
||||
let i = 0;
|
||||
while (i < tokens.length) {
|
||||
const status = tokens[i++];
|
||||
const path = tokens[i++];
|
||||
if (path === undefined) break;
|
||||
if (status[0] === 'R' || status[0] === 'C') {
|
||||
const newPath = tokens[i++];
|
||||
if (newPath !== undefined) {
|
||||
entries.push({ status, path: newPath, origPath: path });
|
||||
}
|
||||
} else {
|
||||
entries.push({ status, path });
|
||||
}
|
||||
}
|
||||
return entries;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse `git status --porcelain -z` output. NUL-delimited, two-char status
|
||||
* code + space + path; for renames the original path follows after another NUL:
|
||||
* "M a.ts\0?? b.ts\0R new.ts\0old.ts\0"
|
||||
*/
|
||||
function parsePorcelainZ(raw: string): ParsedEntry[] {
|
||||
const entries: ParsedEntry[] = [];
|
||||
if (!raw) return entries;
|
||||
const tokens = raw.split('\0').filter((t) => t.length > 0);
|
||||
let i = 0;
|
||||
while (i < tokens.length) {
|
||||
const tok = tokens[i++];
|
||||
// First two chars are the XY status, then a space, then the path.
|
||||
const xy = tok.slice(0, 2);
|
||||
const path = tok.slice(3);
|
||||
// Effective status: prefer staged (X) when not ' ', otherwise unstaged (Y).
|
||||
const code = xy[0] !== ' ' && xy[0] !== '?' ? xy[0] : xy[1];
|
||||
if (xy.startsWith('R') || xy[1] === 'R') {
|
||||
// Rename: next token is the original path.
|
||||
const orig = tokens[i++];
|
||||
entries.push({ status: 'R', path, origPath: orig });
|
||||
} else if (xy === '??') {
|
||||
entries.push({ status: '?', path });
|
||||
} else {
|
||||
entries.push({ status: code, path });
|
||||
}
|
||||
}
|
||||
return entries;
|
||||
}
|
||||
|
||||
function commitExists(repoPath: string, commit: string): boolean {
|
||||
try {
|
||||
execFileSync('git', ['cat-file', '-e', `${commit}^{commit}`], {
|
||||
cwd: repoPath,
|
||||
stdio: ['ignore', 'ignore', 'ignore'],
|
||||
});
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function runGit(repoPath: string, args: string[]): string {
|
||||
return execFileSync('git', args, {
|
||||
cwd: repoPath,
|
||||
stdio: ['ignore', 'pipe', 'ignore'],
|
||||
encoding: 'utf8',
|
||||
// Larger buffer for large repos with many changes.
|
||||
maxBuffer: 100 * 1024 * 1024,
|
||||
});
|
||||
}
|
||||
|
|
@ -1,267 +0,0 @@
|
|||
/**
|
||||
* Incremental-indexing orchestrator helpers.
|
||||
*
|
||||
* High-level helpers used by `run-analyze.ts` to keep the incremental path
|
||||
* out of the main full-rebuild orchestrator function. Each helper has one
|
||||
* responsibility:
|
||||
*
|
||||
* isIncrementalEligible(...) — should this run go incremental?
|
||||
* computeIncrementalClosure(...) — git diff → content-hash → closure
|
||||
* extractIncrementalSubgraph(...) — ctx.graph → "nodes/edges to write"
|
||||
* commitIncrementalProgress(...) — set the dirty flag in meta.json
|
||||
*
|
||||
* See docs/superpowers/specs/2026-05-10-incremental-indexing-design.md.
|
||||
*/
|
||||
|
||||
import { execFileSync } from 'child_process';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
import { createKnowledgeGraph } from '../graph/graph.js';
|
||||
import type { KnowledgeGraph } from '../graph/types.js';
|
||||
import {
|
||||
type RepoMeta,
|
||||
INCREMENTAL_SCHEMA_VERSION,
|
||||
saveMeta,
|
||||
} from '../../storage/repo-manager.js';
|
||||
import { hasGitDir } from '../../storage/git.js';
|
||||
import {
|
||||
getChangedFilesSinceCommit,
|
||||
type ChangedFiles,
|
||||
} from './git-diff.js';
|
||||
import { computeImporterClosure } from './closure.js';
|
||||
import { computeFileHashes } from './file-hash.js';
|
||||
import { queryImporters } from '../lbug/lbug-adapter.js';
|
||||
|
||||
/** Decision returned by isIncrementalEligible. */
|
||||
export interface IncrementalEligibility {
|
||||
/** True iff this run should attempt the incremental path. */
|
||||
eligible: boolean;
|
||||
/** When `eligible === false`, a short reason for logging. */
|
||||
reason?: string;
|
||||
/** The previous lastCommit, when eligible. */
|
||||
lastCommit?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decide whether a run is eligible for the incremental path.
|
||||
*
|
||||
* All conditions must hold:
|
||||
* - --force not passed
|
||||
* - existing meta.json present and previously a full rebuild has populated
|
||||
* surfaceSignatures + schemaVersion (matching CURRENT)
|
||||
* - repo has .git (non-git repos always do full rebuild)
|
||||
* - meta.lastCommit still resolvable (not rebased away)
|
||||
* - no incrementalInProgress flag (would force full rebuild for safety)
|
||||
*/
|
||||
export function isIncrementalEligible(
|
||||
repoPath: string,
|
||||
existingMeta: RepoMeta | null | undefined,
|
||||
optionsForce: boolean | undefined,
|
||||
): IncrementalEligibility {
|
||||
if (optionsForce) {
|
||||
return { eligible: false, reason: '--force passed' };
|
||||
}
|
||||
if (!existingMeta) {
|
||||
return { eligible: false, reason: 'no existing index' };
|
||||
}
|
||||
if (existingMeta.incrementalInProgress) {
|
||||
return {
|
||||
eligible: false,
|
||||
reason: 'previous incremental run did not complete cleanly',
|
||||
};
|
||||
}
|
||||
if (
|
||||
existingMeta.schemaVersion === undefined ||
|
||||
existingMeta.schemaVersion !== INCREMENTAL_SCHEMA_VERSION
|
||||
) {
|
||||
return {
|
||||
eligible: false,
|
||||
reason: `schemaVersion mismatch (have ${existingMeta.schemaVersion}, want ${INCREMENTAL_SCHEMA_VERSION})`,
|
||||
};
|
||||
}
|
||||
if (!existingMeta.surfaceSignatures || Object.keys(existingMeta.surfaceSignatures).length === 0) {
|
||||
return {
|
||||
eligible: false,
|
||||
reason: 'no prior surfaceSignatures in meta.json',
|
||||
};
|
||||
}
|
||||
if (!hasGitDir(repoPath)) {
|
||||
return { eligible: false, reason: 'non-git repo' };
|
||||
}
|
||||
if (!existingMeta.lastCommit) {
|
||||
return { eligible: false, reason: 'no lastCommit recorded' };
|
||||
}
|
||||
if (!commitExists(repoPath, existingMeta.lastCommit)) {
|
||||
return {
|
||||
eligible: false,
|
||||
reason: `lastCommit ${existingMeta.lastCommit.slice(0, 7)} not in repo`,
|
||||
};
|
||||
}
|
||||
return { eligible: true, lastCommit: existingMeta.lastCommit };
|
||||
}
|
||||
|
||||
function commitExists(repoPath: string, commit: string): boolean {
|
||||
try {
|
||||
execFileSync('git', ['cat-file', '-e', `${commit}^{commit}`], {
|
||||
cwd: repoPath,
|
||||
stdio: ['ignore', 'ignore', 'ignore'],
|
||||
});
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
export interface IncrementalSetupResult {
|
||||
/** Files that must be re-parsed in this run. */
|
||||
closure: Set<string>;
|
||||
/** Files deleted on disk since lastCommit (rows must be removed from DB). */
|
||||
deletedFiles: string[];
|
||||
/** Per-file content hashes computed during closure expansion. */
|
||||
newFileHashes: Map<string, string>;
|
||||
/** ChangedFiles result for diagnostics. */
|
||||
changes: ChangedFiles;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute the closure of files to re-parse for an incremental run.
|
||||
*
|
||||
* Algorithm:
|
||||
* 1. git diff lastCommit HEAD ∪ git status → ChangedFiles
|
||||
* 2. closure ← modified ∪ added
|
||||
* 3. For each f in closure: hash(f); if hash differs from previous, query
|
||||
* DB importers and add them to closure. Iterate to fixpoint.
|
||||
*
|
||||
* Throws if `lastCommit` is gone (caller should fall back to full rebuild).
|
||||
*/
|
||||
export async function computeIncrementalClosure(
|
||||
repoPath: string,
|
||||
lastCommit: string,
|
||||
prevSurfaces: Record<string, string>,
|
||||
): Promise<IncrementalSetupResult> {
|
||||
const changes = getChangedFilesSinceCommit(repoPath, lastCommit);
|
||||
|
||||
const initialChangedFiles = new Set<string>([...changes.modified, ...changes.added]);
|
||||
|
||||
// Closure module wants generic parseFile + surfaceFor. v1 uses
|
||||
// file-content hashes (cheap, conservative). The "ParseResult" type
|
||||
// is just the content hash itself.
|
||||
const { closure, newSurfaces } = await computeImporterClosure<string>({
|
||||
initialChangedFiles,
|
||||
prevSurfaces,
|
||||
parseFile: async (filePath) => {
|
||||
const hashes = await computeFileHashes(repoPath, [filePath]);
|
||||
// Missing files (race with delete) → empty hash; counts as "changed".
|
||||
return hashes.get(filePath) ?? '';
|
||||
},
|
||||
surfaceFor: (_filePath, parsed) => parsed,
|
||||
queryImporters: async (filePath) => queryImporters(filePath),
|
||||
});
|
||||
|
||||
return {
|
||||
closure,
|
||||
deletedFiles: changes.deleted,
|
||||
newFileHashes: newSurfaces,
|
||||
changes,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Persist the `incrementalInProgress` dirty flag to meta.json BEFORE any
|
||||
* destructive DB mutation. The flag is cleared on success by overwriting
|
||||
* meta.json with the final state. If the run crashes between, the next
|
||||
* run sees the flag and forces a full rebuild.
|
||||
*/
|
||||
export async function commitIncrementalProgress(
|
||||
storagePath: string,
|
||||
existingMeta: RepoMeta,
|
||||
closure: Set<string>,
|
||||
): Promise<void> {
|
||||
const meta: RepoMeta = {
|
||||
...existingMeta,
|
||||
incrementalInProgress: {
|
||||
closure: [...closure],
|
||||
startedAt: Date.now(),
|
||||
},
|
||||
};
|
||||
await saveMeta(storagePath, meta);
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a subgraph of `ctx.graph` containing ONLY the nodes/edges that
|
||||
* need to be written to LadybugDB in incremental mode:
|
||||
*
|
||||
* - All nodes whose filePath is in `closure` (newly parsed in this run).
|
||||
* - All graph-wide nodes (Community, Process) — they're regenerated by
|
||||
* the communities/processes phases on every run.
|
||||
* - All edges where AT LEAST ONE endpoint is in this set. Edges entirely
|
||||
* between hydrated unchanged-file nodes are NOT included — they're
|
||||
* already in the DB and re-inserting them would PK-conflict.
|
||||
*
|
||||
* This lets us call the existing `loadGraphToLbug` against the filtered
|
||||
* subgraph: the COPY semantics will write only what's missing, while the
|
||||
* unchanged-unchanged DB rows we never deleted stay intact.
|
||||
*/
|
||||
export function extractIncrementalSubgraph(
|
||||
fullGraph: KnowledgeGraph,
|
||||
closure: ReadonlySet<string>,
|
||||
): KnowledgeGraph {
|
||||
const sub = createKnowledgeGraph();
|
||||
|
||||
const isGraphWide = (label: string): boolean => label === 'Community' || label === 'Process';
|
||||
|
||||
// Phase 1: nodes
|
||||
const writableNodeIds = new Set<string>();
|
||||
fullGraph.forEachNode((n: GraphNode) => {
|
||||
const filePath = n.properties?.filePath as string | undefined;
|
||||
const inClosure = filePath ? closure.has(filePath) : false;
|
||||
if (inClosure || isGraphWide(n.label)) {
|
||||
sub.addNode(n);
|
||||
writableNodeIds.add(n.id);
|
||||
}
|
||||
});
|
||||
|
||||
// Phase 2: edges where AT LEAST ONE endpoint is in the writable set.
|
||||
// Edges entirely between hydrated nodes are skipped (already in DB).
|
||||
fullGraph.forEachRelationship((r: GraphRelationship) => {
|
||||
if (writableNodeIds.has(r.sourceId) || writableNodeIds.has(r.targetId)) {
|
||||
sub.addRelationship(r);
|
||||
}
|
||||
});
|
||||
|
||||
return sub;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute the surface signatures for every file in the repo from a graph.
|
||||
* In v1 this is just the per-file content hash (cheap, conservative). v2
|
||||
* will switch to true surface-only signatures derived via `surface.ts`.
|
||||
*
|
||||
* Used after a full rebuild to populate `meta.json.surfaceSignatures` so
|
||||
* the next run can run incrementally.
|
||||
*/
|
||||
export async function computeAllFileSignatures(
|
||||
repoPath: string,
|
||||
filePaths: readonly string[],
|
||||
): Promise<Record<string, string>> {
|
||||
const map = await computeFileHashes(repoPath, filePaths);
|
||||
const out: Record<string, string> = {};
|
||||
for (const [k, v] of map) out[k] = v;
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Merge an updated set of file hashes into a previous snapshot. Used after
|
||||
* an incremental run: `prevSurfaces` is the set in meta.json, `newSurfaces`
|
||||
* is what we computed for closure files this run, and `deletedFiles` are
|
||||
* files that no longer exist on disk.
|
||||
*/
|
||||
export function mergeSurfaceSignatures(
|
||||
prevSurfaces: Record<string, string>,
|
||||
newSurfaces: Map<string, string>,
|
||||
deletedFiles: readonly string[],
|
||||
): Record<string, string> {
|
||||
const merged: Record<string, string> = { ...prevSurfaces };
|
||||
for (const f of deletedFiles) delete merged[f];
|
||||
for (const [f, h] of newSurfaces) merged[f] = h;
|
||||
return merged;
|
||||
}
|
||||
|
|
@ -1,135 +0,0 @@
|
|||
/**
|
||||
* Public-surface signature extraction for incremental indexing.
|
||||
*
|
||||
* The surface signature of a file is a stable hash of everything that could
|
||||
* affect callers in *other* files: the names, signatures, and heritage of
|
||||
* its publicly-visible symbols (functions, classes, methods, interfaces).
|
||||
*
|
||||
* If a file's surface hash is unchanged between runs, no other file's
|
||||
* resolution depends on the changed content — we only need to re-parse the
|
||||
* file itself, not its importers. This is the closure-scoping optimization
|
||||
* driven by `closure.ts`.
|
||||
*
|
||||
* The hash is content-only and excludes formatting, comments, and ordering
|
||||
* (we sort the symbol list before hashing). It does NOT include implementation
|
||||
* bodies — that's the whole point: a body-only edit produces the same surface.
|
||||
*/
|
||||
|
||||
import { createHash } from 'crypto';
|
||||
import type { GraphNode } from 'gitnexus-shared';
|
||||
import type { KnowledgeGraph } from '../graph/types.js';
|
||||
|
||||
/** Node labels whose presence in a file contributes to its public surface. */
|
||||
const SURFACE_LABELS = new Set<string>([
|
||||
'Function',
|
||||
'Class',
|
||||
'Method',
|
||||
'Constructor',
|
||||
'Interface',
|
||||
'TypeAlias',
|
||||
'Enum',
|
||||
'Struct',
|
||||
'Trait',
|
||||
'Namespace',
|
||||
'Module',
|
||||
'Const',
|
||||
'Static',
|
||||
'Property',
|
||||
'Record',
|
||||
'Delegate',
|
||||
'Annotation',
|
||||
'Template',
|
||||
'Union',
|
||||
'Macro',
|
||||
'Typedef',
|
||||
]);
|
||||
|
||||
/**
|
||||
* Extract the surface signature of `filePath` from `graph`.
|
||||
* Returns a stable hex hash; identical-surface inputs → identical output.
|
||||
*/
|
||||
export function extractSurfaceSignature(
|
||||
graph: KnowledgeGraph,
|
||||
filePath: string,
|
||||
): string {
|
||||
const lines: string[] = [];
|
||||
|
||||
// Gather surface-relevant nodes for this file.
|
||||
const surfaceNodes: GraphNode[] = [];
|
||||
graph.forEachNode((node) => {
|
||||
if (
|
||||
node.properties?.filePath === filePath &&
|
||||
SURFACE_LABELS.has(node.label)
|
||||
) {
|
||||
surfaceNodes.push(node);
|
||||
}
|
||||
});
|
||||
|
||||
// Sort deterministically by id so reorder doesn't affect the hash.
|
||||
surfaceNodes.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0));
|
||||
|
||||
for (const n of surfaceNodes) {
|
||||
lines.push(serializeNode(n));
|
||||
}
|
||||
|
||||
// Heritage: edges that fan OUT of this file affect downstream resolution.
|
||||
// (EXTENDS / IMPLEMENTS targets — if the parent class changes, dependents
|
||||
// may resolve methods differently.) These edges are co-located with the
|
||||
// child symbol whose `filePath` matches; we walk them off the source side.
|
||||
const edgeKeys: string[] = [];
|
||||
graph.forEachRelationship((rel) => {
|
||||
if (rel.type !== 'EXTENDS' && rel.type !== 'IMPLEMENTS') return;
|
||||
const src = graph.getNode(rel.sourceId);
|
||||
if (!src) return;
|
||||
if (src.properties?.filePath !== filePath) return;
|
||||
edgeKeys.push(`H ${rel.type} ${rel.sourceId} -> ${rel.targetId}`);
|
||||
});
|
||||
edgeKeys.sort();
|
||||
for (const k of edgeKeys) lines.push(k);
|
||||
|
||||
const h = createHash('sha256');
|
||||
for (const line of lines) {
|
||||
h.update(line);
|
||||
h.update('\n');
|
||||
}
|
||||
return h.digest('hex');
|
||||
}
|
||||
|
||||
/** True iff `current` differs from `prev` (or `prev` is undefined). */
|
||||
export function surfaceChanged(
|
||||
prev: string | undefined,
|
||||
current: string,
|
||||
): boolean {
|
||||
return prev === undefined || prev !== current;
|
||||
}
|
||||
|
||||
/**
|
||||
* Serialize a single surface node to a stable string. We include label,
|
||||
* name, parameter count + types, return type, and visibility-style flags
|
||||
* — anything that could affect how a caller resolves against this node.
|
||||
*
|
||||
* We DO NOT include line numbers, file offsets, or any other position
|
||||
* data: surface should be invariant under whitespace/formatting changes.
|
||||
*/
|
||||
function serializeNode(node: GraphNode): string {
|
||||
const p = (node.properties ?? {}) as Record<string, unknown>;
|
||||
const parts: string[] = [
|
||||
`N`,
|
||||
node.label,
|
||||
String(p.name ?? ''),
|
||||
`id=${node.id}`,
|
||||
];
|
||||
if (p.parameterCount !== undefined) parts.push(`pc=${String(p.parameterCount)}`);
|
||||
if (Array.isArray(p.parameterTypes)) {
|
||||
parts.push(`pt=${(p.parameterTypes as string[]).join(',')}`);
|
||||
}
|
||||
if (p.returnType !== undefined) parts.push(`rt=${String(p.returnType)}`);
|
||||
if (p.visibility !== undefined) parts.push(`v=${String(p.visibility)}`);
|
||||
if (p.isStatic) parts.push('static');
|
||||
if (p.isAbstract) parts.push('abstract');
|
||||
if (p.isReadonly) parts.push('readonly');
|
||||
if (p.isAsync) parts.push('async');
|
||||
// `level` (inheritance depth marker for methods) affects MRO
|
||||
if (p.level !== undefined) parts.push(`lvl=${String(p.level)}`);
|
||||
return parts.join('|');
|
||||
}
|
||||
|
|
@ -1,91 +0,0 @@
|
|||
/**
|
||||
* Phase: hydrate
|
||||
*
|
||||
* In incremental-indexing mode, populates `ctx.graph` with all nodes and
|
||||
* relationships belonging to files OUTSIDE the current closure (i.e., files
|
||||
* that didn't change). This way the parse phase only needs to re-emit nodes
|
||||
* and edges for the closure files, while downstream phases (mro, communities,
|
||||
* processes) see a fully-populated graph and produce results equivalent to a
|
||||
* full rebuild.
|
||||
*
|
||||
* No-op in full-rebuild mode: when `ctx.options.filesToParse` is unset,
|
||||
* the pipeline starts with an empty graph and parses every file (existing
|
||||
* behavior).
|
||||
*
|
||||
* @deps structure
|
||||
* @reads allPaths (from structure)
|
||||
* @writes graph (every node/edge belonging to unchanged files)
|
||||
*/
|
||||
|
||||
import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js';
|
||||
import { getPhaseOutput } from './types.js';
|
||||
import type { StructureOutput } from './structure.js';
|
||||
import { loadGraphFromLbug } from '../../lbug/lbug-adapter.js';
|
||||
import { isDev } from '../utils/env.js';
|
||||
import { logger } from '../../logger.js';
|
||||
|
||||
export interface HydrateOutput {
|
||||
/** True when this run actually loaded prior state (incremental). */
|
||||
readonly hydrated: boolean;
|
||||
/** Number of nodes loaded from DB (0 in full-rebuild mode). */
|
||||
readonly nodesLoaded: number;
|
||||
/** Number of relationships loaded from DB (0 in full-rebuild mode). */
|
||||
readonly edgesLoaded: number;
|
||||
}
|
||||
|
||||
export const hydratePhase: PipelinePhase<HydrateOutput> = {
|
||||
name: 'hydrate',
|
||||
deps: ['structure'],
|
||||
|
||||
async execute(
|
||||
ctx: PipelineContext,
|
||||
deps: ReadonlyMap<string, PhaseResult<unknown>>,
|
||||
): Promise<HydrateOutput> {
|
||||
const filesToParse = ctx.options?.filesToParse;
|
||||
|
||||
// Full-rebuild mode: nothing to hydrate.
|
||||
if (!filesToParse) {
|
||||
return { hydrated: false, nodesLoaded: 0, edgesLoaded: 0 };
|
||||
}
|
||||
|
||||
const { allPaths, totalFiles } = getPhaseOutput<StructureOutput>(deps, 'structure');
|
||||
|
||||
// Compute the unchanged complement: every scanned path NOT in the closure.
|
||||
const unchanged = new Set<string>();
|
||||
for (const p of allPaths) {
|
||||
if (!filesToParse.has(p)) unchanged.add(p);
|
||||
}
|
||||
|
||||
ctx.onProgress({
|
||||
phase: 'hydrate',
|
||||
percent: 22,
|
||||
message: `Hydrating ${unchanged.size} unchanged files from index...`,
|
||||
stats: { filesProcessed: 0, totalFiles, nodesCreated: ctx.graph.nodeCount },
|
||||
});
|
||||
|
||||
const result = await loadGraphFromLbug(ctx.graph, unchanged);
|
||||
|
||||
if (isDev) {
|
||||
logger.info(
|
||||
`💧 Hydrate: ${result.nodesLoaded} nodes, ${result.edgesLoaded} edges loaded for ${unchanged.size} unchanged files (closure: ${filesToParse.size})`,
|
||||
);
|
||||
}
|
||||
|
||||
ctx.onProgress({
|
||||
phase: 'hydrate',
|
||||
percent: 25,
|
||||
message: `Hydrated ${result.nodesLoaded} nodes from previous index`,
|
||||
stats: {
|
||||
filesProcessed: unchanged.size,
|
||||
totalFiles,
|
||||
nodesCreated: ctx.graph.nodeCount,
|
||||
},
|
||||
});
|
||||
|
||||
return {
|
||||
hydrated: true,
|
||||
nodesLoaded: result.nodesLoaded,
|
||||
edgesLoaded: result.edgesLoaded,
|
||||
};
|
||||
},
|
||||
};
|
||||
|
|
@ -9,7 +9,6 @@
|
|||
|
||||
export { scanPhase, type ScanOutput } from './scan.js';
|
||||
export { structurePhase, type StructureOutput } from './structure.js';
|
||||
export { hydratePhase, type HydrateOutput } from './hydrate.js';
|
||||
export { markdownPhase, type MarkdownOutput } from './markdown.js';
|
||||
export { cobolPhase, type CobolOutput } from './cobol.js';
|
||||
export { parsePhase, type ParseOutput } from './parse.js';
|
||||
|
|
|
|||
|
|
@ -96,18 +96,9 @@ export const parsePhase: PipelinePhase<ParseOutput> = {
|
|||
'structure',
|
||||
);
|
||||
|
||||
// Incremental-indexing filter: when `filesToParse` is set, only the
|
||||
// closure files get parsed in this run. Files outside the closure had
|
||||
// their nodes/edges pre-loaded by the `hydrate` phase. See
|
||||
// docs/superpowers/specs/2026-05-10-incremental-indexing-design.md.
|
||||
const filesToParse = ctx.options?.filesToParse;
|
||||
const targetScanned = filesToParse
|
||||
? scannedFiles.filter((f) => filesToParse.has(f.path))
|
||||
: scannedFiles;
|
||||
|
||||
const result = await runChunkedParseAndResolve(
|
||||
ctx.graph,
|
||||
targetScanned,
|
||||
scannedFiles,
|
||||
allPaths,
|
||||
totalFiles,
|
||||
ctx.repoPath,
|
||||
|
|
|
|||
|
|
@ -23,7 +23,6 @@ import {
|
|||
getPhaseOutput,
|
||||
scanPhase,
|
||||
structurePhase,
|
||||
hydratePhase,
|
||||
markdownPhase,
|
||||
cobolPhase,
|
||||
parsePhase,
|
||||
|
|
@ -56,18 +55,6 @@ export interface PipelineOptions {
|
|||
minFiles?: number;
|
||||
minBytes?: number;
|
||||
};
|
||||
/**
|
||||
* Incremental-indexing mode: when set, the parse phase only re-parses
|
||||
* files in this set, and the new `hydrate` phase pre-loads node/edge
|
||||
* state for everything else from the existing LadybugDB index.
|
||||
*
|
||||
* Unset (the default) → full-rebuild mode: parse phase processes every
|
||||
* scanned file and hydrate is a no-op. Set by `runFullAnalysis` when it
|
||||
* detects an eligible incremental run; never set by callers directly.
|
||||
*
|
||||
* See `docs/superpowers/specs/2026-05-10-incremental-indexing-design.md`.
|
||||
*/
|
||||
filesToParse?: ReadonlySet<string>;
|
||||
}
|
||||
|
||||
// ── Phase registry ─────────────────────────────────────────────────────────
|
||||
|
|
@ -87,7 +74,6 @@ function buildPhaseList(options?: PipelineOptions): PipelinePhase[] {
|
|||
const phases: PipelinePhase[] = [
|
||||
scanPhase,
|
||||
structurePhase,
|
||||
hydratePhase,
|
||||
markdownPhase,
|
||||
cobolPhase,
|
||||
parsePhase,
|
||||
|
|
|
|||
|
|
@ -1204,245 +1204,6 @@ export const deleteNodesForFile = async (
|
|||
|
||||
export const getEmbeddingTableName = (): string => EMBEDDING_TABLE_NAME;
|
||||
|
||||
// ============================================================================
|
||||
// Incremental indexing: DB → KnowledgeGraph hydration
|
||||
// ============================================================================
|
||||
|
||||
/**
|
||||
* Node tables that have a `filePath` column and are eligible for hydration.
|
||||
* `Community` and `Process` are graph-wide (no filePath) and are ALWAYS
|
||||
* regenerated by the communities/processes phases — we never load them back.
|
||||
*/
|
||||
const HYDRATABLE_NODE_TABLES: readonly NodeTableName[] = NODE_TABLES.filter(
|
||||
(t) => t !== 'Community' && t !== 'Process',
|
||||
);
|
||||
|
||||
/** Per-table extra columns to project beyond the base (id, name, filePath). */
|
||||
const TABLE_EXTRA_COLUMNS: Record<string, string[]> = {
|
||||
File: ['content'],
|
||||
Folder: [],
|
||||
Function: ['startLine', 'endLine', 'isExported', 'content', 'description'],
|
||||
Class: ['startLine', 'endLine', 'isExported', 'content', 'description'],
|
||||
Interface: ['startLine', 'endLine', 'isExported', 'content', 'description'],
|
||||
Method: [
|
||||
'startLine',
|
||||
'endLine',
|
||||
'isExported',
|
||||
'content',
|
||||
'description',
|
||||
'parameterCount',
|
||||
'returnType',
|
||||
],
|
||||
CodeElement: ['startLine', 'endLine', 'isExported', 'content', 'description'],
|
||||
Section: ['startLine', 'endLine', 'level', 'content', 'description'],
|
||||
Property: ['startLine', 'endLine', 'content', 'description', 'declaredType'],
|
||||
Route: ['responseKeys', 'errorKeys', 'middleware'],
|
||||
Tool: ['description'],
|
||||
};
|
||||
|
||||
/** All other tables share the CODE_ELEMENT_BASE schema (startLine/endLine/content/description). */
|
||||
const DEFAULT_EXTRA_COLUMNS = ['startLine', 'endLine', 'content', 'description'];
|
||||
|
||||
const escapeFilePath = (s: string): string =>
|
||||
s.replace(/\\/g, '\\\\').replace(/'/g, "''");
|
||||
|
||||
/**
|
||||
* Hydrate `graph` with all nodes (and their relationships) belonging to files
|
||||
* in `unchangedFilePaths`. Used by the incremental-indexing pipeline to
|
||||
* pre-populate the in-memory graph with everything that didn't change, so
|
||||
* downstream phases (mro, communities, processes) see a complete picture.
|
||||
*
|
||||
* Skips Community/Process labels and their MEMBER_OF / STEP_IN_PROCESS edges
|
||||
* — those are graph-wide and will be regenerated from scratch by the
|
||||
* pipeline's downstream phases.
|
||||
*
|
||||
* Returns counts for logging/diagnostics.
|
||||
*/
|
||||
export const loadGraphFromLbug = async (
|
||||
graph: KnowledgeGraph,
|
||||
unchangedFilePaths: ReadonlySet<string>,
|
||||
): Promise<{ nodesLoaded: number; edgesLoaded: number }> => {
|
||||
if (!conn) {
|
||||
throw new Error('LadybugDB not initialized. Call initLbug first.');
|
||||
}
|
||||
if (unchangedFilePaths.size === 0) {
|
||||
return { nodesLoaded: 0, edgesLoaded: 0 };
|
||||
}
|
||||
|
||||
let nodesLoaded = 0;
|
||||
let edgesLoaded = 0;
|
||||
|
||||
// Track which node IDs we successfully loaded so the relationship pass
|
||||
// can verify both endpoints are present (cheaper than a DB-side join).
|
||||
const loadedNodeIds = new Set<string>();
|
||||
|
||||
// ── 1. Hydrate nodes per table ─────────────────────────────────────────
|
||||
// Chunk filePaths to keep query size manageable on huge repos.
|
||||
const CHUNK = 200;
|
||||
const filePathArr = [...unchangedFilePaths];
|
||||
|
||||
for (const tableName of HYDRATABLE_NODE_TABLES) {
|
||||
const t = escapeTableName(tableName);
|
||||
const extras = TABLE_EXTRA_COLUMNS[tableName] ?? DEFAULT_EXTRA_COLUMNS;
|
||||
// Always include base columns: id, name, filePath
|
||||
const cols = ['id', 'name', 'filePath', ...extras];
|
||||
const projection = cols.map((c) => `n.${c} AS ${c}`).join(', ');
|
||||
|
||||
for (let i = 0; i < filePathArr.length; i += CHUNK) {
|
||||
const batch = filePathArr.slice(i, i + CHUNK);
|
||||
const inList = batch.map((p) => `'${escapeFilePath(p)}'`).join(',');
|
||||
const cypher = `MATCH (n:${t}) WHERE n.filePath IN [${inList}] RETURN ${projection}`;
|
||||
|
||||
try {
|
||||
const queryResult = await conn.query(cypher);
|
||||
const result = Array.isArray(queryResult) ? queryResult[0] : queryResult;
|
||||
const rows = await result.getAll();
|
||||
for (const row of rows) {
|
||||
const id = typeof row.id === 'string' ? row.id : String(row.id ?? '');
|
||||
if (!id) continue;
|
||||
const properties: Record<string, unknown> = {};
|
||||
for (const c of cols) {
|
||||
const v = row[c];
|
||||
if (v !== undefined && v !== null) properties[c] = v;
|
||||
}
|
||||
// Required for downstream phases: ensure name/filePath always set.
|
||||
if (properties.name === undefined) properties.name = '';
|
||||
graph.addNode({
|
||||
id,
|
||||
label: tableName as unknown as import('gitnexus-shared').NodeLabel,
|
||||
properties: properties as import('gitnexus-shared').NodeProperties,
|
||||
});
|
||||
loadedNodeIds.add(id);
|
||||
nodesLoaded++;
|
||||
}
|
||||
} catch {
|
||||
// Some tables may not exist in the schema or may be empty — that's fine.
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── 2. Hydrate relationships ───────────────────────────────────────────
|
||||
// We pull all CodeRelation rows whose endpoints are nodes we just loaded.
|
||||
// Using SRC.filePath IN [...] AND TGT.filePath IN [...] is precise but
|
||||
// requires LadybugDB to traverse with both endpoint constraints. We
|
||||
// exclude graph-wide edge types (MEMBER_OF, STEP_IN_PROCESS) entirely —
|
||||
// they'll be regenerated.
|
||||
//
|
||||
// Strategy: for each (src filePath chunk × tgt filePath chunk) we'd risk
|
||||
// O(n²) chunking. Instead, we fetch by source-chunk, then verify the
|
||||
// target endpoint is in `loadedNodeIds` JS-side. This keeps the DB query
|
||||
// bounded by source chunks while preserving correctness.
|
||||
for (let i = 0; i < filePathArr.length; i += CHUNK) {
|
||||
const batch = filePathArr.slice(i, i + CHUNK);
|
||||
const inList = batch.map((p) => `'${escapeFilePath(p)}'`).join(',');
|
||||
const cypher = `
|
||||
MATCH (a)-[r:${REL_TABLE_NAME}]->(b)
|
||||
WHERE a.filePath IN [${inList}]
|
||||
AND r.type <> 'MEMBER_OF'
|
||||
AND r.type <> 'STEP_IN_PROCESS'
|
||||
RETURN a.id AS src, b.id AS dst, r.type AS type,
|
||||
r.confidence AS confidence, r.reason AS reason, r.step AS step
|
||||
`;
|
||||
|
||||
try {
|
||||
const queryResult = await conn.query(cypher);
|
||||
const result = Array.isArray(queryResult) ? queryResult[0] : queryResult;
|
||||
const rows = await result.getAll();
|
||||
for (const row of rows) {
|
||||
const src = typeof row.src === 'string' ? row.src : '';
|
||||
const dst = typeof row.dst === 'string' ? row.dst : '';
|
||||
if (!src || !dst) continue;
|
||||
// The other endpoint must be a node we loaded — otherwise the edge
|
||||
// crosses into a closure file (will be re-emitted) or a dropped
|
||||
// graph-wide node (Community/Process), and we skip it.
|
||||
if (!loadedNodeIds.has(dst)) continue;
|
||||
const relType = typeof row.type === 'string' ? row.type : '';
|
||||
if (!relType) continue;
|
||||
const relId = `${src}_${relType}_${dst}`;
|
||||
graph.addRelationship({
|
||||
id: relId,
|
||||
sourceId: src,
|
||||
targetId: dst,
|
||||
type: relType as import('gitnexus-shared').RelationshipType,
|
||||
confidence: typeof row.confidence === 'number' ? row.confidence : 1.0,
|
||||
reason: typeof row.reason === 'string' ? row.reason : '',
|
||||
step:
|
||||
typeof row.step === 'number' && row.step !== 0 ? row.step : undefined,
|
||||
});
|
||||
edgesLoaded++;
|
||||
}
|
||||
} catch {
|
||||
// Continue on chunk failure — best-effort hydration.
|
||||
}
|
||||
}
|
||||
|
||||
return { nodesLoaded, edgesLoaded };
|
||||
};
|
||||
|
||||
/**
|
||||
* Query the IMPORTS edge table for all files that import `targetFilePath`.
|
||||
* Used by the incremental-indexing closure-expansion logic to find files
|
||||
* whose resolution may be stale when `targetFilePath`'s public surface
|
||||
* changes.
|
||||
*
|
||||
* Returns repo-relative paths (i.e. the source file's `filePath`), de-duplicated.
|
||||
*/
|
||||
export const queryImporters = async (targetFilePath: string): Promise<string[]> => {
|
||||
if (!conn) {
|
||||
throw new Error('LadybugDB not initialized. Call initLbug first.');
|
||||
}
|
||||
const escaped = escapeFilePath(targetFilePath);
|
||||
const cypher = `
|
||||
MATCH (a)-[r:${REL_TABLE_NAME}]->(b)
|
||||
WHERE r.type = 'IMPORTS' AND b.filePath = '${escaped}'
|
||||
RETURN DISTINCT a.filePath AS importer
|
||||
`;
|
||||
try {
|
||||
const queryResult = await conn.query(cypher);
|
||||
const result = Array.isArray(queryResult) ? queryResult[0] : queryResult;
|
||||
const rows = await result.getAll();
|
||||
const out: string[] = [];
|
||||
for (const row of rows) {
|
||||
const p = row.importer;
|
||||
if (typeof p === 'string' && p.length > 0) out.push(p);
|
||||
}
|
||||
return out;
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Delete all Community and Process nodes and their MEMBER_OF /
|
||||
* STEP_IN_PROCESS edges. Used at the start of an incremental run so the
|
||||
* communities/processes phases can regenerate them on the merged graph.
|
||||
*/
|
||||
export const deleteAllCommunitiesAndProcesses = async (): Promise<{
|
||||
nodesDeleted: number;
|
||||
}> => {
|
||||
if (!conn) {
|
||||
throw new Error('LadybugDB not initialized. Call initLbug first.');
|
||||
}
|
||||
let nodesDeleted = 0;
|
||||
for (const label of ['Community', 'Process']) {
|
||||
try {
|
||||
const countResult = await conn.query(
|
||||
`MATCH (n:${label}) RETURN count(n) AS cnt`,
|
||||
);
|
||||
const result = Array.isArray(countResult) ? countResult[0] : countResult;
|
||||
const rows = await result.getAll();
|
||||
const count = Number(rows[0]?.cnt ?? rows[0]?.[0] ?? 0);
|
||||
if (count > 0) {
|
||||
await conn.query(`MATCH (n:${label}) DETACH DELETE n`);
|
||||
nodesDeleted += count;
|
||||
}
|
||||
} catch {
|
||||
// table may not exist yet
|
||||
}
|
||||
}
|
||||
return { nodesDeleted };
|
||||
};
|
||||
|
||||
// ============================================================================
|
||||
// Full-Text Search (FTS) Functions
|
||||
// ============================================================================
|
||||
|
|
|
|||
|
|
@ -11,7 +11,6 @@
|
|||
|
||||
import path from 'path';
|
||||
import fs from 'fs/promises';
|
||||
import { execFileSync } from 'child_process';
|
||||
import { runPipelineFromRepo } from './ingestion/pipeline.js';
|
||||
import {
|
||||
initLbug,
|
||||
|
|
@ -21,8 +20,6 @@ import {
|
|||
executeWithReusedStatement,
|
||||
closeLbug,
|
||||
loadCachedEmbeddings,
|
||||
deleteNodesForFile,
|
||||
deleteAllCommunitiesAndProcesses,
|
||||
} from './lbug/lbug-adapter.js';
|
||||
import { createSearchFTSIndexes } from './search/fts-indexes.js';
|
||||
import {
|
||||
|
|
@ -32,7 +29,6 @@ import {
|
|||
ensureGitNexusIgnored,
|
||||
registerRepo,
|
||||
cleanupOldKuzuFiles,
|
||||
INCREMENTAL_SCHEMA_VERSION,
|
||||
} from '../storage/repo-manager.js';
|
||||
import {
|
||||
getCurrentCommit,
|
||||
|
|
@ -45,14 +41,6 @@ import type { CachedEmbedding } from './embeddings/types.js';
|
|||
import { generateAIContextFiles } from '../cli/ai-context.js';
|
||||
import { EMBEDDING_TABLE_NAME } from './lbug/schema.js';
|
||||
import { STALE_HASH_SENTINEL } from './lbug/schema.js';
|
||||
import {
|
||||
isIncrementalEligible,
|
||||
computeIncrementalClosure,
|
||||
commitIncrementalProgress,
|
||||
extractIncrementalSubgraph,
|
||||
mergeSurfaceSignatures,
|
||||
computeAllFileSignatures,
|
||||
} from './incremental/orchestrator.js';
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public types
|
||||
|
|
@ -192,65 +180,20 @@ export async function runFullAnalysis(
|
|||
if (existingMeta && !options.force && existingMeta.lastCommit === currentCommit) {
|
||||
// Non-git folders have currentCommit = '' — always rebuild since we can't detect changes
|
||||
if (currentCommit !== '') {
|
||||
// For git repos, even if HEAD matches lastCommit, the working tree
|
||||
// may have uncommitted changes. Only short-circuit when the working
|
||||
// tree is also clean.
|
||||
const dirty = hasDirtyTree(repoPath);
|
||||
if (!dirty) {
|
||||
await ensureGitNexusIgnored(repoPath);
|
||||
return {
|
||||
// `resolveRepoIdentityRoot` collapses worktree roots to the
|
||||
// canonical repo basename (#1259) but leaves arbitrary subdirs
|
||||
// and `--skip-git` paths unchanged (#1232/#1233 intent preserved).
|
||||
repoName:
|
||||
options.registryName ??
|
||||
getInferredRepoName(repoPath) ??
|
||||
path.basename(resolveRepoIdentityRoot(repoPath)),
|
||||
repoPath,
|
||||
stats: existingMeta.stats ?? {},
|
||||
alreadyUpToDate: true,
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── Incremental branch ─────────────────────────────────────────────
|
||||
// Try the incremental path first. Falls through to full rebuild on:
|
||||
// - --force
|
||||
// - missing/old meta.json
|
||||
// - lastCommit gone (rebased)
|
||||
// - schemaVersion mismatch
|
||||
// - non-git repo
|
||||
// - any error during incremental setup
|
||||
const eligibility = isIncrementalEligible(repoPath, existingMeta, options.force);
|
||||
if (eligibility.eligible && existingMeta) {
|
||||
try {
|
||||
const result = await runIncrementalBranch(
|
||||
await ensureGitNexusIgnored(repoPath);
|
||||
return {
|
||||
// `resolveRepoIdentityRoot` collapses worktree roots to the
|
||||
// canonical repo basename (#1259) but leaves arbitrary subdirs
|
||||
// and `--skip-git` paths unchanged (#1232/#1233 intent preserved).
|
||||
repoName:
|
||||
options.registryName ??
|
||||
getInferredRepoName(repoPath) ??
|
||||
path.basename(resolveRepoIdentityRoot(repoPath)),
|
||||
repoPath,
|
||||
storagePath,
|
||||
lbugPath,
|
||||
existingMeta,
|
||||
currentCommit,
|
||||
options,
|
||||
callbacks,
|
||||
);
|
||||
if (result) return result;
|
||||
// result === null → falls through to full rebuild (e.g. setup failed)
|
||||
} catch (err) {
|
||||
log(
|
||||
`Incremental run failed (${(err as Error).message}); ` +
|
||||
`next run will full-rebuild via dirty-flag.`,
|
||||
);
|
||||
// Re-throw — the dirty flag is set; next run forces full rebuild.
|
||||
try {
|
||||
await closeLbug();
|
||||
} catch {
|
||||
/* swallow */
|
||||
}
|
||||
throw err;
|
||||
stats: existingMeta.stats ?? {},
|
||||
alreadyUpToDate: true,
|
||||
};
|
||||
}
|
||||
} else if (existingMeta) {
|
||||
log(`Incremental skipped: ${eligibility.reason ?? 'unknown reason'}`);
|
||||
}
|
||||
|
||||
// ── Cache embeddings from existing index before rebuild ────────────
|
||||
|
|
@ -511,32 +454,6 @@ export async function runFullAnalysis(
|
|||
const effectiveSemanticMode =
|
||||
semanticMode ??
|
||||
(runtimeCapabilities.semanticMode === 'vector-index' ? 'vector-index' : 'exact-scan');
|
||||
|
||||
// Compute per-file surface signatures so the next run can be incremental.
|
||||
// v1 uses content-hash (cheap, conservative). v2 will switch to a
|
||||
// surface-only signature so body-only edits don't expand the closure.
|
||||
let surfaceSignatures: Record<string, string> | undefined;
|
||||
if (hasGitDir(repoPath)) {
|
||||
try {
|
||||
const allFilePaths: string[] = [];
|
||||
pipelineResult.graph.forEachNode((n) => {
|
||||
const fp = n.properties?.filePath as string | undefined;
|
||||
if (fp && (n.label === 'File' || n.label === 'Folder')) {
|
||||
// File nodes carry their own path; we want every distinct repo-relative
|
||||
// file path that participated in indexing. Use File nodes as the
|
||||
// authoritative source.
|
||||
if (n.label === 'File') allFilePaths.push(fp);
|
||||
}
|
||||
});
|
||||
if (allFilePaths.length > 0) {
|
||||
surfaceSignatures = await computeAllFileSignatures(repoPath, allFilePaths);
|
||||
}
|
||||
} catch {
|
||||
/* surface signatures are best-effort; their absence just means the
|
||||
* next run will fall back to full rebuild. */
|
||||
}
|
||||
}
|
||||
|
||||
const meta = {
|
||||
repoPath,
|
||||
lastCommit: currentCommit,
|
||||
|
|
@ -548,14 +465,6 @@ export async function runFullAnalysis(
|
|||
// origin remote, which is fine: paths-only repos behave as
|
||||
// before.
|
||||
remoteUrl: hasGitDir(repoPath) ? getRemoteUrl(repoPath) : undefined,
|
||||
// Incremental-indexing fields — populated for git repos only. The
|
||||
// next analyze run reads these to decide whether to take the
|
||||
// incremental path. See the Incremental branch above.
|
||||
schemaVersion: surfaceSignatures ? INCREMENTAL_SCHEMA_VERSION : undefined,
|
||||
surfaceSignatures,
|
||||
incrementalInProgress: undefined as
|
||||
| { closure: string[]; startedAt: number }
|
||||
| undefined,
|
||||
stats: {
|
||||
files: pipelineResult.totalFileCount,
|
||||
nodes: stats.nodes,
|
||||
|
|
@ -645,242 +554,3 @@ export async function runFullAnalysis(
|
|||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// Incremental analysis branch — invoked from runFullAnalysis when eligible.
|
||||
// See docs/superpowers/specs/2026-05-10-incremental-indexing-design.md.
|
||||
// ===========================================================================
|
||||
|
||||
/**
|
||||
* Cheap check: does the working tree have uncommitted changes? Used to
|
||||
* decide whether the "lastCommit == HEAD" early-exit is safe to take.
|
||||
*/
|
||||
function hasDirtyTree(repoPath: string): boolean {
|
||||
try {
|
||||
const out = execFileSync('git', ['status', '--porcelain'], {
|
||||
cwd: repoPath,
|
||||
stdio: ['ignore', 'pipe', 'ignore'],
|
||||
encoding: 'utf8',
|
||||
});
|
||||
return out.trim().length > 0;
|
||||
} catch {
|
||||
// If git status fails for any reason, conservatively assume dirty so
|
||||
// we don't accidentally short-circuit a real change.
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute the incremental branch of `runFullAnalysis`. Returns null when
|
||||
* setup determines a full rebuild is needed instead (caller falls through).
|
||||
*
|
||||
* Throws if the run fails after the dirty flag is set — caller is
|
||||
* responsible for surfacing the error; the next run will detect the dirty
|
||||
* flag and force a full rebuild.
|
||||
*/
|
||||
async function runIncrementalBranch(
|
||||
repoPath: string,
|
||||
storagePath: string,
|
||||
lbugPath: string,
|
||||
existingMeta: import('../storage/repo-manager.js').RepoMeta,
|
||||
currentCommit: string,
|
||||
options: AnalyzeOptions,
|
||||
callbacks: AnalyzeCallbacks,
|
||||
): Promise<AnalyzeResult | null> {
|
||||
const log = (msg: string) => callbacks.onLog?.(msg);
|
||||
const progress = (phase: string, percent: number, message: string) =>
|
||||
callbacks.onProgress(phase, percent, message);
|
||||
|
||||
log(
|
||||
`Incremental: probing for changes since ${existingMeta.lastCommit.slice(0, 7)}...`,
|
||||
);
|
||||
|
||||
// 1. Compute closure (parses each file's content hash, walks DB importers).
|
||||
let setup: Awaited<ReturnType<typeof computeIncrementalClosure>>;
|
||||
try {
|
||||
// Open the existing DB so closure expansion can query the IMPORTS edges.
|
||||
await initLbug(lbugPath);
|
||||
setup = await computeIncrementalClosure(
|
||||
repoPath,
|
||||
existingMeta.lastCommit,
|
||||
existingMeta.surfaceSignatures ?? {},
|
||||
);
|
||||
} catch (e) {
|
||||
try {
|
||||
await closeLbug();
|
||||
} catch {
|
||||
/* swallow */
|
||||
}
|
||||
log(
|
||||
`Incremental setup failed: ${(e as Error).message}. Falling back to full rebuild.`,
|
||||
);
|
||||
return null;
|
||||
}
|
||||
|
||||
// 2. No changes? Update lastCommit and return early.
|
||||
if (setup.closure.size === 0 && setup.deletedFiles.length === 0) {
|
||||
log('Incremental: no file changes detected — refreshing meta only.');
|
||||
try {
|
||||
await closeLbug();
|
||||
} catch {
|
||||
/* swallow */
|
||||
}
|
||||
const meta: import('../storage/repo-manager.js').RepoMeta = {
|
||||
...existingMeta,
|
||||
lastCommit: currentCommit,
|
||||
indexedAt: new Date().toISOString(),
|
||||
};
|
||||
await saveMeta(storagePath, meta);
|
||||
await ensureGitNexusIgnored(repoPath);
|
||||
return {
|
||||
repoName:
|
||||
options.registryName ??
|
||||
getInferredRepoName(repoPath) ??
|
||||
path.basename(resolveRepoIdentityRoot(repoPath)),
|
||||
repoPath,
|
||||
stats: existingMeta.stats ?? {},
|
||||
alreadyUpToDate: true,
|
||||
};
|
||||
}
|
||||
|
||||
log(
|
||||
`Incremental: closure=${setup.closure.size} (changed=${setup.changes.modified.length} ` +
|
||||
`+ added=${setup.changes.added.length} + importers=${setup.closure.size - setup.changes.modified.length - setup.changes.added.length}), ` +
|
||||
`deleted=${setup.deletedFiles.length}`,
|
||||
);
|
||||
|
||||
// 3. Mark dirty BEFORE any DB mutation. Closes lbug temporarily to
|
||||
// release the connection while saveMeta writes.
|
||||
await commitIncrementalProgress(storagePath, existingMeta, setup.closure);
|
||||
|
||||
// 4. Delete stale rows from the DB.
|
||||
progress('lbug', 5, 'Removing stale rows for changed files...');
|
||||
for (const file of setup.closure) {
|
||||
try {
|
||||
await deleteNodesForFile(file);
|
||||
} catch {
|
||||
/* file may not have been indexed yet — fine */
|
||||
}
|
||||
}
|
||||
for (const file of setup.deletedFiles) {
|
||||
try {
|
||||
await deleteNodesForFile(file);
|
||||
} catch {
|
||||
/* fine */
|
||||
}
|
||||
}
|
||||
// Always wipe Community / Process — they're regenerated by downstream
|
||||
// pipeline phases and must come from the merged graph for correctness
|
||||
// (Leiden runs on the FULL hydrated + parsed graph).
|
||||
await deleteAllCommunitiesAndProcesses();
|
||||
|
||||
// 5. Run the pipeline with filesToParse set so:
|
||||
// - hydrate phase fills ctx.graph with unchanged-file nodes from DB
|
||||
// - parse phase only parses closure files
|
||||
// - downstream phases (mro, communities, processes) see the full graph
|
||||
const pipelineResult = await runPipelineFromRepo(
|
||||
repoPath,
|
||||
(p) => {
|
||||
const phaseLabel = PHASE_LABELS[p.phase] || p.phase;
|
||||
const scaled = 10 + Math.round(p.percent * 0.55); // 10–65%
|
||||
const message = p.detail ? `${p.message || phaseLabel} (${p.detail})` : p.message || phaseLabel;
|
||||
progress(p.phase, scaled, message);
|
||||
},
|
||||
{ filesToParse: setup.closure },
|
||||
);
|
||||
|
||||
// 6. Extract the subgraph that needs to be written: closure-file nodes
|
||||
// + graph-wide nodes + edges incident to them. Hydrated unchanged-file
|
||||
// nodes are NOT in this subgraph — their rows are still in DB.
|
||||
progress('lbug', 70, 'Writing incremental updates to LadybugDB...');
|
||||
const subgraph = extractIncrementalSubgraph(pipelineResult.graph, setup.closure);
|
||||
await loadGraphToLbug(subgraph, repoPath, storagePath, (msg) => {
|
||||
progress('lbug', 80, msg);
|
||||
});
|
||||
|
||||
// 7. Recreate FTS indexes (cheap; full rebuild over the merged DB state).
|
||||
progress('fts', 90, 'Refreshing search indexes...');
|
||||
try {
|
||||
await createSearchFTSIndexes();
|
||||
} catch {
|
||||
/* FTS is best-effort; log only */
|
||||
}
|
||||
|
||||
// 8. Compute final stats from the live DB state.
|
||||
const stats = await getLbugStats();
|
||||
|
||||
// 9. Compute new meta (merge surface signatures, clear dirty flag).
|
||||
const mergedSurfaces = mergeSurfaceSignatures(
|
||||
existingMeta.surfaceSignatures ?? {},
|
||||
setup.newFileHashes,
|
||||
setup.deletedFiles,
|
||||
);
|
||||
|
||||
const newMeta: import('../storage/repo-manager.js').RepoMeta = {
|
||||
...existingMeta,
|
||||
lastCommit: currentCommit,
|
||||
indexedAt: new Date().toISOString(),
|
||||
remoteUrl: getRemoteUrl(repoPath) ?? existingMeta.remoteUrl,
|
||||
schemaVersion: INCREMENTAL_SCHEMA_VERSION,
|
||||
surfaceSignatures: mergedSurfaces,
|
||||
incrementalInProgress: undefined, // explicit clear
|
||||
stats: {
|
||||
...(existingMeta.stats ?? {}),
|
||||
files: pipelineResult.totalFileCount,
|
||||
nodes: stats.nodes,
|
||||
edges: stats.edges,
|
||||
communities: pipelineResult.communityResult?.stats.totalCommunities,
|
||||
processes: pipelineResult.processResult?.stats.totalProcesses,
|
||||
},
|
||||
};
|
||||
|
||||
// 10. Persist meta + register repo.
|
||||
await saveMeta(storagePath, newMeta);
|
||||
const projectName = await registerRepo(repoPath, newMeta, {
|
||||
name: options.registryName,
|
||||
allowDuplicateName: options.allowDuplicateName,
|
||||
});
|
||||
|
||||
// 11. Best-effort AI-context regeneration so AGENTS.md/CLAUDE.md stay
|
||||
// in sync with the post-incremental graph state.
|
||||
let aggregatedClusterCount = 0;
|
||||
if (pipelineResult.communityResult?.communities) {
|
||||
const groups = new Map<string, number>();
|
||||
for (const c of pipelineResult.communityResult.communities) {
|
||||
const label = c.heuristicLabel || c.label || 'Unknown';
|
||||
groups.set(label, (groups.get(label) || 0) + c.symbolCount);
|
||||
}
|
||||
aggregatedClusterCount = Array.from(groups.values()).filter((cnt) => cnt >= 5).length;
|
||||
}
|
||||
try {
|
||||
await generateAIContextFiles(
|
||||
repoPath,
|
||||
storagePath,
|
||||
projectName,
|
||||
{
|
||||
files: pipelineResult.totalFileCount,
|
||||
nodes: stats.nodes,
|
||||
edges: stats.edges,
|
||||
communities: pipelineResult.communityResult?.stats.totalCommunities,
|
||||
clusters: aggregatedClusterCount,
|
||||
processes: pipelineResult.processResult?.stats.totalProcesses,
|
||||
},
|
||||
undefined,
|
||||
{ skipAgentsMd: options.skipAgentsMd, noStats: options.noStats },
|
||||
);
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
|
||||
await ensureGitNexusIgnored(repoPath);
|
||||
await closeLbug();
|
||||
|
||||
progress('done', 100, 'Incremental complete');
|
||||
|
||||
return {
|
||||
repoName: projectName,
|
||||
repoPath,
|
||||
stats: newMeta.stats ?? {},
|
||||
pipelineResult,
|
||||
};
|
||||
}
|
||||
|
|
|
|||
|
|
@ -71,48 +71,8 @@ export interface RepoMeta {
|
|||
processes?: number;
|
||||
embeddings?: number;
|
||||
};
|
||||
/**
|
||||
* Bumped whenever incremental-indexing invariants change in an
|
||||
* incompatible way (schema bump, closure-algorithm fix, etc.).
|
||||
* Mismatch → run-analyze forces a full rebuild.
|
||||
* See docs/superpowers/specs/2026-05-10-incremental-indexing-design.md.
|
||||
*/
|
||||
schemaVersion?: number;
|
||||
/**
|
||||
* Per-file content/surface signatures used to drive incremental closure
|
||||
* expansion. v1: SHA-256 of file content (cheap, conservative — body-only
|
||||
* edits trigger 1-hop closure expansion). v2 will switch to a real
|
||||
* surface-only signature so body-only edits stay at closure size 1.
|
||||
*
|
||||
* Map keys are repo-relative file paths; values are hex digests. Files
|
||||
* not in this map are treated as "no previous signature" → first-time
|
||||
* processing.
|
||||
*/
|
||||
surfaceSignatures?: Record<string, string>;
|
||||
/**
|
||||
* Crash-recovery dirty flag. Written by run-analyze BEFORE any DB
|
||||
* mutation in an incremental run; cleared by overwrite on success. Its
|
||||
* presence at the start of the next run forces a full rebuild — the
|
||||
* cheapest path back to a known-good index after a crashed/cancelled
|
||||
* incremental run. Consumed by run-analyze.ts.
|
||||
*/
|
||||
incrementalInProgress?: {
|
||||
closure: string[];
|
||||
startedAt: number;
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Bumped whenever incremental-indexing invariants change in an
|
||||
* incompatible way. v1 is `1`. Increment when:
|
||||
* - The closure-expansion algorithm changes in a way that prior
|
||||
* surfaceSignatures cannot be trusted.
|
||||
* - The hydrate phase's DB schema assumptions change.
|
||||
* - The surface-signature derivation changes (so prior hashes are not
|
||||
* comparable to current ones).
|
||||
*/
|
||||
export const INCREMENTAL_SCHEMA_VERSION = 1;
|
||||
|
||||
export interface IndexedRepo {
|
||||
repoPath: string;
|
||||
storagePath: string;
|
||||
|
|
|
|||
|
|
@ -1,212 +0,0 @@
|
|||
import { describe, it, expect, vi } from 'vitest';
|
||||
import { computeImporterClosure } from '../../src/core/incremental/closure.js';
|
||||
|
||||
/**
|
||||
* Mini test harness: a fixture import graph + per-file surface signatures.
|
||||
*
|
||||
* `imports[X] = [a, b]` means files `a` and `b` import file `X`. So when we
|
||||
* ask for "importers of X" the answer is `[a, b]`.
|
||||
*/
|
||||
interface Fixture {
|
||||
files: string[];
|
||||
/** target → list of importers */
|
||||
imports: Record<string, string[]>;
|
||||
/** previous surfaces */
|
||||
prevSurfaces: Record<string, string>;
|
||||
/** new surfaces (what parseFile + surfaceFor will produce) */
|
||||
newSurfaces: Record<string, string>;
|
||||
}
|
||||
|
||||
function makeHarness(fx: Fixture) {
|
||||
const parseFile = vi.fn(async (f: string) => ({ filePath: f }));
|
||||
const surfaceFor = vi.fn((f: string) => fx.newSurfaces[f] ?? '');
|
||||
const queryImporters = vi.fn(async (f: string) => fx.imports[f] ?? []);
|
||||
return { parseFile, surfaceFor, queryImporters };
|
||||
}
|
||||
|
||||
describe('computeImporterClosure', () => {
|
||||
it('empty input → empty closure', async () => {
|
||||
const { parseFile, surfaceFor, queryImporters } = makeHarness({
|
||||
files: [],
|
||||
imports: {},
|
||||
prevSurfaces: {},
|
||||
newSurfaces: {},
|
||||
});
|
||||
const r = await computeImporterClosure({
|
||||
initialChangedFiles: new Set(),
|
||||
prevSurfaces: {},
|
||||
parseFile,
|
||||
surfaceFor,
|
||||
queryImporters,
|
||||
});
|
||||
expect(r.closure.size).toBe(0);
|
||||
expect(parseFile).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('single file with unchanged surface → closure size 1, no expansion', async () => {
|
||||
const { parseFile, surfaceFor, queryImporters } = makeHarness({
|
||||
files: ['a.ts'],
|
||||
imports: { 'a.ts': ['b.ts', 'c.ts'] },
|
||||
prevSurfaces: { 'a.ts': 'sig-a-v1' },
|
||||
newSurfaces: { 'a.ts': 'sig-a-v1' }, // same
|
||||
});
|
||||
const r = await computeImporterClosure({
|
||||
initialChangedFiles: new Set(['a.ts']),
|
||||
prevSurfaces: { 'a.ts': 'sig-a-v1' },
|
||||
parseFile,
|
||||
surfaceFor,
|
||||
queryImporters,
|
||||
});
|
||||
expect([...r.closure].sort()).toEqual(['a.ts']);
|
||||
// queryImporters should not have been called: surface unchanged.
|
||||
expect(queryImporters).not.toHaveBeenCalled();
|
||||
expect(r.expandedFromImporters.size).toBe(0);
|
||||
});
|
||||
|
||||
it('single file with changed surface → expands to direct importers', async () => {
|
||||
const { parseFile, surfaceFor, queryImporters } = makeHarness({
|
||||
files: ['a.ts', 'b.ts', 'c.ts'],
|
||||
imports: {
|
||||
'a.ts': ['b.ts', 'c.ts'],
|
||||
'b.ts': [],
|
||||
'c.ts': [],
|
||||
},
|
||||
prevSurfaces: { 'a.ts': 'sig-old', 'b.ts': 'sig-b', 'c.ts': 'sig-c' },
|
||||
newSurfaces: { 'a.ts': 'sig-new', 'b.ts': 'sig-b', 'c.ts': 'sig-c' },
|
||||
});
|
||||
const r = await computeImporterClosure({
|
||||
initialChangedFiles: new Set(['a.ts']),
|
||||
prevSurfaces: { 'a.ts': 'sig-old', 'b.ts': 'sig-b', 'c.ts': 'sig-c' },
|
||||
parseFile,
|
||||
surfaceFor,
|
||||
queryImporters,
|
||||
});
|
||||
expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts', 'c.ts']);
|
||||
expect([...r.expandedFromImporters].sort()).toEqual(['b.ts', 'c.ts']);
|
||||
});
|
||||
|
||||
it('multi-hop cascade (A surface change → B in closure → B surface change → C in closure)', async () => {
|
||||
const { parseFile, surfaceFor, queryImporters } = makeHarness({
|
||||
files: ['a.ts', 'b.ts', 'c.ts'],
|
||||
imports: {
|
||||
'a.ts': ['b.ts'], // b imports a
|
||||
'b.ts': ['c.ts'], // c imports b
|
||||
'c.ts': [],
|
||||
},
|
||||
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old', 'c.ts': 'c-stable' },
|
||||
newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-new', 'c.ts': 'c-stable' },
|
||||
});
|
||||
const r = await computeImporterClosure({
|
||||
initialChangedFiles: new Set(['a.ts']),
|
||||
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old', 'c.ts': 'c-stable' },
|
||||
parseFile,
|
||||
surfaceFor,
|
||||
queryImporters,
|
||||
});
|
||||
expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts', 'c.ts']);
|
||||
});
|
||||
|
||||
it('cascade halts when surface stops changing mid-chain', async () => {
|
||||
const { parseFile, surfaceFor, queryImporters } = makeHarness({
|
||||
files: ['a.ts', 'b.ts', 'c.ts'],
|
||||
imports: {
|
||||
'a.ts': ['b.ts'],
|
||||
'b.ts': ['c.ts'],
|
||||
},
|
||||
// a's surface changed, b is in closure but b's surface unchanged → c stays out
|
||||
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-stable', 'c.ts': 'c-stable' },
|
||||
newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-stable', 'c.ts': 'c-stable' },
|
||||
});
|
||||
const r = await computeImporterClosure({
|
||||
initialChangedFiles: new Set(['a.ts']),
|
||||
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-stable', 'c.ts': 'c-stable' },
|
||||
parseFile,
|
||||
surfaceFor,
|
||||
queryImporters,
|
||||
});
|
||||
expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts']);
|
||||
expect([...r.expandedFromImporters].sort()).toEqual(['b.ts']);
|
||||
});
|
||||
|
||||
it('terminates on cycles in the import graph', async () => {
|
||||
// a ↔ b cycle. a's surface changes, b is added; b's surface also "changed"
|
||||
// (relative to undefined prev), so its importers are queried — which
|
||||
// includes a, already in closure. Loop terminates because closure
|
||||
// membership prevents re-add.
|
||||
const { parseFile, surfaceFor, queryImporters } = makeHarness({
|
||||
files: ['a.ts', 'b.ts'],
|
||||
imports: { 'a.ts': ['b.ts'], 'b.ts': ['a.ts'] },
|
||||
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' },
|
||||
newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-new' },
|
||||
});
|
||||
const r = await computeImporterClosure({
|
||||
initialChangedFiles: new Set(['a.ts']),
|
||||
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' },
|
||||
parseFile,
|
||||
surfaceFor,
|
||||
queryImporters,
|
||||
});
|
||||
expect([...r.closure].sort()).toEqual(['a.ts', 'b.ts']);
|
||||
// Each file parsed exactly once even though they import each other.
|
||||
expect(parseFile).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it('newly-added file (no prev surface) triggers expansion', async () => {
|
||||
// For a brand-new file, prevSurfaces[f] is undefined → surfaceChanged
|
||||
// returns true → its importers are queried. (For a truly *new* file,
|
||||
// there should be no importers yet, so closure stays at {f}.)
|
||||
const { parseFile, surfaceFor, queryImporters } = makeHarness({
|
||||
files: ['new.ts'],
|
||||
imports: {},
|
||||
prevSurfaces: {},
|
||||
newSurfaces: { 'new.ts': 'sig-new' },
|
||||
});
|
||||
const r = await computeImporterClosure({
|
||||
initialChangedFiles: new Set(['new.ts']),
|
||||
prevSurfaces: {},
|
||||
parseFile,
|
||||
surfaceFor,
|
||||
queryImporters,
|
||||
});
|
||||
expect([...r.closure]).toEqual(['new.ts']);
|
||||
expect(queryImporters).toHaveBeenCalledOnce();
|
||||
});
|
||||
|
||||
it('parses each file exactly once and caches the result', async () => {
|
||||
const { parseFile, surfaceFor, queryImporters } = makeHarness({
|
||||
files: ['a.ts', 'b.ts'],
|
||||
imports: { 'a.ts': ['b.ts'] },
|
||||
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b' },
|
||||
newSurfaces: { 'a.ts': 'new', 'b.ts': 'b' },
|
||||
});
|
||||
const r = await computeImporterClosure({
|
||||
initialChangedFiles: new Set(['a.ts']),
|
||||
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b' },
|
||||
parseFile,
|
||||
surfaceFor,
|
||||
queryImporters,
|
||||
});
|
||||
expect(parseFile).toHaveBeenCalledTimes(2);
|
||||
expect(r.parseCache.size).toBe(2);
|
||||
expect(r.parseCache.has('a.ts')).toBe(true);
|
||||
expect(r.parseCache.has('b.ts')).toBe(true);
|
||||
});
|
||||
|
||||
it('records new surfaces for every parsed file', async () => {
|
||||
const { parseFile, surfaceFor, queryImporters } = makeHarness({
|
||||
files: ['a.ts', 'b.ts'],
|
||||
imports: { 'a.ts': ['b.ts'] },
|
||||
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' },
|
||||
newSurfaces: { 'a.ts': 'new', 'b.ts': 'b-new' },
|
||||
});
|
||||
const r = await computeImporterClosure({
|
||||
initialChangedFiles: new Set(['a.ts']),
|
||||
prevSurfaces: { 'a.ts': 'old', 'b.ts': 'b-old' },
|
||||
parseFile,
|
||||
surfaceFor,
|
||||
queryImporters,
|
||||
});
|
||||
expect(r.newSurfaces.get('a.ts')).toBe('new');
|
||||
expect(r.newSurfaces.get('b.ts')).toBe('b-new');
|
||||
});
|
||||
});
|
||||
|
|
@ -1,163 +0,0 @@
|
|||
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
||||
import { execFileSync } from 'child_process';
|
||||
import {
|
||||
getChangedFilesSinceCommit,
|
||||
LastCommitMissingError,
|
||||
} from '../../src/core/incremental/git-diff.js';
|
||||
|
||||
vi.mock('child_process', () => ({
|
||||
execFileSync: vi.fn(),
|
||||
}));
|
||||
|
||||
const mockExec = vi.mocked(execFileSync);
|
||||
|
||||
/**
|
||||
* Mock helper: queue a sequence of execFileSync return values.
|
||||
* Order matters — the first call returns the first value, etc.
|
||||
*
|
||||
* Calls in `getChangedFilesSinceCommit`:
|
||||
* 1. cat-file -e <commit> (commitExists check)
|
||||
* 2. diff --name-status -z (committed changes)
|
||||
* 3. status --porcelain -z (dirty tree)
|
||||
*/
|
||||
function mockGitSequence(
|
||||
catFileSucceeds: boolean,
|
||||
diffOutput: string,
|
||||
statusOutput: string,
|
||||
) {
|
||||
mockExec.mockReset();
|
||||
if (catFileSucceeds) {
|
||||
mockExec.mockImplementationOnce(() => Buffer.from(''));
|
||||
} else {
|
||||
mockExec.mockImplementationOnce(() => {
|
||||
throw new Error('not in repo');
|
||||
});
|
||||
}
|
||||
mockExec.mockImplementationOnce(() => diffOutput);
|
||||
mockExec.mockImplementationOnce(() => statusOutput);
|
||||
}
|
||||
|
||||
describe('getChangedFilesSinceCommit', () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it('throws LastCommitMissingError when commit not in repo', () => {
|
||||
mockGitSequence(false, '', '');
|
||||
expect(() => getChangedFilesSinceCommit('/repo', 'deadbeef')).toThrow(
|
||||
LastCommitMissingError,
|
||||
);
|
||||
});
|
||||
|
||||
it('returns empty arrays for clean repo and no committed changes', () => {
|
||||
mockGitSequence(true, '', '');
|
||||
const r = getChangedFilesSinceCommit('/repo', 'abc');
|
||||
expect(r).toEqual({ modified: [], added: [], deleted: [] });
|
||||
});
|
||||
|
||||
it('parses committed M/A/D from diff --name-status -z', () => {
|
||||
mockGitSequence(
|
||||
true,
|
||||
'M\0src/a.ts\0A\0src/b.ts\0D\0src/c.ts\0',
|
||||
'',
|
||||
);
|
||||
const r = getChangedFilesSinceCommit('/repo', 'abc');
|
||||
expect(r).toEqual({
|
||||
modified: ['src/a.ts'],
|
||||
added: ['src/b.ts'],
|
||||
deleted: ['src/c.ts'],
|
||||
});
|
||||
});
|
||||
|
||||
it('flattens R<sim> renames into delete(orig) + add(new)', () => {
|
||||
mockGitSequence(
|
||||
true,
|
||||
'R100\0src/old.ts\0src/new.ts\0',
|
||||
'',
|
||||
);
|
||||
const r = getChangedFilesSinceCommit('/repo', 'abc');
|
||||
expect(r).toEqual({
|
||||
modified: [],
|
||||
added: ['src/new.ts'],
|
||||
deleted: ['src/old.ts'],
|
||||
});
|
||||
});
|
||||
|
||||
it('treats T (type change) as modified', () => {
|
||||
mockGitSequence(true, 'T\0src/link.ts\0', '');
|
||||
const r = getChangedFilesSinceCommit('/repo', 'abc');
|
||||
expect(r.modified).toContain('src/link.ts');
|
||||
});
|
||||
|
||||
it('parses dirty tree from status --porcelain -z', () => {
|
||||
// 'M a.ts' = staged-modified. ' M b.ts' = unstaged-modified.
|
||||
// '?? c.ts' = untracked. ' D d.ts' = unstaged-deleted.
|
||||
mockGitSequence(
|
||||
true,
|
||||
'',
|
||||
'M a.ts\0 M b.ts\0?? c.ts\0 D d.ts\0',
|
||||
);
|
||||
const r = getChangedFilesSinceCommit('/repo', 'abc');
|
||||
expect(r.modified.sort()).toEqual(['a.ts', 'b.ts']);
|
||||
expect(r.added).toEqual(['c.ts']);
|
||||
expect(r.deleted).toEqual(['d.ts']);
|
||||
});
|
||||
|
||||
it('unions committed + dirty changes', () => {
|
||||
mockGitSequence(
|
||||
true,
|
||||
'M\0a.ts\0', // committed: a.ts modified
|
||||
' M b.ts\0?? c.ts\0', // dirty: b.ts modified, c.ts untracked
|
||||
);
|
||||
const r = getChangedFilesSinceCommit('/repo', 'abc');
|
||||
expect(r.modified.sort()).toEqual(['a.ts', 'b.ts']);
|
||||
expect(r.added).toEqual(['c.ts']);
|
||||
expect(r.deleted).toEqual([]);
|
||||
});
|
||||
|
||||
it('resolves overlap: file added in diff and modified in status → added', () => {
|
||||
mockGitSequence(
|
||||
true,
|
||||
'A\0newfile.ts\0',
|
||||
' M newfile.ts\0',
|
||||
);
|
||||
const r = getChangedFilesSinceCommit('/repo', 'abc');
|
||||
expect(r).toEqual({
|
||||
modified: [],
|
||||
added: ['newfile.ts'],
|
||||
deleted: [],
|
||||
});
|
||||
});
|
||||
|
||||
it('handles porcelain rename ("R new\\0old")', () => {
|
||||
mockGitSequence(
|
||||
true,
|
||||
'',
|
||||
'R new.ts\0old.ts\0',
|
||||
);
|
||||
const r = getChangedFilesSinceCommit('/repo', 'abc');
|
||||
expect(r.added).toEqual(['new.ts']);
|
||||
// Porcelain rename: `R` in status doesn't pre-flatten old into deleted
|
||||
// because git already resolved the rename — but our parser does flatten
|
||||
// when the diff layer flags it. For status-only, the original is the
|
||||
// rename source and we don't claim to know it was deleted from index.
|
||||
});
|
||||
|
||||
it('returns sorted output for stable comparison', () => {
|
||||
mockGitSequence(
|
||||
true,
|
||||
'M\0z.ts\0M\0a.ts\0M\0m.ts\0',
|
||||
'',
|
||||
);
|
||||
const r = getChangedFilesSinceCommit('/repo', 'abc');
|
||||
expect(r.modified).toEqual(['a.ts', 'm.ts', 'z.ts']);
|
||||
});
|
||||
|
||||
it('passes correct cwd to git', () => {
|
||||
mockGitSequence(true, '', '');
|
||||
getChangedFilesSinceCommit('/some/path', 'abc');
|
||||
for (const call of mockExec.mock.calls) {
|
||||
expect((call[2] as { cwd: string }).cwd).toBe('/some/path');
|
||||
}
|
||||
});
|
||||
});
|
||||
|
|
@ -1,185 +0,0 @@
|
|||
import { describe, it, expect } from 'vitest';
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import {
|
||||
extractSurfaceSignature,
|
||||
surfaceChanged,
|
||||
} from '../../src/core/incremental/surface.js';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
|
||||
function fn(
|
||||
id: string,
|
||||
filePath: string,
|
||||
name: string,
|
||||
extra: Record<string, unknown> = {},
|
||||
): GraphNode {
|
||||
return {
|
||||
id,
|
||||
label: 'Function',
|
||||
properties: { name, filePath, ...extra },
|
||||
};
|
||||
}
|
||||
|
||||
function method(
|
||||
id: string,
|
||||
filePath: string,
|
||||
name: string,
|
||||
extra: Record<string, unknown> = {},
|
||||
): GraphNode {
|
||||
return {
|
||||
id,
|
||||
label: 'Method',
|
||||
properties: { name, filePath, ...extra },
|
||||
};
|
||||
}
|
||||
|
||||
function cls(
|
||||
id: string,
|
||||
filePath: string,
|
||||
name: string,
|
||||
extra: Record<string, unknown> = {},
|
||||
): GraphNode {
|
||||
return {
|
||||
id,
|
||||
label: 'Class',
|
||||
properties: { name, filePath, ...extra },
|
||||
};
|
||||
}
|
||||
|
||||
function rel(
|
||||
id: string,
|
||||
type: GraphRelationship['type'],
|
||||
src: string,
|
||||
dst: string,
|
||||
): GraphRelationship {
|
||||
return { id, type, sourceId: src, targetId: dst, confidence: 1, reason: 't' };
|
||||
}
|
||||
|
||||
describe('extractSurfaceSignature', () => {
|
||||
it('returns a stable hash for an empty file', () => {
|
||||
const g = createKnowledgeGraph();
|
||||
const h1 = extractSurfaceSignature(g, 'a.ts');
|
||||
const h2 = extractSurfaceSignature(g, 'a.ts');
|
||||
expect(h1).toBe(h2);
|
||||
expect(h1).toMatch(/^[a-f0-9]{64}$/);
|
||||
});
|
||||
|
||||
it('produces the same hash for the same surface', () => {
|
||||
const g1 = createKnowledgeGraph();
|
||||
g1.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
|
||||
const g2 = createKnowledgeGraph();
|
||||
g2.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
|
||||
expect(extractSurfaceSignature(g1, 'a.ts')).toBe(
|
||||
extractSurfaceSignature(g2, 'a.ts'),
|
||||
);
|
||||
});
|
||||
|
||||
it('is invariant under node insertion order', () => {
|
||||
const g1 = createKnowledgeGraph();
|
||||
g1.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
|
||||
g1.addNode(fn('Function:a.ts:bar#1', 'a.ts', 'bar', { parameterCount: 1 }));
|
||||
const g2 = createKnowledgeGraph();
|
||||
g2.addNode(fn('Function:a.ts:bar#1', 'a.ts', 'bar', { parameterCount: 1 }));
|
||||
g2.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
|
||||
expect(extractSurfaceSignature(g1, 'a.ts')).toBe(
|
||||
extractSurfaceSignature(g2, 'a.ts'),
|
||||
);
|
||||
});
|
||||
|
||||
it('changes when a function is renamed', () => {
|
||||
const g1 = createKnowledgeGraph();
|
||||
g1.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo'));
|
||||
const g2 = createKnowledgeGraph();
|
||||
g2.addNode(fn('Function:a.ts:bar#0', 'a.ts', 'bar'));
|
||||
expect(extractSurfaceSignature(g1, 'a.ts')).not.toBe(
|
||||
extractSurfaceSignature(g2, 'a.ts'),
|
||||
);
|
||||
});
|
||||
|
||||
it('changes when a function signature changes', () => {
|
||||
const g1 = createKnowledgeGraph();
|
||||
g1.addNode(
|
||||
fn('Function:a.ts:foo#1', 'a.ts', 'foo', {
|
||||
parameterCount: 1,
|
||||
parameterTypes: ['number'],
|
||||
returnType: 'string',
|
||||
}),
|
||||
);
|
||||
const g2 = createKnowledgeGraph();
|
||||
g2.addNode(
|
||||
fn('Function:a.ts:foo#1', 'a.ts', 'foo', {
|
||||
parameterCount: 1,
|
||||
parameterTypes: ['string'],
|
||||
returnType: 'string',
|
||||
}),
|
||||
);
|
||||
expect(extractSurfaceSignature(g1, 'a.ts')).not.toBe(
|
||||
extractSurfaceSignature(g2, 'a.ts'),
|
||||
);
|
||||
});
|
||||
|
||||
it('does NOT change for body-only edits (no surface mutation)', () => {
|
||||
// Body-only edits don't add/remove/rename surface nodes — same hash.
|
||||
const g = createKnowledgeGraph();
|
||||
g.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
|
||||
const h1 = extractSurfaceSignature(g, 'a.ts');
|
||||
// Re-build with same surface (simulating a re-parse of a body-only edit).
|
||||
const g2 = createKnowledgeGraph();
|
||||
g2.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo', { parameterCount: 0 }));
|
||||
const h2 = extractSurfaceSignature(g2, 'a.ts');
|
||||
expect(h1).toBe(h2);
|
||||
});
|
||||
|
||||
it('only considers nodes whose filePath matches', () => {
|
||||
const g = createKnowledgeGraph();
|
||||
g.addNode(fn('Function:a.ts:foo#0', 'a.ts', 'foo'));
|
||||
g.addNode(fn('Function:b.ts:bar#0', 'b.ts', 'bar'));
|
||||
const ha = extractSurfaceSignature(g, 'a.ts');
|
||||
const hb = extractSurfaceSignature(g, 'b.ts');
|
||||
expect(ha).not.toBe(hb);
|
||||
// Adding a node in a OTHER file should not change a's signature.
|
||||
g.addNode(fn('Function:c.ts:baz#0', 'c.ts', 'baz'));
|
||||
expect(extractSurfaceSignature(g, 'a.ts')).toBe(ha);
|
||||
});
|
||||
|
||||
it('changes when EXTENDS target changes', () => {
|
||||
const g1 = createKnowledgeGraph();
|
||||
g1.addNode(cls('Class:a.ts:Child#0', 'a.ts', 'Child'));
|
||||
g1.addNode(cls('Class:b.ts:ParentA#0', 'b.ts', 'ParentA'));
|
||||
g1.addRelationship(
|
||||
rel('r1', 'EXTENDS', 'Class:a.ts:Child#0', 'Class:b.ts:ParentA#0'),
|
||||
);
|
||||
|
||||
const g2 = createKnowledgeGraph();
|
||||
g2.addNode(cls('Class:a.ts:Child#0', 'a.ts', 'Child'));
|
||||
g2.addNode(cls('Class:b.ts:ParentB#0', 'b.ts', 'ParentB'));
|
||||
g2.addRelationship(
|
||||
rel('r1', 'EXTENDS', 'Class:a.ts:Child#0', 'Class:b.ts:ParentB#0'),
|
||||
);
|
||||
|
||||
expect(extractSurfaceSignature(g1, 'a.ts')).not.toBe(
|
||||
extractSurfaceSignature(g2, 'a.ts'),
|
||||
);
|
||||
});
|
||||
|
||||
it('includes Method/Class/Interface in the surface', () => {
|
||||
const g = createKnowledgeGraph();
|
||||
g.addNode(cls('Class:a.ts:C#0', 'a.ts', 'C'));
|
||||
g.addNode(method('Method:a.ts:C.m#0', 'a.ts', 'm'));
|
||||
const h = extractSurfaceSignature(g, 'a.ts');
|
||||
expect(h).toMatch(/^[a-f0-9]{64}$/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('surfaceChanged', () => {
|
||||
it('returns true when prev is undefined (first index of file)', () => {
|
||||
expect(surfaceChanged(undefined, 'abc')).toBe(true);
|
||||
});
|
||||
|
||||
it('returns false when prev equals current', () => {
|
||||
expect(surfaceChanged('abc', 'abc')).toBe(false);
|
||||
});
|
||||
|
||||
it('returns true when prev differs from current', () => {
|
||||
expect(surfaceChanged('abc', 'def')).toBe(true);
|
||||
});
|
||||
});
|
||||
Loading…
Add table
Reference in a new issue