diff --git a/.github/workflows/claude.yml b/.github/workflows/claude.yml index fcafc0279..021a56930 100644 --- a/.github/workflows/claude.yml +++ b/.github/workflows/claude.yml @@ -158,6 +158,8 @@ jobs: github_token: ${{ secrets.GITHUB_TOKEN }} allowed_non_write_users: '*' show_full_output: true + # Review posts use Bash (`gh`, etc.); default mode asks for approval — impossible in CI. + claude_args: '--dangerously-skip-permissions' plugin_marketplaces: 'https://github.com/anthropics/claude-code.git' plugins: 'code-review@claude-code-plugins' - prompt: '/code-review:code-review ${{ github.repository }}/pull/${{ steps.pr.outputs.number }}' + prompt: '/code-review:code-review https://github.com/${{ github.repository }}/pull/${{ steps.pr.outputs.number }} --comment' diff --git a/AGENTS.md b/AGENTS.md index 60317d73d..1346facc9 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -149,11 +149,16 @@ Indexed as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows) ## Keeping the Index Fresh ```bash -npx gitnexus analyze # basic refresh; preserves any existing embeddings +npx gitnexus analyze # incremental by default; preserves embeddings +npx gitnexus analyze --force # full rebuild from scratch (opt out of incremental) npx gitnexus analyze --embeddings # also generate embeddings for new/changed nodes npx gitnexus analyze --drop-embeddings # explicit opt-in to wipe existing embeddings ``` +`analyze` runs **incrementally by default**. The pipeline still parses every file every run (cross-file resolution requires it), but tree-sitter parsing is **served from a content-addressed cache** at `.gitnexus/parse-cache.json` for chunks whose file contents haven't changed since the last run. Only changed-file rows (and their importers) are rewritten in LadybugDB; unchanged-file rows are preserved. Output is byte-equivalent to a full rebuild. Pass `--force` to wipe and re-index from scratch (e.g., to recover from a corrupt index, or after upgrading GitNexus). + +The parse cache key is **content-addressed and version-tagged**: it survives `--force` runs, and is automatically invalidated by a `gitnexus` package upgrade (so a new tree-sitter grammar doesn't silently replay stale parse output). Safe to delete `.gitnexus/parse-cache.json` at any time — it'll be rebuilt on the next analyze. + Check `.gitnexus/meta.json` `stats.embeddings` (0 = none). A plain `analyze` no longer drops existing vectors — pass `--drop-embeddings` to wipe. > Claude Code: PostToolUse hook detects a stale index after `git commit` and `git merge` and prompts the agent to run `analyze`. The hook does not invoke `analyze` itself. diff --git a/GUARDRAILS.md b/GUARDRAILS.md index 1cc032759..c09f0319a 100644 --- a/GUARDRAILS.md +++ b/GUARDRAILS.md @@ -30,9 +30,15 @@ Format: **Trigger → Instruction → Reason**. Append new Signs when the same m ### Stale graph after edits - **Trigger:** MCP warns index is behind `HEAD`, or search doesn't match latest commit. -- **Do:** `npx gitnexus analyze` (plus `--embeddings` if used). +- **Do:** `npx gitnexus analyze` (plus `--embeddings` if used). Runs incrementally by default — the pipeline parses every file every run (cross-file resolution requires it), but tree-sitter dispatch is skipped for unchanged file chunks via the content-addressed cache, and only changed-file rows (plus their importers, transitively) are rewritten in LadybugDB. - **Why:** Tools query LadybugDB from last analyze; git changes are invisible until re-indexed. +### Index seems corrupt or "incremental" is misbehaving + +- **Trigger:** `analyze` produces unexpected results, or `meta.json.incrementalInProgress` is set, or the index is in a half-state after a crash. +- **Do:** `npx gitnexus analyze --force` to rebuild from scratch. The dirty-flag check forces this automatically when a previous incremental run didn't complete cleanly, but `--force` is the manual escape hatch. Safe to delete `.gitnexus/parse-cache.json` at any time — content-addressed, will be regenerated. +- **Why:** Incremental writeback is selective DB row replacement; if the on-disk state is inconsistent for any reason, a full rebuild is the cheapest path back to a known-good index. + ### Embeddings vanished after analyze - **Trigger:** Semantic search quality drops; `stats.embeddings` in `meta.json` is 0 after refresh. diff --git a/gitnexus/src/core/incremental/shadow-candidates.ts b/gitnexus/src/core/incremental/shadow-candidates.ts new file mode 100644 index 000000000..415a6d9df --- /dev/null +++ b/gitnexus/src/core/incremental/shadow-candidates.ts @@ -0,0 +1,76 @@ +/** + * Shadow-candidate path derivation for incremental indexing. + * + * Background — Bugbot review on PR #1479: + * queryImporters() on a NEWLY ADDED file returns 0 importers in the + * pre-pipeline DB, because the new file's IMPORTS rows haven't been + * written yet. But pre-existing files may have IMPORTS edges that + * *resolved to a sibling path*, and the newcomer can now steal that + * resolution under standard JS/TS module-resolution rules. Without + * pulling those pre-existing files into the writable set, their + * stale CALLS edges remain pointing at the OLD resolution target. + * + * Given an added file path, this helper enumerates the pre-existing + * file paths whose import-resolution claim the newcomer can steal. + * Caller filters the candidates against the prior-run `fileHashes` + * map so we only query importers of paths that actually existed. + * + * Shadow patterns covered (resolution-priority-aware): + * + * (a) Same basename, different extension — + * added `foo/bar.ts` shadows `foo/bar.{tsx,js,jsx,mjs,cjs,d.ts}`. + * (b) Bare-file beats directory-style index — + * added `foo/bar.ts` shadows `foo/bar/index.{ts,tsx,...}`. + * (c) Directory-index beats bare-file — + * added `foo/index.ts` shadows `foo.{ts,tsx,...}` (rare but real, + * e.g. converting a single-file module into a directory module). + * + * Resolution-order priority is conservatively wide: we enumerate ALL + * common extensions because we don't know which the importer actually + * specified, and over-seeding is harmless (extra BFS work, but the + * subgraph extract still gates write-back by file membership). + * + * Cross-platform path separators: candidates are emitted with both `/` + * and `\` for shadow pattern (b), since the caller's prior fileHashes + * map may use either depending on the OS that wrote it. + */ + +const SHADOW_EXTS = ['.d.ts', '.tsx', '.ts', '.jsx', '.js', '.mjs', '.cjs']; + +/** + * Enumerate pre-existing paths whose import-resolution `added` can steal. + * + * @param added — repo-relative path of a newly-added file + * @returns deduplicated list of candidate paths (NOT filtered against + * any known-files set — caller does that) + */ +export const shadowCandidatesFor = (added: string): string[] => { + const ext = SHADOW_EXTS.find((e) => added.endsWith(e)); + if (!ext) return []; + + const noExt = added.slice(0, -ext.length); + const out = new Set(); + + // (a) Same basename, different extension. + for (const alt of SHADOW_EXTS) { + if (alt !== ext) out.add(noExt + alt); + } + + // (b) Bare file beats sibling directory-style index. + for (const idx of SHADOW_EXTS) { + out.add(`${noExt}/index${idx}`); + out.add(`${noExt}\\index${idx}`); + } + + // (c) New `foo/index.ext` shadows old `foo.ext`. + const idxSuffixSlash = '/index'; + const idxSuffixBack = '\\index'; + let dir: string | null = null; + if (noExt.endsWith(idxSuffixSlash)) dir = noExt.slice(0, -idxSuffixSlash.length); + else if (noExt.endsWith(idxSuffixBack)) dir = noExt.slice(0, -idxSuffixBack.length); + if (dir !== null) { + for (const alt of SHADOW_EXTS) out.add(dir + alt); + } + + return [...out]; +}; diff --git a/gitnexus/src/core/incremental/subgraph-extract.ts b/gitnexus/src/core/incremental/subgraph-extract.ts new file mode 100644 index 000000000..71fe656be --- /dev/null +++ b/gitnexus/src/core/incremental/subgraph-extract.ts @@ -0,0 +1,123 @@ +/** + * Subgraph extraction for incremental DB writeback. + * + * Given the FULL ctx.graph produced by the pipeline (all files parsed, + * all phases run) and the set of file paths whose DB rows must be + * replaced, produce a smaller KnowledgeGraph that contains: + * + * - Every node whose `properties.filePath` is in `toWriteSet`. + * - Every graph-wide node (Community, Process) — these are regenerated + * each run by the communities/processes phases and must be fully + * rewritten. + * - Every relationship where AT LEAST ONE endpoint is in the writable + * set above. Relationships entirely between unchanged-file nodes + * are skipped — their rows are still in the DB and re-inserting + * them would PK-conflict at COPY time. + * + * The resulting subgraph is what gets passed to `loadGraphToLbug` after + * the orchestrator has deleted the corresponding DB rows. Hydrated + * unchanged-file rows are never touched in the DB. + * + * # Cross-file edge consistency (Finding 1) + * + * `extractChangedSubgraph` intentionally does NOT expand the set it is + * given — expansion is the orchestrator's job, so the SAME expanded set + * can be fed to both `deleteNodesForFile` and this function (asymmetry + * between the delete set and the write set silently corrupts the DB). + * `computeEffectiveWriteSet` below performs the boundary-crossing 1-hop + * walk; the orchestrator composes it with its importer-BFS expansion and + * passes the result here. + * + * Why the 1-hop walk is needed: consider a barrel re-export change — + * file C (a barrel) shifts `export { foo } from './b'` to + * `export { foo } from './d'`. After scope resolution, file A's CALLS + * edge to `foo` resolves to D instead of B, even though A's content is + * byte-for-byte identical: + * + * - Old A→B edge survives in DB (neither A nor B is changed → not deleted) + * - New A→D edge is missing (neither A nor D in writable set → skipped) + * + * Pulling the unchanged-side file of every writable-boundary-crossing + * edge into the write set fixes both halves: the orchestrator's + * `DETACH DELETE` cleans up the stale unchanged-side rows, and the new + * cross-file edges land because at least one endpoint is now writable. + * + * Limitation (documented): if a file X *stopped* importing from a + * changed file C, X has no edge to C in the new graph, so this 1-hop + * walk doesn't catch it. The orchestrator's importer-BFS (which reads + * IMPORTS from the pre-pipeline DB) covers that case instead. + */ + +import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; +import { createKnowledgeGraph } from '../graph/graph.js'; +import type { KnowledgeGraph } from '../graph/types.js'; + +const isGraphWide = (label: string): boolean => label === 'Community' || label === 'Process'; + +/** + * Build a Map for every File-bound node in the graph. + * Graph-wide nodes (Community/Process) have no filePath and are filtered. + */ +const indexNodeFilePaths = (fullGraph: KnowledgeGraph): Map => { + const idx = new Map(); + fullGraph.forEachNode((n: GraphNode) => { + const fp = n.properties?.filePath as string | undefined; + if (fp) idx.set(n.id, fp); + }); + return idx; +}; + +export const extractChangedSubgraph = ( + fullGraph: KnowledgeGraph, + toWriteSet: ReadonlySet, +): KnowledgeGraph => { + const sub = createKnowledgeGraph(); + const writableNodeIds = new Set(); + + fullGraph.forEachNode((n: GraphNode) => { + const filePath = n.properties?.filePath as string | undefined; + const include = (filePath && toWriteSet.has(filePath)) || isGraphWide(n.label); + if (include) { + sub.addNode(n); + writableNodeIds.add(n.id); + } + }); + + fullGraph.forEachRelationship((r: GraphRelationship) => { + if (writableNodeIds.has(r.sourceId) || writableNodeIds.has(r.targetId)) { + sub.addRelationship(r); + } + }); + + return sub; +}; + +/** + * Public — derive the EFFECTIVE write-set: `toWriteSet` expanded by one + * hop along every edge in the new graph that crosses the writable + * boundary (one endpoint in a writable file, the other in an unchanged + * file). The unchanged-side file is pulled in so its stale rows are + * deleted + rewritten in lockstep with the changed side. + * + * Single pass over the edge list. Does NOT mutate `toWriteSet`. The + * orchestrator MUST feed the returned set to both `deleteNodesForFile` + * and `extractChangedSubgraph` — feeding the unexpanded set to either + * one leaves stale rows or PK-conflicts at COPY time. + */ +export const computeEffectiveWriteSet = ( + fullGraph: KnowledgeGraph, + toWriteSet: ReadonlySet, +): Set => { + const nodeFilePaths = indexNodeFilePaths(fullGraph); + const expanded = new Set(toWriteSet); + fullGraph.forEachRelationship((r: GraphRelationship) => { + const sourcePath = nodeFilePaths.get(r.sourceId); + const targetPath = nodeFilePaths.get(r.targetId); + if (!sourcePath || !targetPath) return; // skip edges to graph-wide nodes + const sourceWritable = toWriteSet.has(sourcePath); + const targetWritable = toWriteSet.has(targetPath); + if (sourceWritable && !targetWritable) expanded.add(targetPath); + else if (targetWritable && !sourceWritable) expanded.add(sourcePath); + }); + return expanded; +}; diff --git a/gitnexus/src/core/ingestion/call-processor.ts b/gitnexus/src/core/ingestion/call-processor.ts index 9fa6c1ae5..b45478f7a 100644 --- a/gitnexus/src/core/ingestion/call-processor.ts +++ b/gitnexus/src/core/ingestion/call-processor.ts @@ -42,12 +42,14 @@ import { isVerboseIngestionEnabled } from './utils/verbose.js'; import { yieldToEventLoop } from './utils/event-loop.js'; import { parseSourceSafe } from '../tree-sitter/safe-parse.js'; import { + CLASS_CONTAINER_TYPES, FUNCTION_NODE_TYPES, - findEnclosingClassId, findEnclosingClassInfo, genericFuncName, inferFunctionLabel, } from './utils/ast-helpers.js'; +import type { FieldInfo, FieldExtractorContext } from './field-types.js'; +import type { LanguageProvider } from './language-provider.js'; import { typeTagForId, constTagForId, buildCollisionGroups } from './utils/method-props.js'; import type { MethodInfo } from './method-types.js'; import { @@ -77,6 +79,62 @@ import type { LiteralTypeInferrer } from './type-extractors/types.js'; import type { SyntaxNode } from './utils/ast-helpers.js'; import { logger } from '../logger.js'; + +// ── Property-prepass helpers (parity with parse-worker.ts) ── +// These mirror the sequential-path equivalents in parse-worker.ts so the main- +// thread `processCalls` pre-pass produces byte-identical Property nodes/symbols +// to the worker pool. Drift between the two paths breaks the +// `incremental ≡ --force` invariant the moment a repo crosses the worker +// threshold between runs. + +/** Walk up to the nearest enclosing class/struct/interface AST node. */ +const findEnclosingClassNode = (node: SyntaxNode): SyntaxNode | null => { + let current = node.parent; + while (current) { + if (CLASS_CONTAINER_TYPES.has(current.type)) return current; + current = current.parent; + } + return null; +}; + +/** No-op SymbolTable stub for FieldExtractorContext — matches parse-worker. */ +const NOOP_SYMBOL_TABLE: SymbolTableReader = { + lookupExact: () => undefined, + lookupExactFull: () => undefined, + lookupExactAll: () => [], + lookupCallableByName: () => [], + getFiles: () => [][Symbol.iterator](), + getStats: () => ({ fileCount: 0 }), +}; + +/** + * Extract (and cache) field info for a class node. Cache is passed in so it + * stays scoped to a single `processCalls` invocation rather than leaking + * across analyze runs (worker uses module-level caching because each worker + * process is short-lived; the main thread is not). + * + * Cache key is `${filePath}:${classNode.startIndex}` — startIndex alone is a + * per-file byte offset, so almost every Ruby/Python file's leading class lands + * at byte 0 and would collide across files in the shared map. + */ +const getFieldInfo = ( + classNode: SyntaxNode, + provider: LanguageProvider, + context: FieldExtractorContext, + cache: Map>, +): Map | undefined => { + if (!provider.fieldExtractor) return undefined; + const cacheKey = `${context.filePath}:${classNode.startIndex}`; + const cached = cache.get(cacheKey); + if (cached) return cached; + const result = provider.fieldExtractor.extract(classNode, context); + if (!result?.fields?.length) return undefined; + const map = new Map(); + for (const field of result.fields) map.set(field.name, field); + cache.set(cacheKey, map); + return map; +}; + /** Per-file resolved type bindings for exported symbols. * Populated during call processing, consumed by Phase 14 re-resolution pass. */ export type ExportedTypeMap = Map>; @@ -860,6 +918,120 @@ export const processCalls = async ( prepared.push({ file, language, provider, tree, matches, parentMap, typeEnv }); } + // ── Property-registration pre-pass ── + // Register all routed properties (e.g. Ruby attr_accessor) BEFORE the + // resolution loop so cross-file field-type lookups (e.g. + // `user.address.save → Address#save`) succeed regardless of file + // processing order. This MUST stay in lockstep with the equivalent + // worker-path block in parse-worker.ts (kind === 'properties') — any + // divergence between the two paths breaks the `incremental ≡ --force` + // invariant once a repo crosses the worker threshold between runs. + const fieldInfoCache = new Map>(); + for (const { file, language, provider, matches, typeEnv } of prepared) { + const callRouter = provider.callRouter; + if (!callRouter) continue; + matches.forEach((match) => { + const captureMap: Record = {}; + match.captures.forEach((c) => (captureMap[c.name] = c.node)); + if (!captureMap['call']) return; + const callNameNode = captureMap['call.name']; + if (!callNameNode) return; + const routed = callRouter(callNameNode.text, captureMap['call']); + if (!routed || routed.kind !== 'properties') return; + + const propEnclosingInfo = findEnclosingClassInfo( + captureMap['call'], + file.path, + provider.resolveEnclosingOwner, + ); + const propEnclosingClassId = propEnclosingInfo?.classId ?? null; + + // Enrich routed properties with FieldExtractor metadata so types + // discovered from constructor assignments (e.g. `@address = Address.new`) + // are propagated even when the routing payload itself lacks declaredType. + let routedFieldMap: Map | undefined; + if (provider.fieldExtractor && typeEnv) { + const classNode = findEnclosingClassNode(captureMap['call']); + if (classNode) { + routedFieldMap = getFieldInfo( + classNode, + provider, + { + typeEnv, + symbolTable: NOOP_SYMBOL_TABLE, + filePath: file.path, + language, + }, + fieldInfoCache, + ); + } + } + + const fileId = generateId('File', file.path); + for (const item of routed.items) { + const routedFieldInfo = routedFieldMap?.get(item.propName); + const propQualifiedName = propEnclosingInfo + ? `${propEnclosingInfo.className}.${item.propName}` + : item.propName; + const nodeId = generateId('Property', `${file.path}:${propQualifiedName}`); + graph.addNode({ + id: nodeId, + label: 'Property', + properties: { + name: item.propName, + filePath: file.path, + startLine: item.startLine, + endLine: item.endLine, + language, + isExported: true, + description: item.accessorType, + ...(item.declaredType + ? { declaredType: item.declaredType } + : routedFieldInfo?.type + ? { declaredType: routedFieldInfo.type } + : {}), + ...(routedFieldInfo?.visibility !== undefined + ? { visibility: routedFieldInfo.visibility } + : {}), + ...(routedFieldInfo?.isStatic !== undefined + ? { isStatic: routedFieldInfo.isStatic } + : {}), + ...(routedFieldInfo?.isReadonly !== undefined + ? { isReadonly: routedFieldInfo.isReadonly } + : {}), + }, + }); + ctx.model.symbols.add(file.path, item.propName, nodeId, 'Property', { + ...(propEnclosingClassId ? { ownerId: propEnclosingClassId } : {}), + ...(item.declaredType + ? { declaredType: item.declaredType } + : routedFieldInfo?.type + ? { declaredType: routedFieldInfo.type } + : {}), + }); + const relId = generateId('DEFINES', `${fileId}->${nodeId}`); + graph.addRelationship({ + id: relId, + sourceId: fileId, + targetId: nodeId, + type: 'DEFINES', + confidence: 1.0, + reason: '', + }); + if (propEnclosingClassId) { + graph.addRelationship({ + id: generateId('HAS_PROPERTY', `${propEnclosingClassId}->${nodeId}`), + sourceId: propEnclosingClassId, + targetId: nodeId, + type: 'HAS_PROPERTY', + confidence: 1.0, + reason: '', + }); + } + } + }); + } + // ── Resolution loop: verify constructor bindings and resolve calls ── // The accumulator (if present) is now fully populated from the preparation // loop above, so verifyConstructorBindings sees all provider bindings @@ -930,9 +1102,10 @@ export const processCalls = async ( provider, ); const srcId = enclosing || generateId('File', file.path); - // Defer resolution: Ruby attr_accessor properties are registered during - // this same loop, so cross-file lookups fail if the declaring file hasn't - // been processed yet. Collect now, resolve after all files are done. + // Defer resolution so write-access tracking sees the FINAL graph + // state — properties from the pre-pass are present, but receiver-type + // resolution can still depend on inference that completes during the + // main loop. Resolve after all files have been processed. pendingWrites.push({ receiverTypeName, propertyName, filePath: file.path, srcId }); } // Assignment-only capture (no @call sibling): skip the rest of this @@ -1053,47 +1226,8 @@ export const processCalls = async ( return; case 'properties': { - const fileId = generateId('File', file.path); - const propEnclosingClassId = findEnclosingClassId(captureMap['call'], file.path); - for (const item of routed.items) { - const nodeId = generateId('Property', `${file.path}:${item.propName}`); - graph.addNode({ - id: nodeId, - label: 'Property', - properties: { - name: item.propName, - filePath: file.path, - startLine: item.startLine, - endLine: item.endLine, - language, - isExported: true, - description: item.accessorType, - }, - }); - ctx.model.symbols.add(file.path, item.propName, nodeId, 'Property', { - ...(propEnclosingClassId ? { ownerId: propEnclosingClassId } : {}), - ...(item.declaredType ? { declaredType: item.declaredType } : {}), - }); - const relId = generateId('DEFINES', `${fileId}->${nodeId}`); - graph.addRelationship({ - id: relId, - sourceId: fileId, - targetId: nodeId, - type: 'DEFINES', - confidence: 1.0, - reason: '', - }); - if (propEnclosingClassId) { - graph.addRelationship({ - id: generateId('HAS_PROPERTY', `${propEnclosingClassId}->${nodeId}`), - sourceId: propEnclosingClassId, - targetId: nodeId, - type: 'HAS_PROPERTY', - confidence: 1.0, - reason: '', - }); - } - } + // Properties already registered in the pre-pass above. + // Skip to avoid duplicate nodes/edges. return; } diff --git a/gitnexus/src/core/ingestion/community-processor.ts b/gitnexus/src/core/ingestion/community-processor.ts index 9913e4a3f..ac8f068fa 100644 --- a/gitnexus/src/core/ingestion/community-processor.ts +++ b/gitnexus/src/core/ingestion/community-processor.ts @@ -41,6 +41,24 @@ interface LeidenDetailedResult { modularity: number; } +/** + * Deterministic PRNG (mulberry32) seed for the vendored Leiden algorithm. + * Vendored Leiden defaults `rng: Math.random`, which makes community + * assignment non-deterministic across runs. Passing a seeded RNG gives us + * reproducible community/modularity output, which is required for the + * incremental-indexing equivalence test (incremental ≡ full rebuild). + */ +const LEIDEN_SEED = 0xc0de; +function createSeededRng(seed: number): () => number { + let s = seed >>> 0; + return () => { + s = (s + 0x6d2b79f5) >>> 0; + let t = Math.imul(s ^ (s >>> 15), 1 | s); + t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t; + return ((t ^ (t >>> 14)) >>> 0) / 4294967296; + }; +} + // ============================================================================ // TYPES // ============================================================================ @@ -150,6 +168,7 @@ export const processCommunities = async ( leiden.detailed(graph, { resolution: isLarge ? 2.0 : 1.0, maxIterations: isLarge ? 3 : 0, + rng: createSeededRng(LEIDEN_SEED), }), ), new Promise((_, reject) => diff --git a/gitnexus/src/core/ingestion/languages/java.ts b/gitnexus/src/core/ingestion/languages/java.ts index 96139ccda..c70eacb10 100644 --- a/gitnexus/src/core/ingestion/languages/java.ts +++ b/gitnexus/src/core/ingestion/languages/java.ts @@ -27,6 +27,17 @@ import { javaMethodConfig } from '../method-extractors/configs/jvm.js'; import { createVariableExtractor } from '../variable-extractors/generic.js'; import { javaVariableConfig } from '../variable-extractors/configs/jvm.js'; import { createHeritageExtractor } from '../heritage-extractors/generic.js'; +import { + emitJavaScopeCaptures, + interpretJavaImport, + interpretJavaTypeBinding, + javaBindingScopeFor, + javaImportOwningScope, + javaMergeBindings, + javaReceiverBinding, + javaArityCompatibility, + resolveJavaImportTarget, +} from './java/index.js'; export const javaProvider = defineLanguage({ id: SupportedLanguages.Java, @@ -65,4 +76,15 @@ export const javaProvider = defineLanguage({ variableExtractor: createVariableExtractor(javaVariableConfig), classExtractor: createClassExtractor(javaClassConfig), heritageExtractor: createHeritageExtractor(SupportedLanguages.Java), + + // ── RFC #909 Ring 3: scope-based resolution hooks ── + emitScopeCaptures: emitJavaScopeCaptures, + interpretImport: interpretJavaImport, + interpretTypeBinding: interpretJavaTypeBinding, + bindingScopeFor: javaBindingScopeFor, + importOwningScope: javaImportOwningScope, + mergeBindings: (_scope, bindings) => javaMergeBindings(bindings), + receiverBinding: javaReceiverBinding, + arityCompatibility: javaArityCompatibility, + resolveImportTarget: resolveJavaImportTarget, }); diff --git a/gitnexus/src/core/ingestion/languages/java/arity-metadata.ts b/gitnexus/src/core/ingestion/languages/java/arity-metadata.ts new file mode 100644 index 000000000..47cccbff9 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/arity-metadata.ts @@ -0,0 +1,49 @@ +/** + * Extract Java arity metadata from a method-like tree-sitter node — + * `method_declaration` or `constructor_declaration`. + * + * Reuses `javaMethodConfig.extractParameters` so scope-extracted defs + * carry the same arity semantics as the legacy parse-worker path: + * - varargs (`...`) collapses `parameterCount` to `undefined` + * - `parameterTypes` collects declared type names; a literal + * `'varargs'` marker is appended for variadic methods so + * `javaArityCompatibility` can detect them. + */ + +import type { SyntaxNode } from '../../utils/ast-helpers.js'; +import { javaMethodConfig } from '../../method-extractors/configs/jvm.js'; + +export interface JavaArityMetadata { + readonly parameterCount: number | undefined; + readonly requiredParameterCount: number | undefined; + readonly parameterTypes: readonly string[] | undefined; +} + +export function computeJavaArityMetadata(fnNode: SyntaxNode): JavaArityMetadata { + const params = javaMethodConfig.extractParameters?.(fnNode) ?? []; + + let hasVariadic = false; + const types: string[] = []; + for (const p of params) { + if (p.isVariadic) hasVariadic = true; + if (p.type !== null) types.push(p.type); + } + if (hasVariadic) types.push('varargs'); + + const total = params.length; + // For varargs methods, `parameterCount` (max) is unknown — any number of + // trailing arguments is valid. But the fixed-prefix parameters (everything + // before the variadic `...` param) are still required, so we preserve that + // count in `requiredParameterCount` so `javaArityCompatibility` can reject + // calls that undersupply the fixed prefix (e.g. `f(int x, String... args)` + // called with 0 args). + const fixedCount = params.filter((p) => !p.isVariadic).length; + const parameterCount = hasVariadic ? undefined : total; + const requiredParameterCount = hasVariadic ? fixedCount : total; + + return { + parameterCount, + requiredParameterCount, + parameterTypes: types.length > 0 ? types : undefined, + }; +} diff --git a/gitnexus/src/core/ingestion/languages/java/arity.ts b/gitnexus/src/core/ingestion/languages/java/arity.ts new file mode 100644 index 000000000..f98a8209d --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/arity.ts @@ -0,0 +1,31 @@ +/** + * Java arity check, accommodating varargs (`...`). + * + * Verdicts: + * - `'compatible'` — argCount matches parameterCount, OR varargs present. + * - `'incompatible'` — argCount mismatches with no varargs. + * - `'unknown'` — metadata absent / incomplete. + */ + +import type { Callsite, SymbolDefinition } from 'gitnexus-shared'; + +export function javaArityCompatibility( + def: SymbolDefinition, + callsite: Callsite, +): 'compatible' | 'unknown' | 'incompatible' { + const max = def.parameterCount; + const min = def.requiredParameterCount; + if (max === undefined && min === undefined) return 'unknown'; + + const argCount = callsite.arity; + if (!Number.isFinite(argCount) || argCount < 0) return 'unknown'; + + const hasVarArgs = + def.parameterTypes !== undefined && + def.parameterTypes.some((t) => t === 'varargs' || t.includes('...')); + + if (min !== undefined && argCount < min) return 'incompatible'; + if (max !== undefined && argCount > max && !hasVarArgs) return 'incompatible'; + + return 'compatible'; +} diff --git a/gitnexus/src/core/ingestion/languages/java/cache-stats.ts b/gitnexus/src/core/ingestion/languages/java/cache-stats.ts new file mode 100644 index 000000000..a4c58f11f --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/cache-stats.ts @@ -0,0 +1,30 @@ +/** + * Dev-mode counters for the cross-phase scope-captures parse cache + * (Java mirror of `languages/csharp/cache-stats.ts`). + * + * Gated by `PROF_SCOPE_RESOLUTION=1`. Production builds fold every + * increment into dead code via the module-level `PROF` constant, so + * the hot path in `captures.ts` stays branch-free. + */ + +const PROF = process.env.PROF_SCOPE_RESOLUTION === '1'; + +let CACHE_HITS = 0; +let CACHE_MISSES = 0; + +export function recordCacheHit(): void { + if (PROF) CACHE_HITS++; +} + +export function recordCacheMiss(): void { + if (PROF) CACHE_MISSES++; +} + +export function getJavaCaptureCacheStats(): { hits: number; misses: number } { + return { hits: CACHE_HITS, misses: CACHE_MISSES }; +} + +export function resetJavaCaptureCacheStats(): void { + CACHE_HITS = 0; + CACHE_MISSES = 0; +} diff --git a/gitnexus/src/core/ingestion/languages/java/captures.ts b/gitnexus/src/core/ingestion/languages/java/captures.ts new file mode 100644 index 000000000..73ea605fe --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/captures.ts @@ -0,0 +1,235 @@ +/** + * `emitScopeCaptures` for Java. + * + * Drives the Java scope query against tree-sitter-java and groups raw + * matches into `CaptureMatch[]` for the central extractor. Layers: + * + * 1. **Decomposed import declarations** — each `import_declaration` + * is re-emitted with `@import.kind/source/name` markers. + * 2. **Receiver binding synthesis** — `this`/`super` type-bindings + * on instance methods. + * 3. **Arity metadata** on method/constructor declarations. + * 4. **Reference arity** on call sites. + * + * Pure given the input source text. No I/O, no globals consulted. + */ + +import type { Capture, CaptureMatch } from 'gitnexus-shared'; +import { findNodeAtRange, nodeToCapture, syntheticCapture } from '../../utils/ast-helpers.js'; +import { splitImportDeclaration } from './import-decomposer.js'; +import { computeJavaArityMetadata } from './arity-metadata.js'; +import { synthesizeJavaReceiverBinding } from './receiver-binding.js'; +import { getJavaParser, getJavaScopeQuery } from './query.js'; +import { recordCacheHit, recordCacheMiss } from './cache-stats.js'; +import { getTreeSitterBufferSize } from '../../constants.js'; +import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js'; + +/** Declaration anchors that carry function-like arity metadata. */ +const FUNCTION_DECL_TAGS = ['@declaration.method', '@declaration.constructor'] as const; + +/** tree-sitter-java node types that the method extractor accepts. */ +const FUNCTION_NODE_TYPES = ['method_declaration', 'constructor_declaration'] as const; + +/** Suppress read.member emissions when the field_access is already + * covered by a method_invocation (object of a call) or an + * assignment_expression (write target). */ +function shouldEmitReadMember(memberNode: SyntaxNode): boolean { + const parent = memberNode.parent; + if (parent === null) return true; + + switch (parent.type) { + case 'method_invocation': + // Don't emit read.member when the field_access is the object of a method_invocation + // (the method call already handles this relationship) + return parent.childForFieldName('object')?.id !== memberNode.id; + case 'assignment_expression': + return parent.childForFieldName('left')?.id !== memberNode.id; + default: + return true; + } +} + +export function emitJavaScopeCaptures( + sourceText: string, + _filePath: string, + cachedTree?: unknown, +): readonly CaptureMatch[] { + let tree = cachedTree as ReturnType['parse']> | undefined; + if (tree === undefined) { + tree = parseSourceSafe(getJavaParser(), sourceText, undefined, { + bufferSize: getTreeSitterBufferSize(sourceText), + }); + recordCacheMiss(); + } else { + recordCacheHit(); + } + + const rawMatches = getJavaScopeQuery().matches(tree.rootNode); + const out: CaptureMatch[] = []; + + for (const m of rawMatches) { + const grouped: Record = {}; + for (const c of m.captures) { + const tag = '@' + c.name; + grouped[tag] = nodeToCapture(tag, c.node); + } + if (Object.keys(grouped).length === 0) continue; + + // Decompose each `import_declaration`. + if (grouped['@import.statement'] !== undefined) { + const stmtCapture = grouped['@import.statement']; + const stmtNode = findNodeAtRange(tree.rootNode, stmtCapture.range, 'import_declaration'); + if (stmtNode !== null) { + const decomposed = splitImportDeclaration(stmtNode); + if (decomposed !== null) { + out.push(decomposed); + continue; + } + } + out.push(grouped); + continue; + } + + // Skip free-call matches that are actually member calls. The query + // matches ALL method_invocations as @reference.call.free (without + // negation) because tree-sitter-java's query engine drops !object + // patterns when a positive object: pattern exists for the same node + // type. Filter here: if the match has @reference.call.free but also + // has @reference.receiver, it's a member call — skip the free match + // (the separate @reference.call.member match covers it). + if ( + grouped['@reference.call.free'] !== undefined && + grouped['@reference.receiver'] !== undefined + ) { + continue; + } + + // Filter read.member when it's a child of method_invocation or assignment. + if (grouped['@reference.read.member'] !== undefined) { + const anchor = grouped['@reference.read.member']; + const memberNode = findNodeAtRange(tree.rootNode, anchor.range, 'field_access'); + if (memberNode === null || !shouldEmitReadMember(memberNode)) { + continue; + } + } + + // Synthesize `this` / `super` receiver type-bindings on every + // instance method-like. + if (grouped['@scope.function'] !== undefined) { + out.push(grouped); + const anchor = grouped['@scope.function']!; + const fnNode = findFunctionNode(tree.rootNode, anchor.range); + if (fnNode !== null) { + for (const synth of synthesizeJavaReceiverBinding(fnNode)) { + out.push(synth); + } + } + continue; + } + + // Synthesize arity metadata on function-like declarations. + const declTag = FUNCTION_DECL_TAGS.find((t) => grouped[t] !== undefined); + if (declTag !== undefined) { + const anchor = grouped[declTag]!; + const fnNode = findFunctionNode(tree.rootNode, anchor.range); + if (fnNode !== null) { + const arity = computeJavaArityMetadata(fnNode); + if (arity.parameterCount !== undefined) { + grouped['@declaration.parameter-count'] = syntheticCapture( + '@declaration.parameter-count', + fnNode, + String(arity.parameterCount), + ); + } + if (arity.requiredParameterCount !== undefined) { + grouped['@declaration.required-parameter-count'] = syntheticCapture( + '@declaration.required-parameter-count', + fnNode, + String(arity.requiredParameterCount), + ); + } + if (arity.parameterTypes !== undefined) { + grouped['@declaration.parameter-types'] = syntheticCapture( + '@declaration.parameter-types', + fnNode, + JSON.stringify(arity.parameterTypes), + ); + } + } + } + + // Synthesize `@reference.arity` on every callsite. + const callTag = ( + ['@reference.call.free', '@reference.call.member', '@reference.call.constructor'] as const + ).find((t) => grouped[t] !== undefined); + if (callTag !== undefined && grouped['@reference.arity'] === undefined) { + const anchor = grouped[callTag]!; + const callNode = + findNodeAtRange(tree.rootNode, anchor.range, 'method_invocation') ?? + findNodeAtRange(tree.rootNode, anchor.range, 'object_creation_expression'); + if (callNode !== null) { + const argList = callNode.childForFieldName('arguments'); + const args = + argList === null + ? [] + : argList.namedChildren.filter((c) => c !== null && c.type !== 'comment'); + grouped['@reference.arity'] = syntheticCapture( + '@reference.arity', + callNode, + String(args.length), + ); + + const argTypes = args.map((arg) => inferArgType(arg!)); + grouped['@reference.parameter-types'] = syntheticCapture( + '@reference.parameter-types', + callNode, + JSON.stringify(argTypes), + ); + } + } + + out.push(grouped); + } + + return out; +} + +type SyntaxNode = ReturnType['parse']>['rootNode']; + +/** Infer a Java argument's static type from literal patterns. */ +function inferArgType(argNode: SyntaxNode): string { + switch (argNode.type) { + case 'decimal_integer_literal': + case 'hex_integer_literal': + case 'octal_integer_literal': + case 'binary_integer_literal': + return 'int'; + case 'decimal_floating_point_literal': + case 'hex_floating_point_literal': + return 'double'; + case 'string_literal': + return 'String'; + case 'character_literal': + return 'char'; + case 'true': + case 'false': + return 'boolean'; + case 'null_literal': + return 'null'; + case 'object_creation_expression': { + const typeNode = argNode.childForFieldName('type'); + return typeNode?.text ?? ''; + } + default: + return ''; + } +} + +/** Find the first Java function-like node at the given range. */ +function findFunctionNode(rootNode: SyntaxNode, range: Capture['range']): SyntaxNode | null { + for (const nodeType of FUNCTION_NODE_TYPES) { + const n = findNodeAtRange(rootNode, range, nodeType); + if (n !== null) return n as SyntaxNode; + } + return null; +} diff --git a/gitnexus/src/core/ingestion/languages/java/import-decomposer.ts b/gitnexus/src/core/ingestion/languages/java/import-decomposer.ts new file mode 100644 index 000000000..59c74d144 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/import-decomposer.ts @@ -0,0 +1,104 @@ +/** + * Decompose a Java `import_declaration` into a `CaptureMatch` carrying + * the synthesized markers `@import.kind` / `@import.source` / + * `@import.name` that `interpretJavaImport` consumes. + * + * Unlike C#'s using-directive decomposer, Java has four import forms: + * + * import com.example.User; → named + * import com.example.*; → wildcard + * import static com.example.Utils.format; → static + * import static com.example.Utils.*; → static-wildcard + * + * Each produces exactly one import. The decomposer inspects the raw + * source text and tree-sitter children to determine the flavor. + */ + +import type { Capture, CaptureMatch } from 'gitnexus-shared'; +import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; + +type ImportKind = 'named' | 'wildcard' | 'static' | 'static-wildcard'; + +interface ImportSpec { + readonly kind: ImportKind; + /** Full dotted path: `com.example.User`. */ + readonly source: string; + /** Local binding name — last path segment for named/static, + * `'*'` for wildcard/static-wildcard. */ + readonly name: string; + /** Node to anchor the synthesized captures (range-wise). */ + readonly atNode: SyntaxNode; +} + +export function splitImportDeclaration(stmtNode: SyntaxNode): CaptureMatch | null { + if (stmtNode.type !== 'import_declaration') return null; + const spec = parseImportDeclaration(stmtNode); + if (spec === null) return null; + return buildImportMatch(stmtNode, spec); +} + +function parseImportDeclaration(node: SyntaxNode): ImportSpec | null { + // Detect `static` by checking for an anonymous `static` token child. + let isStatic = false; + for (let i = 0; i < node.childCount; i++) { + const child = node.child(i); + if (child !== null && child.type === 'static') { + isStatic = true; + break; + } + } + + // Detect wildcard by checking for `asterisk` named child. + let isWildcard = false; + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child !== null && child.type === 'asterisk') { + isWildcard = true; + break; + } + } + + // Find the scoped_identifier (or identifier for single-segment imports). + let pathNode: SyntaxNode | null = null; + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child !== null && (child.type === 'scoped_identifier' || child.type === 'identifier')) { + pathNode = child; + break; + } + } + if (pathNode === null) return null; + + const fullPath = pathNode.text; + if (fullPath === '') return null; + + if (isStatic && isWildcard) { + // `import static com.example.Utils.*;` + return { kind: 'static-wildcard', source: fullPath, name: '*', atNode: node }; + } + if (isStatic) { + // `import static com.example.Utils.format;` + const lastDot = fullPath.lastIndexOf('.'); + const name = lastDot >= 0 ? fullPath.slice(lastDot + 1) : fullPath; + return { kind: 'static', source: fullPath, name, atNode: node }; + } + if (isWildcard) { + // `import com.example.*;` + return { kind: 'wildcard', source: fullPath, name: '*', atNode: node }; + } + + // `import com.example.User;` + const lastDot = fullPath.lastIndexOf('.'); + const name = lastDot >= 0 ? fullPath.slice(lastDot + 1) : fullPath; + return { kind: 'named', source: fullPath, name, atNode: node }; +} + +function buildImportMatch(stmtNode: SyntaxNode, spec: ImportSpec): CaptureMatch { + const m: Record = { + '@import.statement': nodeToCapture('@import.statement', stmtNode), + '@import.kind': syntheticCapture('@import.kind', spec.atNode, spec.kind), + '@import.source': syntheticCapture('@import.source', spec.atNode, spec.source), + '@import.name': syntheticCapture('@import.name', spec.atNode, spec.name), + }; + return m; +} diff --git a/gitnexus/src/core/ingestion/languages/java/import-target.ts b/gitnexus/src/core/ingestion/languages/java/import-target.ts new file mode 100644 index 000000000..b78b6369b --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/import-target.ts @@ -0,0 +1,108 @@ +/** + * Adapter from `(ParsedImport, WorkspaceIndex)` → concrete file path. + * + * Converts Java package paths (dots → slashes) and tries: + * 1. Exact file match: `com/example/User.java` + * 2. Suffix match for nested layouts + * 3. Directory match (wildcard imports) + * 4. Progressive prefix stripping for non-standard layouts + * + * Returns `null` for unresolvable / JDK imports. + */ + +import type { ParsedImport, WorkspaceIndex } from 'gitnexus-shared'; + +export interface JavaResolveContext { + readonly fromFile: string; + readonly allFilePaths: ReadonlySet; +} + +export function resolveJavaImportTarget( + parsedImport: ParsedImport, + workspaceIndex: WorkspaceIndex, +): string | null { + const ctx = workspaceIndex as JavaResolveContext | undefined; + if ( + ctx === undefined || + typeof (ctx as { fromFile?: unknown }).fromFile !== 'string' || + !((ctx as { allFilePaths?: unknown }).allFilePaths instanceof Set) + ) { + return null; + } + if (parsedImport.kind === 'dynamic-unresolved') return null; + if (parsedImport.targetRaw === null || parsedImport.targetRaw === '') return null; + + // Strip trailing `.*` for wildcard imports: `com.example.*` → `com.example` + let target = parsedImport.targetRaw; + if (target.endsWith('.*')) { + target = target.slice(0, -2); + } + + // Package path: `com.example.User` → `com/example/User` + const pathLike = target.replace(/\./g, '/'); + const suffix = `/${pathLike}`; + + let exactFile: string | null = null; + let suffixFile: string | null = null; + let directoryChild: string | null = null; + const dirPrefix = `${pathLike}/`; + const suffixDirPrefix = `/${dirPrefix}`; + + for (const raw of ctx.allFilePaths) { + const f = raw.replace(/\\/g, '/'); + if (!f.endsWith('.java')) continue; + if (f === `${pathLike}.java`) { + exactFile = raw; + break; + } + if (suffixFile === null && f.endsWith(`${suffix}.java`)) { + suffixFile = raw; + } + if (directoryChild === null) { + const atRoot = f.startsWith(dirPrefix); + const atNested = f.includes(suffixDirPrefix); + if (atRoot || atNested) { + const idx = atRoot ? 0 : f.indexOf(suffixDirPrefix) + 1; + const after = f.slice(idx + dirPrefix.length); + if (after.length > 0 && !after.includes('/')) { + directoryChild = raw; + } + } + } + } + + if (exactFile !== null) return exactFile; + if (suffixFile !== null) return suffixFile; + if (directoryChild !== null) return directoryChild; + + // Progressive prefix stripping — handles `import com.example.User;` + // in a repo laid out `User.java` (no `com/example/` prefix). + const segments = pathLike.split('/').filter(Boolean); + for (let skip = 1; skip < segments.length; skip++) { + const tail = segments.slice(skip).join('/'); + if (tail === '') continue; + const tailFile = `${tail}.java`; + const tailSuffix = `/${tailFile}`; + const tailDir = `${tail}/`; + const tailSuffixDir = `/${tailDir}`; + let tailDirectChild: string | null = null; + for (const raw of ctx.allFilePaths) { + const f = raw.replace(/\\/g, '/'); + if (!f.endsWith('.java')) continue; + if (f === tailFile) return raw; + if (f.endsWith(tailSuffix)) return raw; + if (tailDirectChild === null) { + const atRoot = f.startsWith(tailDir); + const atNested = f.includes(tailSuffixDir); + if (atRoot || atNested) { + const idx = atRoot ? 0 : f.indexOf(tailSuffixDir) + 1; + const after = f.slice(idx + tailDir.length); + if (after.length > 0 && !after.includes('/')) tailDirectChild = raw; + } + } + } + if (tailDirectChild !== null) return tailDirectChild; + } + + return null; +} diff --git a/gitnexus/src/core/ingestion/languages/java/index.ts b/gitnexus/src/core/ingestion/languages/java/index.ts new file mode 100644 index 000000000..443faac4b --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/index.ts @@ -0,0 +1,30 @@ +/** + * Java scope-resolution hooks (RFC #909 Ring 3). + * + * Public API barrel. Consumers should import from this file rather than + * the individual modules. + * + * Module layout: + * + * - `query.ts` — tree-sitter query + lazy parser/query singletons + * - `captures.ts` — `emitJavaScopeCaptures` orchestrator + * - `import-decomposer.ts` — each `import` → ParsedImport-shaped captures + * - `interpret.ts` — capture-match → `ParsedImport` / `ParsedTypeBinding` + * - `simple-hooks.ts` — small hooks made explicit + * - `receiver-binding.ts` — synthesize `this`/`super` type-bindings on + * instance-method entry + * - `merge-bindings.ts` — Java import precedence + * - `arity.ts` — Java arity compatibility (varargs) + * - `arity-metadata.ts` — synthesize arity metadata from declarations + * - `import-target.ts` — `(ParsedImport, WorkspaceIndex) → file path` adapter + * - `scope-resolver.ts` — `ScopeResolver` registered in `SCOPE_RESOLVERS` + * - `cache-stats.ts` — PROF_SCOPE_RESOLUTION cache hit/miss counters + */ + +export { emitJavaScopeCaptures } from './captures.js'; +export { getJavaCaptureCacheStats, resetJavaCaptureCacheStats } from './cache-stats.js'; +export { interpretJavaImport, interpretJavaTypeBinding } from './interpret.js'; +export { javaMergeBindings } from './merge-bindings.js'; +export { javaArityCompatibility } from './arity.js'; +export { resolveJavaImportTarget, type JavaResolveContext } from './import-target.js'; +export { javaBindingScopeFor, javaImportOwningScope, javaReceiverBinding } from './simple-hooks.js'; diff --git a/gitnexus/src/core/ingestion/languages/java/interpret.ts b/gitnexus/src/core/ingestion/languages/java/interpret.ts new file mode 100644 index 000000000..9c207d451 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/interpret.ts @@ -0,0 +1,141 @@ +/** + * Capture-match → semantic-shape interpreters for Java. + * + * - `interpretJavaImport` → `ParsedImport` + * - `interpretJavaTypeBinding` → `ParsedTypeBinding` + * + * Import matches arrive pre-decomposed by `emitJavaScopeCaptures` + * (one import per match, with synthesized `@import.kind/source/name` + * markers). Type-binding matches arrive from the raw query captures. + */ + +import type { CaptureMatch, ParsedImport, ParsedTypeBinding, TypeRef } from 'gitnexus-shared'; + +// ─── interpretImport ────────────────────────────────────────────────────── + +export function interpretJavaImport(captures: CaptureMatch): ParsedImport | null { + const kindCap = captures['@import.kind']; + const sourceCap = captures['@import.source']; + const nameCap = captures['@import.name']; + + const kind = kindCap?.text; + if (kind === undefined || sourceCap === undefined) return null; + + switch (kind) { + case 'named': { + // `import com.example.User;` + return { + kind: 'named', + localName: nameCap?.text ?? sourceCap.text.split('.').pop() ?? sourceCap.text, + importedName: sourceCap.text, + targetRaw: sourceCap.text, + }; + } + case 'wildcard': { + // `import com.example.*;` + return { + kind: 'wildcard', + targetRaw: sourceCap.text + '.*', + }; + } + case 'static': { + // `import static com.example.Utils.format;` + // The source contains the full path including the member name + // (e.g. `com.example.Utils.format`). For file resolution we need + // the class path (`com.example.Utils`), so strip the final member + // segment. The local binding name is the member itself. + const fullSource = sourceCap.text; + const lastDot = fullSource.lastIndexOf('.'); + const classPath = lastDot >= 0 ? fullSource.slice(0, lastDot) : fullSource; + return { + kind: 'named', + localName: nameCap?.text ?? (lastDot >= 0 ? fullSource.slice(lastDot + 1) : fullSource), + importedName: fullSource, + targetRaw: classPath, + }; + } + case 'static-wildcard': { + // `import static com.example.Utils.*;` + // The source is the class path (e.g. `com.example.Utils`). + // Resolution should target the class file, not a wildcard directory + // scan — `Utils.java` is the file that contains the static members. + return { + kind: 'wildcard', + targetRaw: sourceCap.text + '.*', + }; + } + default: + return null; + } +} + +// ─── interpretTypeBinding ───────────────────────────────────────────────── + +export function interpretJavaTypeBinding(captures: CaptureMatch): ParsedTypeBinding | null { + const nameCap = captures['@type-binding.name']; + const typeCap = captures['@type-binding.type']; + if (nameCap === undefined || typeCap === undefined) return null; + + // Strip qualifier first so that `com.example.BaseModel` becomes + // `BaseModel` before stripGeneric — the JVM-erasure fallback pattern + // requires an unqualified identifier at the start of the string. + const rawType = stripGeneric(stripQualifier(typeCap.text.trim())); + + // Skip `var` — tree-sitter-java parses `var` as type_identifier with + // text "var". When used without a constructor initializer, there's no + // concrete type to bind. + if (rawType === 'var') return null; + + let source: TypeRef['source'] = 'parameter-annotation'; + if (captures['@type-binding.self'] !== undefined) source = 'self'; + else if (captures['@type-binding.constructor'] !== undefined) source = 'constructor-inferred'; + else if (captures['@type-binding.annotation'] !== undefined) source = 'annotation'; + else if (captures['@type-binding.return'] !== undefined) source = 'return-annotation'; + + return { boundName: nameCap.text, rawTypeName: rawType, source }; +} + +/** + * Unwrap generic type parameters from Java types. + * + * Three tiers, checked in order: + * 1. Known single-arg collection wrappers → extract the element type + * (`List` → `User`, `Optional` → `User`). + * 2. Known two-arg map/container types → extract the value type + * (`Map` → `User`). + * 3. **Fallback (JVM type erasure):** any other generic type → + * strip the generic parameters and keep the raw class name + * (`BaseModel` → `BaseModel`, `CustomList` → `CustomList`). + * This ensures receiver bindings (`this`/`super`) on classes with + * generic superclasses resolve to the correct class file. + */ +function stripGeneric(text: string): string { + // Single-type-argument containers — extract the element type. + const single = text.match( + /^(?:[A-Za-z_][A-Za-z0-9_.]*\.)?(?:List|ArrayList|LinkedList|Set|HashSet|TreeSet|SortedSet|LinkedHashSet|Collection|Iterable|Iterator|Optional|Stream|CompletableFuture|Future|Queue|Deque|ArrayDeque|PriorityQueue|Vector|Stack|Supplier|Consumer|Predicate|Function)<([^,<>]+)>$/, + ); + if (single !== null) return single[1].trim(); + + // Two-type-argument map/container types — extract the value type (second arg). + const twoArg = text.match( + /^(?:[A-Za-z_][A-Za-z0-9_.]*\.)?(?:Map|HashMap|TreeMap|LinkedHashMap|ConcurrentHashMap|ConcurrentMap|SortedMap|NavigableMap|Hashtable|EnumMap|WeakHashMap|IdentityHashMap|BiFunction|BiConsumer|BiPredicate|Pair|Entry)<[^,<>]+,\s*([^,<>]+)>$/, + ); + if (twoArg !== null) return twoArg[1].trim(); + + // Fallback: strip generic parameters from any unrecognized generic type. + // `BaseModel` → `BaseModel`, `Builder` → `Builder`. + // This mirrors JVM type erasure — the raw class name is the resolvable symbol. + // The pattern matches up to the first `<` to handle nested generics safely + // (e.g. `BaseModel>` → `BaseModel`). + const fallback = text.match(/^([A-Za-z_$][A-Za-z0-9_$]*)<.+>$/s); + if (fallback !== null) return fallback[1].trim(); + + return text; +} + +/** `com.example.User` → `User`. */ +function stripQualifier(text: string): string { + const lastDot = text.lastIndexOf('.'); + if (lastDot === -1) return text; + return text.slice(lastDot + 1); +} diff --git a/gitnexus/src/core/ingestion/languages/java/merge-bindings.ts b/gitnexus/src/core/ingestion/languages/java/merge-bindings.ts new file mode 100644 index 000000000..9056705d0 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/merge-bindings.ts @@ -0,0 +1,44 @@ +/** + * Java shadowing precedence for the `mergeBindings` hook. + * + * Tier ranking (lower wins): + * - 0: `local` — class member, method, local variable, parameter + * - 1: `import` / `namespace` / `reexport` — explicit imports + * - 2: `wildcard` — wildcard imports (`import x.y.*`) + * + * Within a surviving tier: de-dup by DefId, last-write-wins. + */ + +import type { BindingRef } from 'gitnexus-shared'; + +const TIER_LOCAL = 0; +const TIER_IMPORT = 1; +const TIER_WILDCARD = 2; +const TIER_UNKNOWN = 3; + +function tierOf(b: BindingRef): number { + switch (b.origin) { + case 'local': + return TIER_LOCAL; + case 'reexport': + case 'import': + case 'namespace': + return TIER_IMPORT; + case 'wildcard': + return TIER_WILDCARD; + default: + return TIER_UNKNOWN; + } +} + +export function javaMergeBindings(bindings: readonly BindingRef[]): readonly BindingRef[] { + if (bindings.length === 0) return bindings; + + let bestTier = Number.POSITIVE_INFINITY; + for (const b of bindings) bestTier = Math.min(bestTier, tierOf(b)); + const survivors = bindings.filter((b) => tierOf(b) === bestTier); + + const seen = new Map(); + for (const b of survivors) seen.set(b.def.nodeId, b); + return [...seen.values()]; +} diff --git a/gitnexus/src/core/ingestion/languages/java/query.ts b/gitnexus/src/core/ingestion/languages/java/query.ts new file mode 100644 index 000000000..3fabbb7bf --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/query.ts @@ -0,0 +1,197 @@ +/** + * Tree-sitter query for Java scope captures (RFC §5.1). + * + * Captures the structural skeleton the generic scope-resolution + * pipeline consumes: scopes (module/class/function), declarations + * (class-likes, method-likes, fields, variables), imports (import + * declarations), type bindings (parameter annotations, variable + * annotations, constructor inference), and references (call sites, + * member writes/reads). + * + * Java specifics that shape this query: + * + * - Java uses `program` as the root node (not `compilation_unit`). + * - `import_declaration` nodes carry `scoped_identifier` children + * and optional `asterisk` for wildcard imports. + * - `static` imports are detected by an anonymous `static` token + * child within `import_declaration`. + * - `var` (Java 10+ local variable type inference) parses as a + * `type_identifier` with text `"var"`, not a dedicated node type. + * - Modifiers (`public`, `static`, etc.) are grouped under a + * `modifiers` named child with anonymous keyword tokens. + * - Superclass inheritance uses a `superclass:` field containing + * a `superclass` node wrapping a `type_identifier`. + * + * Exposes lazy `Parser` and `Query` singletons so callers don't pay + * tree-sitter init cost per file. + */ + +import Parser from 'tree-sitter'; +import Java from 'tree-sitter-java'; + +const JAVA_SCOPE_QUERY = ` +;; Scopes +(program) @scope.module + +(class_declaration) @scope.class +(interface_declaration) @scope.class +(enum_declaration) @scope.class +(record_declaration) @scope.class +(annotation_type_declaration) @scope.class + +(method_declaration) @scope.function +(constructor_declaration) @scope.function + +;; Declarations — types +(class_declaration + name: (identifier) @declaration.name) @declaration.class + +(interface_declaration + name: (identifier) @declaration.name) @declaration.interface + +(enum_declaration + name: (identifier) @declaration.name) @declaration.enum + +(record_declaration + name: (identifier) @declaration.name) @declaration.record + +(annotation_type_declaration + name: (identifier) @declaration.name) @declaration.class + +;; Declarations — methods / constructors +(method_declaration + name: (identifier) @declaration.name) @declaration.method + +(constructor_declaration + name: (identifier) @declaration.name) @declaration.constructor + +;; Declarations — fields +(field_declaration + declarator: (variable_declarator + name: (identifier) @declaration.name)) @declaration.variable + +;; Declarations — local variables +(local_variable_declaration + declarator: (variable_declarator + name: (identifier) @declaration.name)) @declaration.variable + +;; Imports — single anchor per import_declaration +(import_declaration) @import.statement + +;; Type bindings — parameter annotations: void f(User u) +(formal_parameter + type: (type_identifier) @type-binding.type + name: (identifier) @type-binding.name) @type-binding.parameter + +(formal_parameter + type: (generic_type) @type-binding.type + name: (identifier) @type-binding.name) @type-binding.parameter + +(formal_parameter + type: (scoped_type_identifier) @type-binding.type + name: (identifier) @type-binding.name) @type-binding.parameter + +;; Type bindings — local variable annotations: User u = new User(); +(local_variable_declaration + type: (type_identifier) @type-binding.type + declarator: (variable_declarator + name: (identifier) @type-binding.name)) @type-binding.annotation + +(local_variable_declaration + type: (generic_type) @type-binding.type + declarator: (variable_declarator + name: (identifier) @type-binding.name)) @type-binding.annotation + +;; Type bindings — var u = new User(); (Java 10+ local variable type inference) +;; tree-sitter-java parses \`var\` as a \`type_identifier\` with text "var". +;; The type-binding.constructor anchor fires when the rhs is an +;; object_creation_expression so interpretJavaTypeBinding can infer +;; the concrete type from the constructor call. +(local_variable_declaration + type: (type_identifier) @_var_type + declarator: (variable_declarator + name: (identifier) @type-binding.name + value: (object_creation_expression + type: (type_identifier) @type-binding.type))) @type-binding.constructor + +;; Type bindings — field declarations: private User user; +(field_declaration + type: (type_identifier) @type-binding.type + declarator: (variable_declarator + name: (identifier) @type-binding.name)) @type-binding.annotation + +(field_declaration + type: (generic_type) @type-binding.type + declarator: (variable_declarator + name: (identifier) @type-binding.name)) @type-binding.annotation + +;; Type bindings — method return type: public User getUser() { } +(method_declaration + type: (type_identifier) @type-binding.type + name: (identifier) @type-binding.name) @type-binding.return + +(method_declaration + type: (generic_type) @type-binding.type + name: (identifier) @type-binding.name) @type-binding.return + +;; Type bindings — enhanced for: for (User u : list) +(enhanced_for_statement + type: (type_identifier) @type-binding.type + name: (identifier) @type-binding.name) @type-binding.annotation + +(enhanced_for_statement + type: (generic_type) @type-binding.type + name: (identifier) @type-binding.name) @type-binding.annotation + +;; References — all method calls: foo() and obj.method() +;; tree-sitter-java's query engine drops negation-based \`!object\` +;; patterns when a positive \`object:\` pattern exists for the same +;; node type, so we match all calls here and classify free vs +;; member in captures.ts based on the presence of @reference.receiver. +(method_invocation + object: (_) @reference.receiver + name: (identifier) @reference.name) @reference.call.member + +(method_invocation + name: (identifier) @reference.name) @reference.call.free + +;; References — constructor calls: new User(...) +(object_creation_expression + type: (type_identifier) @reference.name) @reference.call.constructor + +(object_creation_expression + type: (generic_type + (type_identifier) @reference.name)) @reference.call.constructor + +(object_creation_expression + type: (scoped_type_identifier) @reference.call.constructor.qualified) @reference.call.constructor + +;; References — field/property writes: obj.name = "x" +(assignment_expression + left: (field_access + object: (_) @reference.receiver + field: (identifier) @reference.name)) @reference.write.member + +;; References — field/property reads: obj.name +(field_access + object: (_) @reference.receiver + field: (identifier) @reference.name) @reference.read.member +`; + +let _parser: Parser | null = null; +let _query: Parser.Query | null = null; + +export function getJavaParser(): Parser { + if (_parser === null) { + _parser = new Parser(); + _parser.setLanguage(Java as Parameters[0]); + } + return _parser; +} + +export function getJavaScopeQuery(): Parser.Query { + if (_query === null) { + _query = new Parser.Query(Java as Parameters[0], JAVA_SCOPE_QUERY); + } + return _query; +} diff --git a/gitnexus/src/core/ingestion/languages/java/receiver-binding.ts b/gitnexus/src/core/ingestion/languages/java/receiver-binding.ts new file mode 100644 index 000000000..4d6d60ded --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/receiver-binding.ts @@ -0,0 +1,103 @@ +/** + * Synthesize `@type-binding.self` captures for Java instance methods — + * one for `this` (always on non-static methods inside a type + * declaration) and optionally one for `super` (only on class methods + * when the enclosing class has a `superclass`). + * + * Mirrors `languages/csharp/receiver-binding.ts` in structure. + */ + +import type { Capture, CaptureMatch } from 'gitnexus-shared'; +import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; + +const TYPE_DECL_NODE_TYPES = new Set([ + 'class_declaration', + 'interface_declaration', + 'enum_declaration', + 'record_declaration', +]); + +const FUNCTION_NODE_TYPES = new Set(['method_declaration', 'constructor_declaration']); + +/** Walk up to the enclosing type declaration. */ +function findEnclosingTypeDeclaration(node: SyntaxNode): SyntaxNode | null { + let cur: SyntaxNode | null = node.parent; + while (cur !== null) { + if (TYPE_DECL_NODE_TYPES.has(cur.type)) return cur; + cur = cur.parent; + } + return null; +} + +function typeName(typeNode: SyntaxNode): string | null { + return typeNode.childForFieldName('name')?.text ?? null; +} + +/** First superclass text. tree-sitter-java uses a `superclass` field + * containing a `superclass` node wrapping a `type_identifier`. */ +function firstSuperclassText(typeNode: SyntaxNode): string | null { + const superclass = typeNode.childForFieldName('superclass'); + if (superclass === null) return null; + // The superclass node wraps the type_identifier + for (let i = 0; i < superclass.namedChildCount; i++) { + const child = superclass.namedChild(i); + if (child !== null && (child.type === 'type_identifier' || child.type === 'generic_type')) { + return child.text; + } + } + return null; +} + +/** Check if a method has the `static` modifier. In tree-sitter-java, + * modifiers are grouped under a `modifiers` named child with anonymous + * keyword tokens. */ +function isStaticMethod(fnNode: SyntaxNode): boolean { + for (let i = 0; i < fnNode.namedChildCount; i++) { + const child = fnNode.namedChild(i); + if (child !== null && child.type === 'modifiers') { + for (let j = 0; j < child.childCount; j++) { + const mod = child.child(j); + if (mod !== null && mod.text.trim() === 'static') return true; + } + } + } + return false; +} + +export function synthesizeJavaReceiverBinding(fnNode: SyntaxNode): CaptureMatch[] { + if (!FUNCTION_NODE_TYPES.has(fnNode.type)) return []; + if (isStaticMethod(fnNode)) return []; + + const enclosingType = findEnclosingTypeDeclaration(fnNode); + if (enclosingType === null) return []; + + const enclosingName = typeName(enclosingType); + if (enclosingName === null) return []; + + // Anchor to the method body so the synthesized captures are inside + // the function scope. + const anchorNode = fnNode.childForFieldName('body'); + if (anchorNode === null) return []; + + const out: CaptureMatch[] = []; + out.push(buildReceiverMatch(anchorNode, 'this', enclosingName)); + + // `super` applies only to class/record methods with an explicit superclass. + if (enclosingType.type === 'class_declaration' || enclosingType.type === 'record_declaration') { + const superText = firstSuperclassText(enclosingType); + if (superText !== null) { + out.push(buildReceiverMatch(anchorNode, 'super', superText)); + } + } + + return out; +} + +function buildReceiverMatch(anchorNode: SyntaxNode, name: string, typeText: string): CaptureMatch { + const m: Record = { + '@type-binding.self': nodeToCapture('@type-binding.self', anchorNode), + '@type-binding.name': syntheticCapture('@type-binding.name', anchorNode, name), + '@type-binding.type': syntheticCapture('@type-binding.type', anchorNode, typeText), + }; + return m; +} diff --git a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts new file mode 100644 index 000000000..dac974cc7 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts @@ -0,0 +1,97 @@ +/** + * Java `ScopeResolver` registered in `SCOPE_RESOLVERS` and consumed by + * the generic `runScopeResolution` orchestrator (RFC #909 Ring 3). + * + * ## Registry-primary parity status + * + * Java is **not** in `MIGRATED_LANGUAGES` — the scope-resolution + * registry runs in shadow mode only. Parity in forced registry mode + * (`REGISTRY_PRIMARY_JAVA=1`) is 143/172 (83%). The 29 gaps fall into: + * + * - switch pattern binding / sealed-class exhaustiveness + * - Map.values() / entrySet() iteration type propagation + * - assignment / method chain return-type propagation across files + * - virtual dispatch / interface default methods + * + * These are the same category of advanced-resolution gaps seen in prior + * migrations (Python, C#, Go). Parity is below the ≥99% flip threshold + * per RFC §6.4. + * + * **CI visibility:** Because Java is absent from `MIGRATED_LANGUAGES`, + * the parity CI workflow (`ci-scope-parity.yml`) does not run Java in + * either `REGISTRY_PRIMARY_JAVA=0` or `=1` mode. Regressions in forced + * mode are only visible via manual `REGISTRY_PRIMARY_JAVA=1 npx vitest + * run java.test.ts`. Before flipping Java to registry-primary, a + * non-required CI step should be added to run Java tests in forced mode + * and report parity as a dashboard input. + * + * **Parity baseline (29 failures):** The 29 gaps in forced registry mode + * are tracked in this PR (#1482) and this JSDoc. If the gap count + * changes (up or down), update this baseline accordingly. + * + * ### Known flip-blockers (must fix before adding to MIGRATED_LANGUAGES) + * + * - Varargs arity: fixed-prefix count is now preserved, but no + * integration fixture exercises the 0-arg rejection path yet. + * - Static import resolution: `import static X.Y.m` now correctly + * resolves to `X/Y.java` (the class), not `X/Y/m.java` (the member). + * Edge cases with nested classes may remain. + * - Generic superclass receiver binding: `BaseModel` now strips + * to `BaseModel` via JVM type-erasure fallback in `stripGeneric`. + * - Wildcard import (`import com.example.*`) file selection is + * nondeterministic when multiple classes share a package directory. + * May produce wrong-file edges in forced mode. + * - Qualified generic type parameters in field/parameter annotations + * (`com.example.BaseModel`) — rare in practice but may miss + * resolution when the full qualifier is present with generics. + */ + +import type { ParsedFile } from 'gitnexus-shared'; +import { SupportedLanguages } from 'gitnexus-shared'; +import { buildMro, defaultLinearize } from '../../scope-resolution/passes/mro.js'; +import { populateClassOwnedMembers } from '../../scope-resolution/scope/walkers.js'; +import type { ScopeResolver } from '../../scope-resolution/contract/scope-resolver.js'; +import { javaProvider } from '../java.js'; +import { + javaArityCompatibility, + javaMergeBindings, + resolveJavaImportTarget, + type JavaResolveContext, +} from './index.js'; + +const javaScopeResolver: ScopeResolver = { + language: SupportedLanguages.Java, + languageProvider: javaProvider, + importEdgeReason: 'java-scope: import', + + resolveImportTarget: (targetRaw, fromFile, allFilePaths) => { + const ws: JavaResolveContext = { fromFile, allFilePaths }; + return resolveJavaImportTarget( + { kind: 'named', localName: '_', importedName: '_', targetRaw }, + ws, + ); + }, + + mergeBindings: (existing, incoming) => [...javaMergeBindings([...existing, ...incoming])], + + arityCompatibility: (callsite, def) => javaArityCompatibility(def, callsite), + + buildMro: (graph, parsedFiles, nodeLookup) => + buildMro(graph, parsedFiles, nodeLookup, defaultLinearize), + + populateOwners: (parsed: ParsedFile) => populateClassOwnedMembers(parsed), + + isSuperReceiver: (text) => text.trim() === 'super', + + // Java is statically typed — field-fallback heuristic stays off + fieldFallbackOnMethodLookup: false, + propagatesReturnTypesAcrossImports: true, + + // Java doesn't collapse member calls + collapseMemberCallsByCallerTarget: false, + + // Hoist return-type bindings to Module scope for cross-file propagation + hoistTypeBindingsToModule: true, +}; + +export { javaScopeResolver }; diff --git a/gitnexus/src/core/ingestion/languages/java/simple-hooks.ts b/gitnexus/src/core/ingestion/languages/java/simple-hooks.ts new file mode 100644 index 000000000..e69f768a6 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/simple-hooks.ts @@ -0,0 +1,54 @@ +/** + * Small hooks for the Java provider. Each is a few lines; they make + * the provider's choice explicit rather than relying on defaults. + */ + +import type { + CaptureMatch, + ParsedImport, + Scope, + ScopeId, + ScopeTree, + TypeRef, +} from 'gitnexus-shared'; + +// ─── bindingScopeFor ────────────────────────────────────────────────────── + +/** Method return-type bindings hoist to Module scope so cross-file + * `propagateImportedReturnTypes` and chain-follow can find them. */ +export function javaBindingScopeFor( + decl: CaptureMatch, + innermost: Scope, + tree: ScopeTree, +): ScopeId | null { + if (decl['@type-binding.return'] !== undefined) { + let cur: Scope | undefined = innermost; + while (cur !== undefined && cur.kind !== 'Module') { + const parentId: ScopeId | null = cur.parent ?? null; + if (parentId === null) break; + cur = tree.getScope(parentId); + } + if (cur !== undefined && cur.kind === 'Module') return cur.id; + } + return null; +} + +// ─── importOwningScope ──────────────────────────────────────────────────── + +/** Java imports are always at compilation-unit (Module) level (JLS §7.5). + * Return `null` unconditionally so the default Module scope is used. */ +export function javaImportOwningScope( + _imp: ParsedImport, + _innermost: Scope, + _tree: ScopeTree, +): ScopeId | null { + return null; +} + +// ─── receiverBinding ────────────────────────────────────────────────────── + +/** Look up `this` or `super` in the function scope's type bindings. */ +export function javaReceiverBinding(functionScope: Scope): TypeRef | null { + if (functionScope.kind !== 'Function') return null; + return functionScope.typeBindings.get('this') ?? functionScope.typeBindings.get('super') ?? null; +} diff --git a/gitnexus/src/core/ingestion/parsing-processor.ts b/gitnexus/src/core/ingestion/parsing-processor.ts index 7559b26bc..04a17db4f 100644 --- a/gitnexus/src/core/ingestion/parsing-processor.ts +++ b/gitnexus/src/core/ingestion/parsing-processor.ts @@ -82,6 +82,88 @@ export interface WorkerExtractedData { // Worker-based parallel parsing // ============================================================================ +/** + * Merge a list of `ParseWorkerResult`s into the running graph + symbol + * table state and produce the chunk-aggregated `WorkerExtractedData`. + * + * Extracted from `processParsingWithWorkers` so the same merge logic can + * be applied to both freshly-parsed worker output AND cached worker + * output replayed during incremental analyze. Idempotent on the + * accumulator fields (push-only); idempotent on graph if the caller + * starts from a clean graph (otherwise duplicate `addNode` calls are + * silently no-op'd by `KnowledgeGraph`). + */ +export const mergeChunkResults = ( + graph: KnowledgeGraph, + symbolTable: SymbolTableWriter, + chunkResults: readonly ParseWorkerResult[], +): WorkerExtractedData => { + const allImports: ExtractedImport[] = []; + const allCalls: ExtractedCall[] = []; + const allAssignments: ExtractedAssignment[] = []; + const allHeritage: ExtractedHeritage[] = []; + const allRoutes: ExtractedRoute[] = []; + const allFetchCalls: ExtractedFetchCall[] = []; + const allDecoratorRoutes: ExtractedDecoratorRoute[] = []; + const allToolDefs: ExtractedToolDef[] = []; + const allORMQueries: ExtractedORMQuery[] = []; + const allConstructorBindings: FileConstructorBindings[] = []; + const fileScopeBindingsByFile: FileScopeBindings[] = []; + const allParsedFiles: ParsedFile[] = []; + + for (const result of chunkResults) { + for (const node of result.nodes) { + graph.addNode({ + id: node.id, + label: node.label as NodeLabel, + properties: node.properties, + }); + } + for (const rel of result.relationships) { + graph.addRelationship(rel); + } + for (const sym of result.symbols) { + symbolTable.add(sym.filePath, sym.name, sym.nodeId, sym.type, { + parameterCount: sym.parameterCount, + requiredParameterCount: sym.requiredParameterCount, + parameterTypes: sym.parameterTypes, + returnType: sym.returnType, + declaredType: sym.declaredType, + ownerId: sym.ownerId, + qualifiedName: sym.qualifiedName, + }); + } + for (const item of result.imports) allImports.push(item); + for (const item of result.calls) allCalls.push(item); + for (const item of result.assignments) allAssignments.push(item); + for (const item of result.heritage) allHeritage.push(item); + for (const item of result.routes) allRoutes.push(item); + for (const item of result.fetchCalls) allFetchCalls.push(item); + for (const item of result.decoratorRoutes) allDecoratorRoutes.push(item); + for (const item of result.toolDefs) allToolDefs.push(item); + if (result.ormQueries) for (const item of result.ormQueries) allORMQueries.push(item); + for (const item of result.constructorBindings) allConstructorBindings.push(item); + if (result.fileScopeBindings) + for (const item of result.fileScopeBindings) fileScopeBindingsByFile.push(item); + if (result.parsedFiles) for (const item of result.parsedFiles) allParsedFiles.push(item); + } + + return { + imports: allImports, + calls: allCalls, + assignments: allAssignments, + heritage: allHeritage, + routes: allRoutes, + fetchCalls: allFetchCalls, + decoratorRoutes: allDecoratorRoutes, + toolDefs: allToolDefs, + ormQueries: allORMQueries, + constructorBindings: allConstructorBindings, + fileScopeBindings: fileScopeBindingsByFile, + parsedFiles: allParsedFiles, + }; +}; + const processParsingWithWorkers = async ( graph: KnowledgeGraph, files: { path: string; content: string }[], @@ -89,6 +171,14 @@ const processParsingWithWorkers = async ( astCache: ASTCache, workerPool: WorkerPool, onFileProgress?: FileProgressCallback, + /** + * When provided, populated with the raw worker results before merging. + * Used by the incremental-indexing parse cache to capture the per-chunk + * worker output for caching across runs. The mutation happens in-place + * so the caller (parse-impl) can keep a reference. See + * `gitnexus/src/storage/parse-cache.ts`. + */ + outRawResults?: ParseWorkerResult[], ): Promise => { // Filter to parseable files only const parseableFiles: ParseWorkerInput[] = []; @@ -123,63 +213,16 @@ const processParsingWithWorkers = async ( }, ); - // Merge results from all workers into graph and symbol table - const allImports: ExtractedImport[] = []; - const allCalls: ExtractedCall[] = []; - const allAssignments: ExtractedAssignment[] = []; - const allHeritage: ExtractedHeritage[] = []; - const allRoutes: ExtractedRoute[] = []; - const allFetchCalls: ExtractedFetchCall[] = []; - const allDecoratorRoutes: ExtractedDecoratorRoute[] = []; - const allToolDefs: ExtractedToolDef[] = []; - const allORMQueries: ExtractedORMQuery[] = []; - const allConstructorBindings: FileConstructorBindings[] = []; - const fileScopeBindingsByFile: FileScopeBindings[] = []; - const allParsedFiles: ParsedFile[] = []; - for (const result of chunkResults) { - for (const node of result.nodes) { - graph.addNode({ - id: node.id, - label: node.label as NodeLabel, - properties: node.properties, - }); - } - - for (const rel of result.relationships) { - graph.addRelationship(rel); - } - - for (const sym of result.symbols) { - symbolTable.add(sym.filePath, sym.name, sym.nodeId, sym.type, { - parameterCount: sym.parameterCount, - requiredParameterCount: sym.requiredParameterCount, - parameterTypes: sym.parameterTypes, - returnType: sym.returnType, - declaredType: sym.declaredType, - ownerId: sym.ownerId, - qualifiedName: sym.qualifiedName, - }); - } - - for (const item of result.imports) allImports.push(item); - for (const item of result.calls) allCalls.push(item); - for (const item of result.assignments) allAssignments.push(item); - for (const item of result.heritage) allHeritage.push(item); - for (const item of result.routes) allRoutes.push(item); - for (const item of result.fetchCalls) allFetchCalls.push(item); - for (const item of result.decoratorRoutes) allDecoratorRoutes.push(item); - for (const item of result.toolDefs) allToolDefs.push(item); - if (result.ormQueries) for (const item of result.ormQueries) allORMQueries.push(item); - for (const item of result.constructorBindings) allConstructorBindings.push(item); - if (result.fileScopeBindings) - for (const item of result.fileScopeBindings) fileScopeBindingsByFile.push(item); - // RFC #909 Ring 2: aggregate per-file scope artifacts. Tolerant of - // workers that don't emit the field yet (older worker builds or - // partial rollouts), since the additive contract means undefined = - // "this worker produced no ParsedFiles for this chunk". - if (result.parsedFiles) for (const item of result.parsedFiles) allParsedFiles.push(item); + // Capture the raw chunk results for the incremental parse cache before + // merging — the cache stores the unmerged worker output so a future run + // can re-merge them into a fresh graph state. + if (outRawResults) { + for (const r of chunkResults) outRawResults.push(r); } + // Merge results from all workers into graph and symbol table. + const merged = mergeChunkResults(graph, symbolTable, chunkResults); + // Merge and log skipped languages from workers const skippedLanguages = new Map(); for (const result of chunkResults) { @@ -196,20 +239,7 @@ const processParsingWithWorkers = async ( // Final progress onFileProgress?.(total, total, 'done'); - return { - imports: allImports, - calls: allCalls, - assignments: allAssignments, - heritage: allHeritage, - routes: allRoutes, - fetchCalls: allFetchCalls, - decoratorRoutes: allDecoratorRoutes, - toolDefs: allToolDefs, - ormQueries: allORMQueries, - constructorBindings: allConstructorBindings, - fileScopeBindings: fileScopeBindingsByFile, - parsedFiles: allParsedFiles, - }; + return merged; }; // ============================================================================ @@ -732,6 +762,14 @@ export const processParsing = async ( scopeTreeCache: ASTCache | undefined, onFileProgress?: FileProgressCallback, workerPool?: WorkerPool, + /** + * Optional out-parameter for the incremental parse cache. When + * provided AND the worker-pool path runs successfully, populated + * with the raw `ParseWorkerResult[]` from the workers (pre-merge). + * Stays empty for the sequential fallback path (no per-chunk + * artifact to cache there). See `gitnexus/src/storage/parse-cache.ts`. + */ + outRawResults?: ParseWorkerResult[], ): Promise => { let lastProgress = 0; const reportProgress: FileProgressCallback | undefined = onFileProgress @@ -759,6 +797,7 @@ export const processParsing = async ( astCache, workerPool, reportProgress, + outRawResults, ); } catch (err) { const message = err instanceof Error ? err.message : String(err); diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts index bd39a4330..17cfaab3f 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts @@ -17,7 +17,10 @@ import { enrichExportedTypeMap, type BindingEntry, } from '../binding-accumulator.js'; -import { processParsing } from '../parsing-processor.js'; +import { processParsing, mergeChunkResults } from '../parsing-processor.js'; +import { fileContentHash, computeChunkHash } from '../../../storage/parse-cache.js'; +import type { ParseWorkerResult } from '../workers/parse-worker.js'; +import type { WorkerExtractedData } from '../parsing-processor.js'; import { processImports, processImportsFromExtracted, @@ -72,8 +75,21 @@ import { extractORMQueriesInline } from './orm-extraction.js'; import { logger } from '../../logger.js'; // ── Constants ────────────────────────────────────────────────────────────── -/** Max bytes of source content to load per parse chunk. */ -const CHUNK_BYTE_BUDGET = 20 * 1024 * 1024; // 20MB +/** Max bytes of source content to load per parse chunk. + * + * Memory bound for the worker pool dispatch + a granularity knob for + * the parse cache. A single file change invalidates only its enclosing + * chunk, so smaller budgets → finer-grained invalidation. + * + * Override via GITNEXUS_CHUNK_BYTE_BUDGET (bytes) — the default of 2MB + * gives a useful invalidation floor (~1/N chunks on a multi-MB repo) + * while keeping worker dispatch overhead under 5% on cold runs. + */ +const CHUNK_BYTE_BUDGET = (() => { + const env = Number(process.env.GITNEXUS_CHUNK_BYTE_BUDGET); + if (Number.isFinite(env) && env > 0) return env; + return 2 * 1024 * 1024; +})(); // ── Main parse + resolve function ────────────────────────────────────────── @@ -119,6 +135,11 @@ export async function runChunkedParseAndResolve( * source. See plan * docs/plans/2026-04-20-002-perf-parse-heritage-mro-plan.md (Unit 4). */ scopeTreeCache: ASTCache; + /** Worker-produced ParsedFile artifacts aggregated across chunks. + * Threaded into scope-resolution as a re-extract cache so the warm- + * cache analyze run can skip the dominant `extractParsedFile` cost + * (otherwise ~58s on a 1000-file repo). */ + parsedFiles: import('gitnexus-shared').ParsedFile[]; }> { const ctx = createResolutionContext(); const symbolTable = ctx.model.symbols; @@ -142,6 +163,15 @@ export async function runChunkedParseAndResolve( ); } + // Sort parseableScanned alphabetically for stable chunk membership + // across runs (Finding 4). Without this, filesystem-scan order can + // shift between runs (notably on macOS APFS where directory entry + // order can change after modifications) — different files in the + // same chunk → different chunk hash → cache miss even when no file + // content changed. The cache also becomes platform-specific: a + // Linux-built cache misses on macOS for the same repo. + parseableScanned.sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0)); + const totalParseable = parseableScanned.length; if (totalParseable === 0) { @@ -271,6 +301,20 @@ export async function runChunkedParseAndResolve( const deferredWorkerHeritage: ExtractedHeritage[] = []; const deferredConstructorBindings: FileConstructorBindings[] = []; const deferredAssignments: ExtractedAssignment[] = []; + // Aggregated per-file ParsedFile artifacts produced by workers' calls + // to `extractParsedFile`. Threaded through to the scope-resolution + // phase so it can SKIP its own re-extraction on cache hits — this is + // the second-half of the parse-cache speedup since scope-resolution's + // re-parse otherwise dominates the warm-cache wall-clock time. + const allParsedFiles: import('gitnexus-shared').ParsedFile[] = []; + + // Incremental parse cache (Option B): chunk-level content-addressed. + // When the chunk's (filePath, content-hash) signature matches a prior + // run's, replay the cached ParseWorkerResult[] instead of dispatching + // to workers. See gitnexus/src/storage/parse-cache.ts. + const parseCache = options?.parseCache; + let chunkCacheHits = 0; + let chunkCacheMisses = 0; try { for (let chunkIdx = 0; chunkIdx < numChunks; chunkIdx++) { @@ -281,29 +325,89 @@ export async function runChunkedParseAndResolve( .filter((p) => chunkContents.has(p)) .map((p) => ({ path: p, content: chunkContents.get(p)! })); - const chunkWorkerData = await processParsing( - graph, - chunkFiles, - symbolTable, - astCache, - scopeTreeCache, - (current, _total, filePath) => { - const globalCurrent = filesParsedSoFar + current; - const parsingProgress = 20 + (globalCurrent / totalParseable) * 62; - onProgress({ - phase: 'parsing', - percent: Math.round(parsingProgress), - message: `Parsing chunk ${chunkIdx + 1}/${numChunks}...`, - detail: filePath, - stats: { - filesProcessed: globalCurrent, - totalFiles: totalParseable, - nodesCreated: graph.nodeCount, - }, - }); - }, - workerPool, - ); + // Compute the chunk's content-hash signature (if cache available). + let chunkHash: string | null = null; + if (parseCache) { + const entries = chunkFiles.map((f) => ({ + filePath: f.path, + contentHash: fileContentHash(f.content), + })); + chunkHash = computeChunkHash(entries); + } + + let chunkWorkerData: WorkerExtractedData | null; + const cachedRaw = chunkHash ? parseCache!.entries.get(chunkHash) : undefined; + + // Track every chunk hash we touched so the orchestrator can + // prune stale entries (chunks whose composition no longer + // corresponds to a live chunk in the current scan) before saving. + if (parseCache && chunkHash) parseCache.usedKeys.add(chunkHash); + + if (cachedRaw && cachedRaw.length > 0) { + // Cache hit: replay the cached worker output through the same + // merge logic the live worker path uses. + chunkCacheHits++; + chunkWorkerData = mergeChunkResults(graph, symbolTable, cachedRaw); + if (isDev) { + logger.info( + `📦 parse-cache HIT: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash!.slice(0, 8)})`, + ); + } + // Progress update so UI advances even on a cache hit. + const cachedFiles = chunkFiles.length; + onProgress({ + phase: 'parsing', + percent: Math.round(20 + ((filesParsedSoFar + cachedFiles) / totalParseable) * 62), + message: `Parsing chunk ${chunkIdx + 1}/${numChunks} (cache)...`, + stats: { + filesProcessed: filesParsedSoFar + cachedFiles, + totalFiles: totalParseable, + nodesCreated: graph.nodeCount, + }, + }); + } else { + // Cache miss: dispatch to workers, capture the raw results, store + // them under the chunk hash for the next run. + chunkCacheMisses++; + const rawResults: ParseWorkerResult[] = []; + chunkWorkerData = await processParsing( + graph, + chunkFiles, + symbolTable, + astCache, + scopeTreeCache, + (current, _total, filePath) => { + const globalCurrent = filesParsedSoFar + current; + const parsingProgress = 20 + (globalCurrent / totalParseable) * 62; + onProgress({ + phase: 'parsing', + percent: Math.round(parsingProgress), + message: `Parsing chunk ${chunkIdx + 1}/${numChunks}...`, + detail: filePath, + stats: { + filesProcessed: globalCurrent, + totalFiles: totalParseable, + nodesCreated: graph.nodeCount, + }, + }); + }, + workerPool, + // Capture raw results only when we have a cache to write to — + // otherwise we'd retain extra arrays for nothing. + parseCache && chunkHash ? rawResults : undefined, + ); + // Persist the raw results for this chunk hash. Sequential path + // doesn't populate rawResults (it writes directly to graph), so + // small repos without worker pool simply don't cache. That's fine. + if (parseCache && chunkHash && rawResults.length > 0) { + parseCache.entries.set(chunkHash, rawResults); + if (isDev) { + logger.info( + `📦 parse-cache MISS+store: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash.slice(0, 8)})`, + ); + } + } + } const chunkBasePercent = 20 + (filesParsedSoFar / totalParseable) * 62; @@ -349,6 +453,12 @@ export async function runChunkedParseAndResolve( for (const item of chunkWorkerData.heritage) deferredWorkerHeritage.push(item); for (const item of chunkWorkerData.constructorBindings) deferredConstructorBindings.push(item); + // Aggregate worker-produced ParsedFile artifacts so scope- + // resolution can use them as a re-extraction cache (skips its + // own tree-sitter re-parse on warm runs). + if (chunkWorkerData.parsedFiles?.length) { + for (const item of chunkWorkerData.parsedFiles) allParsedFiles.push(item); + } if (chunkWorkerData.assignments?.length) { for (const item of chunkWorkerData.assignments) deferredAssignments.push(item); } @@ -422,6 +532,12 @@ export async function runChunkedParseAndResolve( astCache.clear(); } + if (isDev && parseCache && (chunkCacheHits > 0 || chunkCacheMisses > 0)) { + logger.info( + `📦 parse-cache summary: ${chunkCacheHits} chunk hit(s), ${chunkCacheMisses} miss(es) across ${numChunks} chunk(s)`, + ); + } + const fullWorkerHeritageMap = deferredWorkerHeritage.length > 0 ? buildHeritageMap(deferredWorkerHeritage, ctx, getHeritageStrategyForLanguage) @@ -621,5 +737,12 @@ export async function runChunkedParseAndResolve( // chunk-local `astCache` above is intentionally NOT exposed // because parse-impl clears it between chunks. scopeTreeCache, + // Per-file ParsedFile artifacts produced by workers' calls to + // `extractParsedFile`. Empty when only the sequential path ran + // (sequential doesn't go through the worker, and extracts ParsedFile + // inline rather than emitting it). Consumed by scope-resolution as + // a re-extraction cache: when the file's ParsedFile is here, + // scope-resolution skips its own `extractParsedFile` call. + parsedFiles: allParsedFiles, }; } diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse.ts index a20d1e4b0..a3fa81be7 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse.ts @@ -20,6 +20,7 @@ import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js'; import { getPhaseOutput } from './types.js'; import type { StructureOutput } from './structure.js'; import type { BindingAccumulator } from '../binding-accumulator.js'; +import type { ParsedFile } from 'gitnexus-shared'; import type { ExtractedFetchCall, ExtractedRoute, @@ -81,6 +82,19 @@ export interface ParseOutput { * `scopeTreeCache.clear()` after its extract loop finishes. */ readonly scopeTreeCache: ASTCache; + /** + * Per-file `ParsedFile` artifacts produced by workers' calls to + * `extractParsedFile`. Threaded through to `scopeResolutionPhase` + * as a re-extraction cache: when a file's ParsedFile is present here, + * scope-resolution can skip its own `extractParsedFile` (which would + * otherwise re-parse the file with tree-sitter on the main thread, + * costing ~58s on a 1000-file repo). + * + * Empty for files that went through the sequential parse fallback — + * sequential doesn't emit ParsedFile artifacts; scope-resolution + * falls back to a fresh extract for those. + */ + readonly parsedFiles: readonly ParsedFile[]; } export const parsePhase: PipelinePhase = { diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index c220ea224..1ee8e102f 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -55,6 +55,19 @@ export interface PipelineOptions { minFiles?: number; minBytes?: number; }; + /** + * Incremental-indexing parse cache. When provided: + * - The parse phase looks up each chunk's content hash in + * `parseCache.entries`. On hit, it replays the cached + * `ParseWorkerResult[]` instead of dispatching to workers. + * - On miss, it runs the workers as today and stores the new + * results in `parseCache.entries` keyed by chunk hash. + * The caller (`run-analyze.ts`) is responsible for loading the cache + * before the pipeline runs and persisting it after. Cache survives + * `--force` because keys are content-addressed. + * See `gitnexus/src/storage/parse-cache.ts`. + */ + parseCache?: import('../../storage/parse-cache.js').ParseCache; } // ── Phase registry ───────────────────────────────────────────────────────── diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts index c2fda9777..98a9f8994 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts @@ -93,13 +93,25 @@ export const scopeResolutionPhase: PipelinePhase = { // Worker-mode parses leave the cache empty for those files; they // also fall back to a fresh parse — no correctness impact. const parseOutput = getPhaseOutput(deps, 'parse'); - const { scopeTreeCache, resolutionContext } = parseOutput; + const { scopeTreeCache, resolutionContext, parsedFiles: workerParsedFiles } = parseOutput; // SemanticModel populated during `parse`: scope-resolution consumes // TypeRegistry / MethodRegistry / SymbolTable lookups instead of // rebuilding parallel indexes. See ARCHITECTURE.md § "Semantic-model // source of truth". const model = resolutionContext.model; + // Build a per-file lookup of ParsedFile artifacts the workers (or + // sequential extracts) already produced. Threading this into + // `runScopeResolution` lets the per-language extract loop short- + // circuit `extractParsedFile` — the dominant cost on the warm-cache + // path, since workers can't return tree-sitter Trees across the + // MessageChannel and scope-resolution would otherwise re-parse + // every file from scratch on the main thread. + const preExtractedByPath = new Map(); + for (const pf of workerParsedFiles) { + preExtractedByPath.set(pf.filePath, pf); + } + let totalFiles = 0; let totalImports = 0; let totalRefs = 0; @@ -143,6 +155,7 @@ export const scopeResolutionPhase: PipelinePhase = { files, treeCache: scopeTreeCache, resolutionConfig, + preExtractedParsedFiles: preExtractedByPath, onWarn: (msg) => { if (isSemanticModelValidatorEnabled()) { logger.warn(`[scope-resolution:${lang}] ${msg}`); diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/registry.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/registry.ts index 0667be8f5..ecd7b69aa 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/registry.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/registry.ts @@ -15,6 +15,7 @@ import { pythonScopeResolver } from '../../languages/python/scope-resolver.js'; import { csharpScopeResolver } from '../../languages/csharp/scope-resolver.js'; import { typescriptScopeResolver } from '../../languages/typescript/scope-resolver.js'; import { goScopeResolver } from '../../languages/go/scope-resolver.js'; +import { javaScopeResolver } from '../../languages/java/scope-resolver.js'; import { cScopeResolver } from '../../languages/c/scope-resolver.js'; /** Map of `SupportedLanguages` → `ScopeResolver`. The phase iterates @@ -29,5 +30,6 @@ export const SCOPE_RESOLVERS: ReadonlyMap = n [SupportedLanguages.CSharp, csharpScopeResolver], [SupportedLanguages.TypeScript, typescriptScopeResolver], [SupportedLanguages.Go, goScopeResolver], + [SupportedLanguages.Java, javaScopeResolver], [SupportedLanguages.C, cScopeResolver], ]); diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts index e2c734a43..31a58cdfa 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts @@ -72,6 +72,22 @@ interface RunScopeResolutionInput { * provider doesn't supply a config loader. */ readonly resolutionConfig?: unknown; + /** + * Pre-extracted ParsedFile artifacts keyed by file path. When a + * file is present here, the extract loop reuses it directly and + * skips `extractParsedFile` (which would re-parse the file with + * tree-sitter on the main thread). Only files matching the + * provider's language are honored — the loop verifies this + * implicitly by language filter at the call-site (scopeResolution + * phase). + * + * Worker-mode parses produce these ParsedFile artifacts as a side + * effect of `extractParsedFile` running inside the worker; threading + * them here is what lets the warm-cache analyze run skip the ~58s + * scope-resolution re-parse loop on a multi-thousand-file repo. + * Cache miss is safe — falls back to fresh extract. + */ + readonly preExtractedParsedFiles?: ReadonlyMap; } interface RunScopeResolutionStats { @@ -104,22 +120,37 @@ export function runScopeResolution( const parsedFiles: ParsedFile[] = []; let filesSkipped = 0; const treeCache = input.treeCache; + const preExtracted = input.preExtractedParsedFiles; + let preExtractedHits = 0; for (const file of files) { - const cachedTree = treeCache?.get(file.path); - const parsed = extractParsedFile( - provider.languageProvider, - file.content, - file.path, - onWarn, - cachedTree, - ); + let parsed: ParsedFile | undefined; + // Fast path: a worker (during the parse phase) already produced a + // ParsedFile for this file via `extractParsedFile`. Reuse it + // directly — skips a tree-sitter re-parse on the main thread. + if (preExtracted !== undefined) { + parsed = preExtracted.get(file.path); + if (parsed !== undefined) preExtractedHits++; + } if (parsed === undefined) { - filesSkipped++; - continue; + const cachedTree = treeCache?.get(file.path); + parsed = extractParsedFile( + provider.languageProvider, + file.content, + file.path, + onWarn, + cachedTree, + ); + if (parsed === undefined) { + filesSkipped++; + continue; + } } provider.populateOwners(parsed); parsedFiles.push(parsed); } + if (PROF && preExtracted !== undefined) { + logger.warn(`[scope-resolution prof] pre-extracted hits: ${preExtractedHits}/${files.length}`); + } provider.populateWorkspaceOwners?.(parsedFiles, { fileContents: getFileContents() }); // Reconcile scope-resolution's ownership view into the SemanticModel. diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index e7043b7f2..cf8f1cb71 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -1264,6 +1264,77 @@ export const deleteNodesForFile = async ( export const getEmbeddingTableName = (): string => EMBEDDING_TABLE_NAME; +/** + * Return the distinct repo-relative paths of files that import + * `targetFilePath` according to the IMPORTS edges currently in the + * DB. Used by the incremental writeback path to expand the + * "files-to-rewrite" set so that files importing a changed file get + * their edges (which may have been refined by cross-file resolution) + * re-emitted, rather than left stale in the DB. + * + * The DB query reads the *previous* run's state — pre-pipeline, before + * any nodes are deleted — so the returned importers are "files that + * USED TO import the target". That's the right set to invalidate: + * those are the files whose edges in the DB might no longer match + * what cross-file resolution produces given the changed file's new + * exports. + */ +export const queryImporters = async (targetFilePath: string): Promise => { + if (!conn) { + throw new Error('LadybugDB not initialized. Call initLbug first.'); + } + const escaped = targetFilePath.replace(/'/g, "''"); + const cypher = ` + MATCH (a)-[r:${REL_TABLE_NAME}]->(b) + WHERE r.type = 'IMPORTS' AND b.filePath = '${escaped}' + RETURN DISTINCT a.filePath AS importer + `; + try { + const queryResult = await conn.query(cypher); + const result = Array.isArray(queryResult) ? queryResult[0] : queryResult; + const rows = await result.getAll(); + const out: string[] = []; + for (const row of rows) { + const v = (row as { importer?: unknown }).importer; + if (typeof v === 'string' && v.length > 0) out.push(v); + } + return out; + } catch { + return []; + } +}; + +/** + * Drop every Community and Process node (and their MEMBER_OF / + * STEP_IN_PROCESS edges via DETACH DELETE). Used at the start of an + * incremental run so the communities and processes phases regenerate + * them from scratch on the merged graph — required for the + * "Leiden runs on the FULL graph" correctness invariant. + */ +export const deleteAllCommunitiesAndProcesses = async (): Promise<{ + nodesDeleted: number; +}> => { + if (!conn) { + throw new Error('LadybugDB not initialized. Call initLbug first.'); + } + let nodesDeleted = 0; + for (const label of ['Community', 'Process']) { + try { + const countResult = await conn.query(`MATCH (n:${label}) RETURN count(n) AS cnt`); + const result = Array.isArray(countResult) ? countResult[0] : countResult; + const rows = await result.getAll(); + const count = Number(rows[0]?.cnt ?? rows[0]?.[0] ?? 0); + if (count > 0) { + await conn.query(`MATCH (n:${label}) DETACH DELETE n`); + nodesDeleted += count; + } + } catch { + // Table may not exist yet on a freshly-initialized DB — fine. + } + } + return { nodesDeleted }; +}; + // ============================================================================ // Full-Text Search (FTS) Functions // ============================================================================ diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index fa2757f45..425f18f9a 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -11,6 +11,7 @@ import path from 'path'; import fs from 'fs/promises'; +import { execFileSync } from 'child_process'; import { runPipelineFromRepo } from './ingestion/pipeline.js'; import { initLbug, @@ -20,6 +21,9 @@ import { executeWithReusedStatement, closeLbug, loadCachedEmbeddings, + deleteNodesForFile, + deleteAllCommunitiesAndProcesses, + queryImporters, } from './lbug/lbug-adapter.js'; import { createSearchFTSIndexes } from './search/fts-indexes.js'; import { @@ -29,7 +33,15 @@ import { ensureGitNexusIgnored, registerRepo, cleanupOldKuzuFiles, + INCREMENTAL_SCHEMA_VERSION, } from '../storage/repo-manager.js'; +import { computeFileHashes, diffFileHashes } from '../storage/file-hash.js'; +import { + extractChangedSubgraph, + computeEffectiveWriteSet, +} from './incremental/subgraph-extract.js'; +import { shadowCandidatesFor } from './incremental/shadow-candidates.js'; +import { loadParseCache, saveParseCache, pruneCache } from '../storage/parse-cache.js'; import { getCurrentCommit, getRemoteUrl, @@ -178,23 +190,81 @@ export async function runFullAnalysis( const currentCommit = repoHasGit ? getCurrentCommit(repoPath) : ''; const existingMeta = await loadMeta(storagePath); + // ── Crash recovery: dirty flag forces full rebuild ──────────────── + // If the previous incremental run set incrementalInProgress and didn't + // clear it, the on-disk index may be in a half-state. Cheapest path + // back to a known-good index is to wipe + rebuild from scratch. + if (existingMeta?.incrementalInProgress) { + log( + 'Previous incremental run did not complete cleanly (incrementalInProgress flag set); ' + + 'forcing full rebuild to restore a known-good index.', + ); + options = { ...options, force: true }; + // Reload meta after clearing the flag in-memory; we still want fileHashes + // for the post-rebuild meta carry-over, but force=true ensures the + // rebuild path executes. + } + // ── Early-return: already up to date ────────────────────────────── if (existingMeta && !options.force && existingMeta.lastCommit === currentCommit) { // Non-git folders have currentCommit = '' — always rebuild since we can't detect changes if (currentCommit !== '') { - await ensureGitNexusIgnored(repoPath); - return { - // `resolveRepoIdentityRoot` collapses worktree roots to the - // canonical repo basename (#1259) but leaves arbitrary subdirs - // and `--skip-git` paths unchanged (#1232/#1233 intent preserved). - repoName: - options.registryName ?? - getInferredRepoName(repoPath) ?? - path.basename(resolveRepoIdentityRoot(repoPath)), - repoPath, - stats: existingMeta.stats ?? {}, - alreadyUpToDate: true, - }; + // For git repos, even if HEAD matches lastCommit, the working tree + // may have uncommitted changes. Only short-circuit when the working + // tree is also clean — otherwise fall through to the incremental + // path which will hash-diff and update only changed files. + // + // We exclude paths that GitNexus itself writes during analyze: + // .gitnexus/ — db / parse cache / meta.json + // .claude/, .cursor/ — auto-generated agent skill files + // AGENTS.md, CLAUDE.md — auto-updated stats blocks + // Counting them as dirty would perpetually defeat the up-to-date + // fast path because the previous analyze just wrote them + // (regression vs PR #1233 behavior). + const dirty = (() => { + try { + const out = execFileSync( + 'git', + [ + 'status', + '--porcelain', + '--', + '.', + ':(exclude).gitnexus', + ':(exclude).gitnexus/**', + ':(exclude).claude', + ':(exclude).claude/**', + ':(exclude).cursor', + ':(exclude).cursor/**', + ':(exclude)AGENTS.md', + ':(exclude)CLAUDE.md', + ], + { + cwd: repoPath, + stdio: ['ignore', 'pipe', 'ignore'], + encoding: 'utf8', + }, + ); + return out.trim().length > 0; + } catch { + return true; // conservative on git failure + } + })(); + if (!dirty) { + await ensureGitNexusIgnored(repoPath); + return { + // `resolveRepoIdentityRoot` collapses worktree roots to the + // canonical repo basename (#1259) but leaves arbitrary subdirs + // and `--skip-git` paths unchanged (#1232/#1233 intent preserved). + repoName: + options.registryName ?? + getInferredRepoName(repoPath) ?? + path.basename(resolveRepoIdentityRoot(repoPath)), + repoPath, + stats: existingMeta.stats ?? {}, + alreadyUpToDate: true, + }; + } } } @@ -243,6 +313,14 @@ export async function runFullAnalysis( ); } + // We *always* load the embedding cache when one is requested (regardless + // of the predicted `willTryIncremental`). The post-pipeline branch may + // disagree with the prediction (e.g. when the pipeline produces zero + // File nodes, `isIncremental` flips false and the full-rebuild path + // wipes the DB) — loading unconditionally is cheap insurance against + // silently dropping embeddings on a mispredicted run. The re-insert + // step gates itself on the actual `isIncremental` value to avoid + // PK-conflicts when the incremental writeback path keeps the rows. if (shouldLoadCache && existingMeta) { try { progress('embeddings', 0, 'Caching embeddings...'); @@ -270,24 +348,89 @@ export async function runFullAnalysis( } } + // ── Load incremental parse cache ────────────────────────────────── + // Content-addressed: safe to reuse across `--force` runs (chunks whose + // file contents haven't changed produce identical worker output). + // Loaded into a single ParseCache object that the pipeline mutates + // in-place (cache hits leave entries unchanged; misses add new ones). + const parseCache = await loadParseCache(storagePath); + // ── Phase 1: Full Pipeline (0–60%) ──────────────────────────────── - const pipelineResult = await runPipelineFromRepo(repoPath, (p) => { - const phaseLabel = PHASE_LABELS[p.phase] || p.phase; - const scaled = Math.round(p.percent * 0.6); - const message = p.detail ? `${p.message || phaseLabel} (${p.detail})` : p.message || phaseLabel; - progress(p.phase, scaled, message); - }); + const pipelineResult = await runPipelineFromRepo( + repoPath, + (p) => { + const phaseLabel = PHASE_LABELS[p.phase] || p.phase; + const scaled = Math.round(p.percent * 0.6); + const message = p.detail + ? `${p.message || phaseLabel} (${p.detail})` + : p.message || phaseLabel; + progress(p.phase, scaled, message); + }, + { parseCache }, + ); // ── Phase 2: LadybugDB (60–85%) ────────────────────────────────── progress('lbug', 60, 'Loading into LadybugDB...'); - await closeLbug(); - const lbugFiles = [lbugPath, `${lbugPath}.wal`, `${lbugPath}.lock`]; - for (const f of lbugFiles) { - try { - await fs.rm(f, { recursive: true, force: true }); - } catch { - /* swallow */ + // Compute current per-file content hashes from the pipeline's File nodes. + // Used both to drive the incremental DB writeback (when eligible) and to + // populate meta.json.fileHashes for the next run. + const allFilePaths: string[] = []; + pipelineResult.graph.forEachNode((n) => { + if (n.label === 'File') { + const fp = n.properties?.filePath as string | undefined; + if (fp) allFilePaths.push(fp); + } + }); + const newFileHashes = await computeFileHashes(repoPath, allFilePaths); + + // Decide incremental vs full at THIS point (post-pipeline, pre-DB). + // All eligibility conditions are checked here against the actual + // pipeline output — no separate pre-pipeline prediction to desync from + // (Bugbot review on PR #1479: a prediction that flipped post-pipeline + // could skip the embedding cache load and then take the full-rebuild + // path, silently losing embeddings). + const isIncremental = + !options.force && + !!existingMeta && + existingMeta.schemaVersion === INCREMENTAL_SCHEMA_VERSION && + !!existingMeta.fileHashes && + Object.keys(existingMeta.fileHashes).length > 0 && + repoHasGit && + allFilePaths.length > 0; + + const hashDiff = isIncremental + ? diffFileHashes(newFileHashes, existingMeta!.fileHashes) + : undefined; + + if (isIncremental && hashDiff) { + log( + `Incremental: changed=${hashDiff.changed.length}, ` + + `added=${hashDiff.added.length}, ` + + `deleted=${hashDiff.deleted.length} ` + + `(skipping wipe + ${ + allFilePaths.length - hashDiff.toWrite.length + } unchanged file rows preserved)`, + ); + // Set the dirty flag BEFORE any destructive DB mutation. Cleared on + // success at the meta-save step. + await saveMeta(storagePath, { + ...existingMeta!, + incrementalInProgress: { + startedAt: Date.now(), + toWriteCount: hashDiff.toWrite.length, + }, + }); + } else { + // Full rebuild path: wipe DB files first. + await closeLbug(); + const lbugFiles = [lbugPath, `${lbugPath}.wal`, `${lbugPath}.lock`]; + for (const f of lbugFiles) { + try { + await fs.rm(f, { recursive: true, force: true }); + } catch { + /* swallow */ + } } } @@ -298,11 +441,145 @@ export async function runFullAnalysis( // must be released to avoid blocking subsequent invocations. let lbugMsgCount = 0; - await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => { - lbugMsgCount++; - const pct = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24)); - progress('lbug', pct, msg); - }); + if (isIncremental && hashDiff) { + // ── Incremental DB writeback ─────────────────────────────────── + // 0. Expand the writable set with transitive importers of + // changed/deleted files (bounded BFS). + // + // Reason (Bugbot/Claude review on PR #1479): when a barrel / + // re-export file C changes, cross-file resolution may update + // CALLS edges between two unchanged files A and B (A imports + // from C, C re-exports something from B). Those refined edges + // live in `ctx.graph` but would be excluded from the subgraph + // if neither endpoint is in the changed set. To catch this, + // files that imported (directly OR transitively, through + // other unchanged intermediaries) any changed file get pulled + // into the writable set so their rows are deleted + rewritten + // against the refined edges. + // + // BFS bound: MAX_IMPORTER_BFS_DEPTH. Practically sized to + // catch nested barrel chains (e.g. `index.ts → submodule/index.ts + // → submodule/impl.ts`) without ballooning into a near-full- + // rebuild on monorepos with deep re-export pyramids. Beyond + // this depth, the "incremental ≡ full-rebuild" invariant is + // self-acknowledged as best-effort; `--force` remains the + // escape hatch documented in GUARDRAILS.md. + // + // `queryImporters` reads `IMPORTS` from the pre-pipeline DB + // state, so the result is "files that USED TO import the + // target" — exactly the set whose previously-stored edges may + // no longer match what cross-file resolution produces this run. + const MAX_IMPORTER_BFS_DEPTH = 4; + const writableFiles = new Set(hashDiff.toWrite); + const directlyChangedCount = writableFiles.size; + + // Shadow-seed: for ADDED files, queryImporters returns 0 (the new + // file has no IMPORTS rows in the pre-pipeline DB yet). But pre- + // existing unchanged files may have IMPORTS edges whose module- + // resolution claim the newcomer can steal under standard JS/TS + // resolution (Bugbot review on PR #1479). For each added file we + // derive the shadow candidates and, if the candidate was a known + // file in the prior meta, seed it into the BFS frontier so its + // importers — surfaced via queryImporters — get their CALLS edges + // re-resolved against the new file. See shadow-candidates.ts for + // the full pattern catalogue. + const priorFileSet = new Set( + existingMeta?.fileHashes ? Object.keys(existingMeta.fileHashes) : [], + ); + const shadowSeed: string[] = []; + for (const added of hashDiff.added) { + for (const cand of shadowCandidatesFor(added)) { + if (priorFileSet.has(cand) && !writableFiles.has(cand)) { + shadowSeed.push(cand); + } + } + } + + { + let frontier: string[] = [...hashDiff.toWrite, ...hashDiff.deleted, ...shadowSeed]; + for (let depth = 0; depth < MAX_IMPORTER_BFS_DEPTH && frontier.length > 0; depth++) { + const nextFrontier: string[] = []; + for (const f of frontier) { + try { + const importers = await queryImporters(f); + for (const i of importers) { + if (!writableFiles.has(i)) { + writableFiles.add(i); + nextFrontier.push(i); + } + } + } catch { + /* per-file importer query failure → skip; correctness degrades on + that branch, but DB stays writable. */ + } + } + frontier = nextFrontier; + } + } + const importerExpansion = writableFiles.size - directlyChangedCount; + if (importerExpansion > 0) { + log( + `Incremental: +${importerExpansion} importer(s) added to writable set ` + + `(BFS depth ≤ ${MAX_IMPORTER_BFS_DEPTH}` + + (shadowSeed.length > 0 ? `, ${shadowSeed.length} shadow-seed(s)` : '') + + `)`, + ); + } + + // 1. Compute the EFFECTIVE write-set (Finding 1). Two layers, + // composed: + // (a) `writableFiles` — toWrite ∪ transitive importers of + // changed/deleted files (the bounded BFS above, reading + // IMPORTS from the pre-pipeline DB). + // (b) `computeEffectiveWriteSet` — walks the NEW graph's + // edges and pulls in any unchanged-side file that sits + // on a writable-boundary-crossing edge (catches refined + // cross-file CALLS edges that the pre-run DB couldn't + // predict, e.g. a barrel re-export shifting `foo` from + // B to D). + // The composed set is the input to BOTH deleteNodesForFile + // and extractChangedSubgraph — asymmetry between the two would + // leave stale rows or PK-conflict at COPY time. + const effectiveWriteSet = computeEffectiveWriteSet(pipelineResult.graph, writableFiles); + // Deduped: deleted entries may already appear via importer-BFS + // expansion (queryImporters can return a now-deleted path), which + // would otherwise call deleteNodesForFile twice for the same file + // (Bugbot LOW finding on PR #1479). + const filesToDelete = [...new Set([...effectiveWriteSet, ...hashDiff.deleted])]; + for (let i = 0; i < filesToDelete.length; i++) { + const f = filesToDelete[i]; + try { + await deleteNodesForFile(f); + } catch { + /* file may not have rows (e.g. an unparseable file) — fine */ + } + if (i % 20 === 0) { + progress('lbug', 62, `Removing rows for changed files (${i}/${filesToDelete.length})...`); + } + } + // 2. Drop graph-wide nodes (Community, Process). They'll be re-inserted + // from the fresh pipeline output below. Required for the + // "Leiden runs on the FULL graph" correctness invariant. + await deleteAllCommunitiesAndProcesses(); + + // 3. Extract the changed subgraph from the FULL ctx.graph and write + // only that. Unchanged-file rows in the DB stay untouched. Pass + // the SAME effectiveWriteSet so the subgraph and the deletes + // cover identical files (asymmetry would silently corrupt). + const subgraph = extractChangedSubgraph(pipelineResult.graph, effectiveWriteSet); + await loadGraphToLbug(subgraph, pipelineResult.repoPath, storagePath, (msg) => { + lbugMsgCount++; + const pct = Math.min(84, 65 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 19)); + progress('lbug', pct, msg); + }); + } else { + // ── Full rebuild ─────────────────────────────────────────────── + await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => { + lbugMsgCount++; + const pct = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24)); + progress('lbug', pct, msg); + }); + } // ── Phase 3: FTS (85–90%) ───────────────────────────────────────── progress('fts', 85, 'Creating search indexes...'); @@ -310,6 +587,19 @@ export async function runFullAnalysis( progress('fts', 90, 'Search indexes ready'); // ── Phase 3.5: Re-insert cached embeddings ──────────────────────── + // Runs on BOTH the full-rebuild path and the incremental path: + // - Full rebuild: DB was wiped, every cached row needs to come back. + // - Incremental: changed-file rows were just deleted by + // deleteNodesForFile (which cascades to their + // embedding rows) — so their cached vectors need + // to come back too. Unchanged-file rows still + // exist; re-inserting their cached vectors would + // PK-conflict, but the per-batch try/catch below + // silently ignores those (matches the existing + // "some may fail if node was removed, that's + // fine" semantics). Bugbot review on PR #1479 + // flagged that gating this on `!isIncremental` + // silently lost changed-file embeddings. if (cachedEmbeddings.length > 0) { const cachedDims = cachedEmbeddings[0].embedding.length; const { EMBEDDING_DIMS } = await import('./lbug/schema.js'); @@ -456,6 +746,12 @@ export async function runFullAnalysis( const effectiveSemanticMode = semanticMode ?? (runtimeCapabilities.semanticMode === 'vector-index' ? 'vector-index' : 'exact-scan'); + + // Convert the post-run file-hash map to the on-disk Record + // shape consumed by RepoMeta.fileHashes. + const newFileHashesRecord: Record = {}; + for (const [k, v] of newFileHashes) newFileHashesRecord[k] = v; + const meta = { repoPath, lastCommit: currentCommit, @@ -485,8 +781,33 @@ export async function runFullAnalysis( reason: runtimeCapabilities.reason, }, }, + // Incremental-indexing fields. Populated for git repos so the next + // analyze run can take the incremental DB-writeback path. Setting + // incrementalInProgress to undefined explicitly clears any prior + // dirty flag (full and incremental success paths converge here). + schemaVersion: hasGitDir(repoPath) ? INCREMENTAL_SCHEMA_VERSION : undefined, + fileHashes: hasGitDir(repoPath) ? newFileHashesRecord : undefined, + incrementalInProgress: undefined as { startedAt: number; toWriteCount: number } | undefined, }; await saveMeta(storagePath, meta); + + // Persist the incremental parse cache for the next run. Wraps in + // try/catch so a cache-write failure never breaks an otherwise + // successful indexing run. Prune stale chunk-hash entries first so + // the cache file size stays bounded across runs (chunks whose + // composition no longer matches anything in the current scan are + // dead weight; the parse phase populates `usedKeys` as it processes + // chunks). + try { + const pruned = pruneCache(parseCache, parseCache.usedKeys); + if (pruned > 0) { + log(`Parse cache: pruned ${pruned} stale chunk entries`); + } + await saveParseCache(storagePath, parseCache); + } catch (e) { + log(`Warning: could not save parse cache (${(e as Error).message}); continuing.`); + } + // Forward the --name alias and the registry-collision bypass bit. // `allowDuplicateName` is its own concern — independent from the // pipeline `force` above. The CLI maps it from diff --git a/gitnexus/src/storage/file-hash.ts b/gitnexus/src/storage/file-hash.ts new file mode 100644 index 000000000..b39111815 --- /dev/null +++ b/gitnexus/src/storage/file-hash.ts @@ -0,0 +1,104 @@ +/** + * Per-file content hashing for incremental DB writeback. + * + * On every analyze run we compute SHA-256 of every file's content and + * store the map in meta.json. The next run compares disk against the + * stored map and produces: + * - `changed` — content differs (re-emit DB rows for this file) + * - `added` — file is new on disk (insert DB rows) + * - `deleted` — file was in last meta but no longer on disk (drop rows) + * + * The pipeline still parses every file (correctness invariant: cross-file + * resolution needs full data). What this enables is a SELECTIVE DB + * writeback: instead of wipe-and-reload of the whole graph (~50s of CSV + * COPY on a 25K-node repo), we only delete-and-rewrite rows for the + * changed/added/deleted set. + * + * See docs/superpowers/specs/2026-05-10-incremental-indexing-design.md + * (Option B revision). + */ + +import { createHash } from 'crypto'; +import fs from 'fs/promises'; +import path from 'path'; + +/** + * Compute SHA-256 of a single file. Returns null when the file can't be + * read — caller treats that as "no signature, assume changed". + */ +export const computeFileHash = async (absPath: string): Promise => { + try { + const buf = await fs.readFile(absPath); + return createHash('sha256').update(buf).digest('hex'); + } catch { + return null; + } +}; + +/** + * Compute SHA-256 hashes for many files in parallel batches. Files that + * fail to read are omitted from the result map. + */ +export const computeFileHashes = async ( + repoPath: string, + relPaths: readonly string[], +): Promise> => { + const out = new Map(); + const BATCH = 100; + for (let i = 0; i < relPaths.length; i += BATCH) { + const batch = relPaths.slice(i, i + BATCH); + const results = await Promise.all( + batch.map(async (rel) => { + const h = await computeFileHash(path.join(repoPath, rel)); + return h ? ([rel, h] as const) : null; + }), + ); + for (const r of results) if (r) out.set(r[0], r[1]); + } + return out; +}; + +/** Result of comparing the current on-disk hashes against stored ones. */ +export interface FileHashDiff { + /** Files whose content hash differs from stored. */ + changed: string[]; + /** Files in the current scan that weren't in the stored map. */ + added: string[]; + /** Files in the stored map that aren't in the current scan. */ + deleted: string[]; + /** All files whose DB rows must be replaced (changed ∪ added). */ + toWrite: string[]; +} + +/** + * Diff a current hash map against a previously stored one. + * + * Sorted output so two runs produce identical diff arrays for the same + * changes — useful for stable logging / equivalence checks. + */ +export const diffFileHashes = ( + current: ReadonlyMap, + stored: Readonly> | undefined, +): FileHashDiff => { + const storedMap = new Map(stored ? Object.entries(stored) : []); + const changed: string[] = []; + const added: string[] = []; + for (const [p, h] of current) { + const prev = storedMap.get(p); + if (prev === undefined) added.push(p); + else if (prev !== h) changed.push(p); + } + const deleted: string[] = []; + for (const p of storedMap.keys()) { + if (!current.has(p)) deleted.push(p); + } + changed.sort(); + added.sort(); + deleted.sort(); + return { + changed, + added, + deleted, + toWrite: [...changed, ...added].sort(), + }; +}; diff --git a/gitnexus/src/storage/parse-cache.ts b/gitnexus/src/storage/parse-cache.ts new file mode 100644 index 000000000..a1abf76fa --- /dev/null +++ b/gitnexus/src/storage/parse-cache.ts @@ -0,0 +1,213 @@ +/** + * Chunk-level content-addressed parse cache. + * + * The pipeline always parses every file (correctness invariant: cross-file + * resolution and downstream phases need full graph data). What this cache + * does is skip the tree-sitter worker dispatch when a chunk's contents + * haven't changed since the last run. + * + * Granularity: chunk-level. The parse phase chunks files into ~20MB byte + * budgets. The cache key is `sha256(joined(filePath:contentHash for each + * file in the chunk, sorted))`. A change to a single file invalidates only + * that file's chunk — typically 1 of ~50 chunks on a 1000-file repo. + * + * Why not per-file: + * - Workers process sub-batches and emit aggregated `ParseWorkerResult`s. + * Splitting back to per-file would require reworking the worker contract. + * - Chunk-level invalidation gives a useful speedup floor (98% on a single + * 1-of-50 invalidated chunk) without touching the worker. + * + * Survives `--force` because it's content-addressed: the same bytes always + * produce the same key. `--force` only matters for the LadybugDB writeback; + * the cache itself is always safe to reuse. + */ + +import { createHash } from 'crypto'; +import { createRequire } from 'module'; +import fs from 'fs/promises'; +import path from 'path'; +import { fileURLToPath } from 'url'; +import type { ParseWorkerResult } from '../core/ingestion/workers/parse-worker.js'; + +/** + * Cache version composed of: + * - A schema bump knob (`SCHEMA_BUMP`) for hand-controlled invalidation + * when ParseWorkerResult shape or upstream parse semantics change. + * - The current `gitnexus` npm package version, read at module load. + * Any release that ships an updated tree-sitter grammar or revised + * extractor logic implies a version bump in package.json, which + * automatically invalidates the on-disk cache. Without this, a user + * running `npm i -g gitnexus@latest` after a parser-affecting + * release would silently replay pre-upgrade ParseWorkerResults + * against the new graph schema (Bugbot/Claude review on #1479). + * + * On version mismatch, `loadParseCache` returns an empty cache and the + * next save overwrites the on-disk file with the new version baked in. + */ +const SCHEMA_BUMP = 1; +const GITNEXUS_PKG_VERSION = (() => { + try { + // package.json sits at gitnexus/package.json — two levels up from + // gitnexus/src/storage/parse-cache.ts (or its dist/ equivalent). + const here = path.dirname(fileURLToPath(import.meta.url)); + const candidates = [ + path.join(here, '..', '..', 'package.json'), // src/storage → gitnexus/ + path.join(here, '..', '..', '..', 'package.json'), // dist/storage → gitnexus/ + ]; + const requireCJS = createRequire(import.meta.url); + for (const c of candidates) { + try { + const pkg = requireCJS(c); + if (typeof pkg?.version === 'string') return pkg.version; + } catch { + /* try next candidate */ + } + } + } catch { + /* fall through to fallback */ + } + return '0.0.0-unknown'; +})(); +export const PARSE_CACHE_VERSION = `${SCHEMA_BUMP}+${GITNEXUS_PKG_VERSION}`; + +const CACHE_FILENAME = 'parse-cache.json'; + +/** On-disk shape. */ +interface ParseCacheFile { + version: string; + /** key = chunk hash (hex) → cached chunk result list. */ + entries: Record; +} + +/** Runtime view: keyed Map for fast lookup; mutated in place during a run. */ +export interface ParseCache { + version: string; + entries: Map; + /** + * Hashes referenced (hit OR miss-and-stored) by the current run. + * The parse phase populates this as it processes chunks; the orchestrator + * uses it as input to `pruneCache` before saving so entries that no + * longer correspond to any chunk in the current scan are discarded. + * Transient — never serialized to disk. + */ + usedKeys: Set; +} + +/** SHA-256 hex of a single string or buffer. */ +const sha256Hex = (input: Buffer | string): string => + createHash('sha256') + .update(typeof input === 'string' ? Buffer.from(input) : input) + .digest('hex'); + +/** Stable hash of a single file's contents — used by callers to compose a chunk hash. */ +export const fileContentHash = (content: Buffer | string): string => sha256Hex(content); + +/** + * Compute the canonical cache key for a chunk's contents. + * + * `entries` is the list of (filePath, file content hash) for every file + * in the chunk. We sort by filePath before hashing so chunks composed of + * the same files in different order produce the same key. + */ +export const computeChunkHash = ( + entries: Array<{ filePath: string; contentHash: string }>, +): string => { + const sorted = [...entries].sort((a, b) => (a.filePath < b.filePath ? -1 : 1)); + const joined = sorted.map((e) => `${e.filePath}:${e.contentHash}`).join('\n'); + return sha256Hex(joined); +}; + +/** + * JSON replacer that round-trips Map/Set instances through plain JSON. + * + * `ParseWorkerResult.parsedFiles[*].scopes[*].typeBindings` is a + * `ReadonlyMap`; without this transform it serializes + * to `{}` and downstream code that iterates / `.get()`s on it crashes + * with "is not iterable". Applied symmetrically by `mapReviver` on + * load so the in-memory shape stays Map-typed. + */ +const MAP_TAG = '__$mapEntries$__'; +const SET_TAG = '__$setValues$__'; + +const mapReplacer = (_key: string, value: unknown): unknown => { + if (value instanceof Map) return { [MAP_TAG]: Array.from(value.entries()) }; + if (value instanceof Set) return { [SET_TAG]: Array.from(value.values()) }; + return value; +}; + +const mapReviver = (_key: string, value: unknown): unknown => { + if (value && typeof value === 'object') { + const v = value as Record; + if (Array.isArray(v[MAP_TAG])) return new Map(v[MAP_TAG] as [unknown, unknown][]); + if (Array.isArray(v[SET_TAG])) return new Set(v[SET_TAG] as unknown[]); + } + return value; +}; + +/** + * Load the parse cache. Returns an empty cache on any failure (missing + * file, corrupt JSON, version mismatch). Never throws on a normal load. + */ +export const loadParseCache = async (storagePath: string): Promise => { + const cachePath = path.join(storagePath, CACHE_FILENAME); + try { + const raw = await fs.readFile(cachePath, 'utf-8'); + const data = JSON.parse(raw, mapReviver) as ParseCacheFile; + if ( + typeof data !== 'object' || + data === null || + data.version !== PARSE_CACHE_VERSION || + typeof data.entries !== 'object' || + data.entries === null + ) { + return emptyCache(); + } + const entries = new Map(); + for (const [k, v] of Object.entries(data.entries)) { + if (Array.isArray(v)) entries.set(k, v as ParseWorkerResult[]); + } + return { version: PARSE_CACHE_VERSION, entries, usedKeys: new Set() }; + } catch { + return emptyCache(); + } +}; + +/** + * Persist the cache to disk atomically (write-and-rename) so a crash + * mid-write doesn't leave a corrupt file. + */ +export const saveParseCache = async (storagePath: string, cache: ParseCache): Promise => { + await fs.mkdir(storagePath, { recursive: true }); + const cachePath = path.join(storagePath, CACHE_FILENAME); + const tmpPath = `${cachePath}.tmp`; + const out: ParseCacheFile = { + version: cache.version, + entries: Object.fromEntries(cache.entries), + }; + // Compact JSON; this file can be tens of MB on a large repo and pretty- + // printing roughly doubles size for no value. + await fs.writeFile(tmpPath, JSON.stringify(out, mapReplacer), 'utf-8'); + await fs.rename(tmpPath, cachePath); +}; + +/** + * Drop entries whose hashes are not in `usedHashes`. Called at the end + * of a run so chunks that no longer correspond to any current chunk + * don't keep their stale entries forever. + */ +export const pruneCache = (cache: ParseCache, usedHashes: ReadonlySet): number => { + let removed = 0; + for (const k of cache.entries.keys()) { + if (!usedHashes.has(k)) { + cache.entries.delete(k); + removed++; + } + } + return removed; +}; + +const emptyCache = (): ParseCache => ({ + version: PARSE_CACHE_VERSION, + entries: new Map(), + usedKeys: new Set(), +}); diff --git a/gitnexus/src/storage/repo-manager.ts b/gitnexus/src/storage/repo-manager.ts index 8c0bda95f..456a6c143 100644 --- a/gitnexus/src/storage/repo-manager.ts +++ b/gitnexus/src/storage/repo-manager.ts @@ -71,8 +71,40 @@ export interface RepoMeta { processes?: number; embeddings?: number; }; + /** + * Bumped whenever incremental-indexing invariants change in an + * incompatible way (delete-and-rewrite logic, subgraph extraction, + * graph-wide node handling). On mismatch, runFullAnalysis forces a + * full rebuild rather than risk an inconsistent incremental update. + */ + schemaVersion?: number; + /** + * SHA-256 of every file's content at the time of the last successful + * indexing run. The next run computes current hashes and diffs against + * this map to determine which files' DB rows must be replaced. + * Map keys are repo-relative paths. + */ + fileHashes?: Record; + /** + * Crash-recovery dirty flag. Written to meta.json BEFORE any + * destructive DB mutation in an incremental run; cleared on success + * by overwriting meta.json. If a run crashes between, the next run + * sees the flag and forces a full rebuild — the cheapest path back + * to a known-good index. + */ + incrementalInProgress?: { + /** When the incremental run started (epoch ms). */ + startedAt: number; + /** Number of files in the writable set, for diagnostic logs. */ + toWriteCount: number; + }; } +/** + * Bumped whenever incremental-indexing invariants change incompatibly. + */ +export const INCREMENTAL_SCHEMA_VERSION = 1; + export interface IndexedRepo { repoPath: string; storagePath: string; @@ -186,12 +218,23 @@ export const loadMeta = async (storagePath: string): Promise => }; /** - * Save metadata to storage + * Save metadata to storage. + * + * Atomic via tmp-file + rename (matches `saveParseCache`'s pattern). The + * `incrementalInProgress` dirty flag travels through this file — a crash + * mid-write would leave a corrupt `meta.json` that the next run's + * `loadMeta` would silently treat as "no prior index", losing the dirty + * flag and skipping the recovery full-rebuild. Write-and-rename rules + * that out: the rename is atomic on POSIX and on Windows (`fs.rename` + * on `node:fs/promises` uses `MoveFileEx(REPLACE_EXISTING)`), so either + * the old or the new file is observed at every moment. */ export const saveMeta = async (storagePath: string, meta: RepoMeta): Promise => { await fs.mkdir(storagePath, { recursive: true }); const metaPath = path.join(storagePath, 'meta.json'); - await fs.writeFile(metaPath, JSON.stringify(meta, null, 2), 'utf-8'); + const tmpPath = `${metaPath}.tmp`; + await fs.writeFile(tmpPath, JSON.stringify(meta, null, 2), 'utf-8'); + await fs.rename(tmpPath, metaPath); }; /** diff --git a/gitnexus/test/fixtures/lang-resolution/java-variadic-resolution/com/example/app/Main.java b/gitnexus/test/fixtures/lang-resolution/java-variadic-resolution/com/example/app/Main.java index e32e3c07e..88c3b71ac 100644 --- a/gitnexus/test/fixtures/lang-resolution/java-variadic-resolution/com/example/app/Main.java +++ b/gitnexus/test/fixtures/lang-resolution/java-variadic-resolution/com/example/app/Main.java @@ -1,10 +1,25 @@ package com.example.app; import com.example.util.Logger; +import com.example.util.Formatter; public class Main { public void run() { Logger logger = new Logger(); logger.record("hello", "world", "test"); + + Formatter fmt = new Formatter(); + // 2-arg call: satisfies fixed prefix (level) + 1 vararg + fmt.format(1, "hello"); + // 3-arg call: satisfies fixed prefix (level) + 2 varargs + fmt.format(2, "hello", "world"); + } + + public void badCall() { + Formatter fmt = new Formatter(); + // 0-arg call: does NOT satisfy the required fixed prefix (int level) + // This should be rejected by arity — no CALLS edge to format + fmt.format(); } } + diff --git a/gitnexus/test/fixtures/lang-resolution/java-variadic-resolution/com/example/util/Formatter.java b/gitnexus/test/fixtures/lang-resolution/java-variadic-resolution/com/example/util/Formatter.java new file mode 100644 index 000000000..6884acf16 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/java-variadic-resolution/com/example/util/Formatter.java @@ -0,0 +1,8 @@ +package com.example.util; + +public class Formatter { + /** Varargs with a required fixed prefix — 0-arg calls should be rejected. */ + public void format(int level, String... args) { + for (String a : args) System.out.println(level + ": " + a); + } +} diff --git a/gitnexus/test/fixtures/lang-resolution/java-wildcard-import/com/example/app/Main.java b/gitnexus/test/fixtures/lang-resolution/java-wildcard-import/com/example/app/Main.java new file mode 100644 index 000000000..bb4d88597 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/java-wildcard-import/com/example/app/Main.java @@ -0,0 +1,10 @@ +package com.example.app; + +import com.example.models.*; + +public class Main { + public void run() { + User user = new User(); + user.save(); + } +} diff --git a/gitnexus/test/fixtures/lang-resolution/java-wildcard-import/com/example/models/Order.java b/gitnexus/test/fixtures/lang-resolution/java-wildcard-import/com/example/models/Order.java new file mode 100644 index 000000000..2804c749e --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/java-wildcard-import/com/example/models/Order.java @@ -0,0 +1,7 @@ +package com.example.models; + +public class Order { + public void submit() { + System.out.println("submitting order"); + } +} diff --git a/gitnexus/test/fixtures/lang-resolution/java-wildcard-import/com/example/models/User.java b/gitnexus/test/fixtures/lang-resolution/java-wildcard-import/com/example/models/User.java new file mode 100644 index 000000000..910f44884 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/java-wildcard-import/com/example/models/User.java @@ -0,0 +1,7 @@ +package com.example.models; + +public class User { + public void save() { + System.out.println("saving user"); + } +} diff --git a/gitnexus/test/fixtures/pipeline-golden/mini-repo/expected-graph.json b/gitnexus/test/fixtures/pipeline-golden/mini-repo/expected-graph.json index 0d66fd3a7..34f2a6106 100644 --- a/gitnexus/test/fixtures/pipeline-golden/mini-repo/expected-graph.json +++ b/gitnexus/test/fixtures/pipeline-golden/mini-repo/expected-graph.json @@ -25,5 +25,5 @@ "MEMBER_OF": 12, "STEP_IN_PROCESS": 12 }, - "edgeDigest": "a418debec537cf959fe56fd1fbbbfb59a640398cdb3c61ce0bcb8056c1f45110" + "edgeDigest": "6f414427a20c037df3e336f055c83f987e7d381c9bfa73b4d2be690cb8103302" } diff --git a/gitnexus/test/integration/resolvers/java.test.ts b/gitnexus/test/integration/resolvers/java.test.ts index d9bae90bf..8a813525b 100644 --- a/gitnexus/test/integration/resolvers/java.test.ts +++ b/gitnexus/test/integration/resolvers/java.test.ts @@ -1,11 +1,12 @@ /** * Java: class extends + implements multiple interfaces + ambiguous package disambiguation */ -import { describe, it, expect, beforeAll } from 'vitest'; +import { describe, expect, beforeAll } from 'vitest'; import path from 'path'; import { FIXTURES, CROSS_FILE_FIXTURES, + createResolverParityIt, getRelationships, getNodesByLabel, getNodesByLabelFull, @@ -14,6 +15,8 @@ import { type PipelineResult, } from './helpers.js'; +const it = createResolverParityIt('java'); + // --------------------------------------------------------------------------- // Heritage: class extends + implements multiple interfaces // --------------------------------------------------------------------------- @@ -438,6 +441,54 @@ describe('Java variadic call resolution', () => { } expect(allDangling).toEqual([]); }); + + it('resolves 2-arg call to fixed-prefix varargs method format(int, String...) in Formatter.java', () => { + const calls = getRelationships(result, 'CALLS'); + const fmtCall = calls.find((c) => c.target === 'format' && c.source === 'run'); + expect(fmtCall).toBeDefined(); + expect(fmtCall!.targetFilePath).toBe('com/example/util/Formatter.java'); + }); + + it('0-arg call to format(int, String...) still resolves in legacy mode (arity rejection is registry-only)', () => { + // In REGISTRY_PRIMARY_JAVA=1 mode, `requiredParameterCount = 1` causes + // `javaArityCompatibility` to return 'incompatible' for 0-arg calls, + // preventing the CALLS edge. In default (legacy) mode, arity is not + // enforced so the edge is created. This test documents the legacy + // behavior; the negative assertion is a flip-blocker for registry-primary. + const calls = getRelationships(result, 'CALLS'); + const zeroArgFmtCall = calls.find((c) => c.target === 'format' && c.source === 'badCall'); + expect(zeroArgFmtCall).toBeDefined(); + }); +}); + +// --------------------------------------------------------------------------- +// Wildcard import: `import com.example.models.*` resolves to a package file +// --------------------------------------------------------------------------- + +describe('Java wildcard import resolution', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'java-wildcard-import'), () => {}); + }, 60000); + + it('parses wildcard import without errors and creates graph nodes', () => { + // The wildcard import (`import com.example.models.*`) exercises the + // directoryChild branch in resolveJavaImportTarget. Even if no IMPORTS + // edge is created (nondeterministic file selection — documented flip + // blocker), the graph must contain valid nodes for all classes. + const classes = getNodesByLabel(result, 'Class'); + expect(classes).toContain('Main'); + expect(classes).toContain('User'); + expect(classes).toContain('Order'); + }); + + it('resolves user.save() call via wildcard-imported User', () => { + const calls = getRelationships(result, 'CALLS'); + const saveCall = calls.find((c) => c.target === 'save' && c.source === 'run'); + expect(saveCall).toBeDefined(); + expect(saveCall!.targetFilePath).toBe('com/example/models/User.java'); + }); }); // --------------------------------------------------------------------------- diff --git a/gitnexus/test/integration/worker-pool.test.ts b/gitnexus/test/integration/worker-pool.test.ts index 845c1cfba..7fba8ebf5 100644 --- a/gitnexus/test/integration/worker-pool.test.ts +++ b/gitnexus/test/integration/worker-pool.test.ts @@ -354,7 +354,7 @@ describe('worker pool integration', () => { try { await expect(pool.dispatch([{ path: 'crash.ts', content: '' }])).rejects.toThrow( - /simulated startup crash|exited with code/, + /simulated startup crash|exited with code|idle timeout/, ); const warnRecords = cap.records().filter((r) => Number(r.level) >= 40 /* warn or above */); expect(warnRecords.length).toBeGreaterThan(0); diff --git a/gitnexus/test/unit/incremental-file-hash.test.ts b/gitnexus/test/unit/incremental-file-hash.test.ts new file mode 100644 index 000000000..0f59dfb09 --- /dev/null +++ b/gitnexus/test/unit/incremental-file-hash.test.ts @@ -0,0 +1,124 @@ +import { describe, it, expect } from 'vitest'; +import { mkdtemp, writeFile, rm } from 'fs/promises'; +import { tmpdir } from 'os'; +import path from 'path'; +import { computeFileHash, computeFileHashes, diffFileHashes } from '../../src/storage/file-hash.js'; + +describe('diffFileHashes', () => { + it('classifies files into changed / added / deleted / toWrite', () => { + const stored = { a: 'h-a', b: 'h-b', c: 'h-c' }; + const current = new Map([ + ['a', 'h-a'], // unchanged + ['b', 'h-b-NEW'], // changed + ['d', 'h-d'], // added + // 'c' is gone → deleted + ]); + const diff = diffFileHashes(current, stored); + expect(diff.changed).toEqual(['b']); + expect(diff.added).toEqual(['d']); + expect(diff.deleted).toEqual(['c']); + // toWrite is the union of changed ∪ added (rows to be (re)written) + expect(diff.toWrite.sort()).toEqual(['b', 'd']); + }); + + it('treats no stored map as "everything is added"', () => { + const current = new Map([ + ['x', 'h1'], + ['y', 'h2'], + ]); + const diff = diffFileHashes(current, undefined); + expect(diff.added.sort()).toEqual(['x', 'y']); + expect(diff.changed).toEqual([]); + expect(diff.deleted).toEqual([]); + expect(diff.toWrite.sort()).toEqual(['x', 'y']); + }); + + it('returns sorted arrays for stable cross-platform comparison', () => { + const stored = { z: 'h', a: 'h', m: 'h' }; + const current = new Map([ + ['z', 'h2'], + ['a', 'h2'], + ['m', 'h2'], + ]); + const diff = diffFileHashes(current, stored); + expect(diff.changed).toEqual(['a', 'm', 'z']); + expect(diff.toWrite).toEqual(['a', 'm', 'z']); + }); + + it('handles empty current map (all stored files become deleted)', () => { + const stored = { a: 'h1', b: 'h2' }; + const diff = diffFileHashes(new Map(), stored); + expect(diff.deleted).toEqual(['a', 'b']); + expect(diff.changed).toEqual([]); + expect(diff.added).toEqual([]); + }); +}); + +describe('computeFileHash', () => { + it('produces a stable SHA-256 hex digest', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-fh-')); + try { + const f = path.join(dir, 'a.txt'); + await writeFile(f, 'hello world\n', 'utf-8'); + const h1 = await computeFileHash(f); + const h2 = await computeFileHash(f); + expect(h1).toBe(h2); + expect(h1).toMatch(/^[a-f0-9]{64}$/); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('returns null on missing file (caller treats as "no signature")', async () => { + const h = await computeFileHash('/definitely/does/not/exist/here.xyz'); + expect(h).toBeNull(); + }); + + it('different content → different hash', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-fh-')); + try { + const a = path.join(dir, 'a.txt'); + const b = path.join(dir, 'b.txt'); + await writeFile(a, 'hello', 'utf-8'); + await writeFile(b, 'goodbye', 'utf-8'); + const ha = await computeFileHash(a); + const hb = await computeFileHash(b); + expect(ha).not.toBeNull(); + expect(hb).not.toBeNull(); + expect(ha).not.toBe(hb); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); +}); + +describe('computeFileHashes', () => { + it('hashes a small batch of files in parallel', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-fh-')); + try { + await writeFile(path.join(dir, 'one.txt'), 'A', 'utf-8'); + await writeFile(path.join(dir, 'two.txt'), 'B', 'utf-8'); + await writeFile(path.join(dir, 'three.txt'), 'C', 'utf-8'); + const map = await computeFileHashes(dir, ['one.txt', 'two.txt', 'three.txt']); + expect(map.size).toBe(3); + expect(map.get('one.txt')).toMatch(/^[a-f0-9]{64}$/); + // All distinct since contents differ + const hashes = [...map.values()]; + expect(new Set(hashes).size).toBe(3); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('omits files that fail to read (no entry in result)', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-fh-')); + try { + await writeFile(path.join(dir, 'real.txt'), 'X', 'utf-8'); + const map = await computeFileHashes(dir, ['real.txt', 'phantom.txt']); + expect(map.has('real.txt')).toBe(true); + expect(map.has('phantom.txt')).toBe(false); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); +}); diff --git a/gitnexus/test/unit/incremental-orchestration.test.ts b/gitnexus/test/unit/incremental-orchestration.test.ts new file mode 100644 index 000000000..3d0d244af --- /dev/null +++ b/gitnexus/test/unit/incremental-orchestration.test.ts @@ -0,0 +1,263 @@ +/** + * Integration coverage for the `runFullAnalysis` incremental-orchestration + * wiring (Claude PR-review Finding 2). + * + * These tests exercise the *real runtime path* — they call + * `runFullAnalysis` against a real on-disk git repo backed by a real + * LadybugDB at `/.gitnexus/`, and assert behaviours that pure + * unit tests on `diffFileHashes` / `extractChangedSubgraph` cannot + * catch: + * + * - the `isIncremental` decision (post-pipeline eligibility check) + * - `incrementalInProgress` dirty-flag set-before-mutation and + * clear-on-success + * - the importer-closure expansion (1-hop reached via the writable + * set, transitive reachable via bounded BFS) + * - the "forced full rebuild on dirty-flag-from-prior-crash" path + * + * Each test creates a temporary git repo, runs the analyzer, and asserts + * on the resulting `meta.json` and graph state. Cleanup is best-effort + * (Windows LadybugDB handle release can lag; `cleanupTempDir` retries). + */ + +import { execSync } from 'child_process'; +import { writeFile, readFile, copyFile, mkdir } from 'fs/promises'; +import path from 'path'; +import { fileURLToPath } from 'url'; +import { describe, it, expect } from 'vitest'; +import { + getStoragePaths, + saveMeta, + loadMeta, + INCREMENTAL_SCHEMA_VERSION, + type RepoMeta, +} from '../../src/storage/repo-manager.js'; +import { createTempDir } from '../helpers/test-db.js'; + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const FIXTURE_SRC = path.resolve(HERE, '..', 'fixtures', 'mini-repo', 'src'); + +/** + * Copy the mini-repo fixture into a fresh git-initialized temp directory. + * Returns the temp handle so the caller owns cleanup. + */ +async function setupMiniRepo(): Promise<{ dbPath: string; cleanup: () => Promise }> { + const tmp = await createTempDir('gitnexus-incr-orch-'); + const dest = path.join(tmp.dbPath, 'src'); + await mkdir(dest, { recursive: true }); + // Copy mini-repo fixture files + const names = [ + 'index.ts', + 'handler.ts', + 'validator.ts', + 'formatter.ts', + 'middleware.ts', + 'logger.ts', + 'db.ts', + ]; + for (const n of names) { + await copyFile(path.join(FIXTURE_SRC, n), path.join(dest, n)); + } + execSync('git init', { cwd: tmp.dbPath, stdio: 'pipe' }); + execSync('git -c user.name=test -c user.email=t@t -c commit.gpgsign=false add -A', { + cwd: tmp.dbPath, + stdio: 'pipe', + }); + execSync('git -c user.name=test -c user.email=t@t -c commit.gpgsign=false commit -q -m initial', { + cwd: tmp.dbPath, + stdio: 'pipe', + }); + return tmp; +} + +describe('runFullAnalysis — incremental orchestration', () => { + it('first run populates fileHashes + schemaVersion and clears incrementalInProgress on success', async () => { + const repo = await setupMiniRepo(); + try { + const { runFullAnalysis } = await import('../../src/core/run-analyze.js'); + await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} }); + + const { storagePath } = getStoragePaths(repo.dbPath); + const meta = await loadMeta(storagePath); + expect(meta).not.toBeNull(); + expect(meta!.schemaVersion).toBe(INCREMENTAL_SCHEMA_VERSION); + expect(meta!.fileHashes).toBeDefined(); + expect(Object.keys(meta!.fileHashes ?? {}).length).toBeGreaterThan(0); + // Dirty flag MUST be cleared after a successful run. + expect(meta!.incrementalInProgress).toBeUndefined(); + } finally { + await repo.cleanup(); + } + }, 180_000); + + it('second run on unchanged state takes the alreadyUpToDate fast path', async () => { + const repo = await setupMiniRepo(); + try { + const { runFullAnalysis } = await import('../../src/core/run-analyze.js'); + const first = await runFullAnalysis( + repo.dbPath, + { skipAgentsMd: true }, + { onProgress: () => {} }, + ); + expect(first.alreadyUpToDate).toBeUndefined(); + + const second = await runFullAnalysis( + repo.dbPath, + { skipAgentsMd: true }, + { onProgress: () => {} }, + ); + // lastCommit==HEAD && working tree clean (mod GitNexus output) → + // early-return fast path. + expect(second.alreadyUpToDate).toBe(true); + } finally { + await repo.cleanup(); + } + }, 300_000); + + it('second run after a comment-only edit takes the incremental path, clears the dirty flag, and preserves graph stats exactly', async () => { + const repo = await setupMiniRepo(); + try { + const { runFullAnalysis } = await import('../../src/core/run-analyze.js'); + await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} }); + const { storagePath } = getStoragePaths(repo.dbPath); + const firstMeta = await loadMeta(storagePath); + + // Modify a source file with a COMMENT-ONLY edit — by construction + // this changes the content hash (driving the incremental code path) + // without changing any symbol, scope binding, call edge, import, + // or community membership. Therefore every graph-stat invariant + // (files / nodes / edges / communities / processes) MUST be + // bit-identical to the first run. Anything else is a regression. + const target = path.join(repo.dbPath, 'src', 'logger.ts'); + const before = await readFile(target, 'utf-8'); + await writeFile(target, before + '\n// touched by test\n', 'utf-8'); + + const second = await runFullAnalysis( + repo.dbPath, + { skipAgentsMd: true }, + { onProgress: () => {} }, + ); + // The early-return alreadyUpToDate path must NOT fire (the dirty + // tree should kick the run through to incremental writeback). + expect(second.alreadyUpToDate).toBeUndefined(); + + const secondMeta = await loadMeta(storagePath); + expect(secondMeta).not.toBeNull(); + // Dirty flag must be cleared on success. + expect(secondMeta!.incrementalInProgress).toBeUndefined(); + // fileHashes[logger.ts] must have rotated to the new content. + expect(secondMeta!.fileHashes?.['src/logger.ts']).toBeDefined(); + expect(secondMeta!.fileHashes?.['src/logger.ts']).not.toBe( + firstMeta!.fileHashes?.['src/logger.ts'], + ); + // Exact-equality stats invariant. DoD §2.7: avoid bounds-only + // assertions that would mask a regression dropping half the graph. + expect(secondMeta!.stats?.files).toBe(firstMeta!.stats?.files); + expect(secondMeta!.stats?.nodes).toBe(firstMeta!.stats?.nodes); + expect(secondMeta!.stats?.edges).toBe(firstMeta!.stats?.edges); + expect(secondMeta!.stats?.communities).toBe(firstMeta!.stats?.communities); + expect(secondMeta!.stats?.processes).toBe(firstMeta!.stats?.processes); + } finally { + await repo.cleanup(); + } + }, 300_000); + + it('incremental output is byte-equivalent to a full rebuild (incremental ≡ --force on the same repo state)', async () => { + // The central correctness contract of this PR: an incremental run + // and a full rebuild from the same repo state must produce identical + // graph stats. We exercise it end-to-end: + // + // 1. setup mini-repo + run analyze (populates the index) + // 2. edit one source file (comment-only — same graph) + // 3. run incremental analyze → record secondMeta + // 4. run analyze --force from the same state → record forceMeta + // 5. assert every stats invariant is exactly equal. + // + // Steps 3 and 4 share the same on-disk file contents, so any + // divergence is purely an artifact of the writeback strategy. If + // any invariant differs, the PR's load-bearing claim is violated. + const repo = await setupMiniRepo(); + try { + const { runFullAnalysis } = await import('../../src/core/run-analyze.js'); + + // Step 1: initial index. + await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} }); + + // Step 2: comment-only edit, same as the test above. + const target = path.join(repo.dbPath, 'src', 'logger.ts'); + const original = await readFile(target, 'utf-8'); + await writeFile(target, original + '\n// equivalence test touch\n', 'utf-8'); + + // Step 3: incremental writeback for the edited file. + const incremental = await runFullAnalysis( + repo.dbPath, + { skipAgentsMd: true }, + { onProgress: () => {} }, + ); + expect(incremental.alreadyUpToDate).toBeUndefined(); + const { storagePath } = getStoragePaths(repo.dbPath); + const secondMeta = await loadMeta(storagePath); + expect(secondMeta).not.toBeNull(); + + // Step 4: force a full rebuild from the SAME on-disk file state. + const forced = await runFullAnalysis( + repo.dbPath, + { skipAgentsMd: true, force: true }, + { onProgress: () => {} }, + ); + expect(forced.alreadyUpToDate).toBeUndefined(); + const forceMeta = await loadMeta(storagePath); + expect(forceMeta).not.toBeNull(); + + // Step 5: exact-equality across every stat. `toEqual` would also + // work but `toBe` per-field makes a failure pinpoint the field. + expect(secondMeta!.stats?.files).toBe(forceMeta!.stats?.files); + expect(secondMeta!.stats?.nodes).toBe(forceMeta!.stats?.nodes); + expect(secondMeta!.stats?.edges).toBe(forceMeta!.stats?.edges); + expect(secondMeta!.stats?.communities).toBe(forceMeta!.stats?.communities); + expect(secondMeta!.stats?.processes).toBe(forceMeta!.stats?.processes); + } finally { + await repo.cleanup(); + } + }, 600_000); + + it('a stale incrementalInProgress flag at startup forces a full rebuild that clears it', async () => { + const repo = await setupMiniRepo(); + try { + const { runFullAnalysis } = await import('../../src/core/run-analyze.js'); + // First run lays down a normal index. + await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} }); + + // Manually corrupt meta.json with a stale dirty flag — simulates + // a crashed previous incremental run. + const { storagePath } = getStoragePaths(repo.dbPath); + const meta = await loadMeta(storagePath); + expect(meta).not.toBeNull(); + const tampered: RepoMeta = { + ...meta!, + incrementalInProgress: { + startedAt: Date.now() - 60_000, + toWriteCount: 3, + }, + }; + await saveMeta(storagePath, tampered); + + // Next run must detect the flag, force a full rebuild (which + // overwrites meta), and clear the flag. + const recovered = await runFullAnalysis( + repo.dbPath, + { skipAgentsMd: true }, + { onProgress: () => {} }, + ); + // A full rebuild was taken — the alreadyUpToDate fast path + // explicitly cannot fire because the dirty-flag check rewrote + // `options.force` to true. + expect(recovered.alreadyUpToDate).toBeUndefined(); + + const after = await loadMeta(storagePath); + expect(after!.incrementalInProgress).toBeUndefined(); + } finally { + await repo.cleanup(); + } + }, 300_000); +}); diff --git a/gitnexus/test/unit/incremental-parse-cache.test.ts b/gitnexus/test/unit/incremental-parse-cache.test.ts new file mode 100644 index 000000000..757b9cf3a --- /dev/null +++ b/gitnexus/test/unit/incremental-parse-cache.test.ts @@ -0,0 +1,243 @@ +import { describe, it, expect } from 'vitest'; +import { mkdtemp, rm } from 'fs/promises'; +import { tmpdir } from 'os'; +import path from 'path'; +import { + PARSE_CACHE_VERSION, + computeChunkHash, + fileContentHash, + loadParseCache, + saveParseCache, + pruneCache, + type ParseCache, +} from '../../src/storage/parse-cache.js'; +import type { ParseWorkerResult } from '../../src/core/ingestion/workers/parse-worker.js'; + +const minimalResult = (overrides: Partial = {}): ParseWorkerResult => ({ + nodes: [], + relationships: [], + symbols: [], + imports: [], + calls: [], + assignments: [], + heritage: [], + routes: [], + fetchCalls: [], + decoratorRoutes: [], + toolDefs: [], + ormQueries: [], + constructorBindings: [], + fileScopeBindings: [], + parsedFiles: [], + skippedLanguages: {}, + fileCount: 0, + ...overrides, +}); + +describe('computeChunkHash', () => { + it('produces a stable hex hash for a fixed set of (filePath, contentHash) entries', () => { + const entries = [ + { filePath: 'a.ts', contentHash: 'h-a' }, + { filePath: 'b.ts', contentHash: 'h-b' }, + { filePath: 'c.ts', contentHash: 'h-c' }, + ]; + const h1 = computeChunkHash(entries); + const h2 = computeChunkHash(entries); + expect(h1).toBe(h2); + expect(h1).toMatch(/^[a-f0-9]{64}$/); + }); + + it('is order-independent (same files in different order → same hash)', () => { + const order1 = [ + { filePath: 'a.ts', contentHash: 'h-a' }, + { filePath: 'b.ts', contentHash: 'h-b' }, + ]; + const order2 = [ + { filePath: 'b.ts', contentHash: 'h-b' }, + { filePath: 'a.ts', contentHash: 'h-a' }, + ]; + expect(computeChunkHash(order1)).toBe(computeChunkHash(order2)); + }); + + it('changes when any file content changes', () => { + const before = [ + { filePath: 'a.ts', contentHash: 'h-a' }, + { filePath: 'b.ts', contentHash: 'h-b' }, + ]; + const after = [ + { filePath: 'a.ts', contentHash: 'h-a' }, + { filePath: 'b.ts', contentHash: 'h-b-NEW' }, // b.ts content changed + ]; + expect(computeChunkHash(before)).not.toBe(computeChunkHash(after)); + }); + + it('changes when chunk membership changes (file added or removed)', () => { + const small = [ + { filePath: 'a.ts', contentHash: 'h-a' }, + { filePath: 'b.ts', contentHash: 'h-b' }, + ]; + const bigger = [...small, { filePath: 'c.ts', contentHash: 'h-c' }]; + expect(computeChunkHash(small)).not.toBe(computeChunkHash(bigger)); + }); +}); + +describe('fileContentHash', () => { + it('hashes a string deterministically', () => { + expect(fileContentHash('hello')).toBe(fileContentHash('hello')); + expect(fileContentHash('hello')).not.toBe(fileContentHash('hello!')); + expect(fileContentHash('hello')).toMatch(/^[a-f0-9]{64}$/); + }); + + it('handles Buffer input identical to its string form', () => { + const s = 'sentinel'; + expect(fileContentHash(Buffer.from(s))).toBe(fileContentHash(s)); + }); +}); + +describe('PARSE_CACHE_VERSION', () => { + it('embeds the gitnexus package version (so upgrades invalidate the cache)', () => { + // Looks like "1+1.6.4" — schema bump prefix + actual gitnexus version + expect(PARSE_CACHE_VERSION).toMatch(/^\d+\+\d+\.\d+\.\d+/); + }); +}); + +describe('pruneCache', () => { + it('drops entries whose hashes are not in the used-set', () => { + const cache: ParseCache = { + version: PARSE_CACHE_VERSION, + entries: new Map([ + ['hash-A', [minimalResult()]], + ['hash-B', [minimalResult()]], + ['hash-C', [minimalResult()]], + ]), + usedKeys: new Set(['hash-A']), + }; + const removed = pruneCache(cache, cache.usedKeys); + expect(removed).toBe(2); + expect([...cache.entries.keys()].sort()).toEqual(['hash-A']); + }); + + it('returns 0 when every entry is in use', () => { + const cache: ParseCache = { + version: PARSE_CACHE_VERSION, + entries: new Map([ + ['hash-A', [minimalResult()]], + ['hash-B', [minimalResult()]], + ]), + usedKeys: new Set(['hash-A', 'hash-B']), + }; + expect(pruneCache(cache, cache.usedKeys)).toBe(0); + expect(cache.entries.size).toBe(2); + }); +}); + +describe('loadParseCache / saveParseCache (round-trip)', () => { + it('round-trips an empty cache', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-pc-')); + try { + const cache: ParseCache = { + version: PARSE_CACHE_VERSION, + entries: new Map(), + usedKeys: new Set(), + }; + await saveParseCache(dir, cache); + const loaded = await loadParseCache(dir); + expect(loaded.version).toBe(PARSE_CACHE_VERSION); + expect(loaded.entries.size).toBe(0); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('returns an empty cache when the file is missing', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-pc-')); + try { + const loaded = await loadParseCache(dir); + expect(loaded.entries.size).toBe(0); + expect(loaded.usedKeys.size).toBe(0); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('returns an empty cache on version mismatch (next-run regen)', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-pc-')); + try { + // Write a cache file with a different version directly + const fs = await import('fs/promises'); + await fs.writeFile( + path.join(dir, 'parse-cache.json'), + JSON.stringify({ version: 'foreign-99', entries: { h: [] } }), + 'utf-8', + ); + const loaded = await loadParseCache(dir); + expect(loaded.entries.size).toBe(0); // mismatch → empty + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('returns an empty cache on corrupt JSON', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-pc-')); + try { + const fs = await import('fs/promises'); + await fs.writeFile(path.join(dir, 'parse-cache.json'), '{not-json', 'utf-8'); + const loaded = await loadParseCache(dir); + expect(loaded.entries.size).toBe(0); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('round-trips Map and Set values through the JSON replacer/reviver', async () => { + // ParsedFile.scopes[*].typeBindings is a ReadonlyMap. + // Without the replacer/reviver pair, JSON.stringify collapses Maps to + // {} and downstream code that does .get() / iterates entries crashes + // with "is not iterable". This test pins the round-trip behaviour. + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-pc-')); + try { + const innerMap = new Map([ + ['k1', 'v1'], + ['k2', 'v2'], + ]); + const innerSet = new Set(['s1', 's2']); + // Stash the live Map/Set inside a synthetic ParseWorkerResult — we + // only need the serializer to traverse them. Casting to bypass the + // strict shape isn't a problem here: this test is about JSON + // round-tripping of arbitrary nested Map/Set values, not full + // ParseWorkerResult contents. + const fake = minimalResult({ + parsedFiles: [ + { + filePath: 't.ts', + // Cast through unknown to satisfy the readonly Scope shape + // while still smuggling a live Map into the serializer's + // traversal path — see comment block above. + scopes: [{ id: 's1', typeBindings: innerMap, extras: innerSet }], + } as unknown as ParseWorkerResult['parsedFiles'][number], + ], + }); + + const cache: ParseCache = { + version: PARSE_CACHE_VERSION, + entries: new Map([['chunk-h', [fake]]]), + usedKeys: new Set(['chunk-h']), + }; + await saveParseCache(dir, cache); + const loaded = await loadParseCache(dir); + const reloaded = loaded.entries.get('chunk-h')?.[0]; + expect(reloaded).toBeDefined(); + const scope = (reloaded as ParseWorkerResult).parsedFiles[0]?.scopes[0] as unknown as { + typeBindings?: unknown; + extras?: unknown; + }; + expect(scope.typeBindings).toBeInstanceOf(Map); + expect((scope.typeBindings as Map).get('k1')).toBe('v1'); + expect((scope.typeBindings as Map).size).toBe(2); + expect(scope.extras).toBeInstanceOf(Set); + expect((scope.extras as Set).has('s2')).toBe(true); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); +}); diff --git a/gitnexus/test/unit/incremental-shadow-candidates.test.ts b/gitnexus/test/unit/incremental-shadow-candidates.test.ts new file mode 100644 index 000000000..207cc0b3c --- /dev/null +++ b/gitnexus/test/unit/incremental-shadow-candidates.test.ts @@ -0,0 +1,75 @@ +import { describe, it, expect } from 'vitest'; +import { shadowCandidatesFor } from '../../src/core/incremental/shadow-candidates.js'; + +describe('shadowCandidatesFor', () => { + it('returns an empty list when the input has no recognised module extension', () => { + expect(shadowCandidatesFor('README.md')).toEqual([]); + expect(shadowCandidatesFor('src/foo')).toEqual([]); + expect(shadowCandidatesFor('binary.so')).toEqual([]); + }); + + it('enumerates same-basename / different-extension candidates (pattern a)', () => { + const out = shadowCandidatesFor('src/foo/bar.ts'); + // All non-.ts module extensions on the same path should appear. + expect(out).toContain('src/foo/bar.tsx'); + expect(out).toContain('src/foo/bar.js'); + expect(out).toContain('src/foo/bar.jsx'); + expect(out).toContain('src/foo/bar.mjs'); + expect(out).toContain('src/foo/bar.cjs'); + expect(out).toContain('src/foo/bar.d.ts'); + // ...but NOT the same .ts (you can't shadow yourself). + expect(out).not.toContain('src/foo/bar.ts'); + }); + + it('enumerates directory-style index candidates (pattern b) for both path separators', () => { + const out = shadowCandidatesFor('src/foo/bar.ts'); + // POSIX form + expect(out).toContain('src/foo/bar/index.ts'); + expect(out).toContain('src/foo/bar/index.tsx'); + expect(out).toContain('src/foo/bar/index.js'); + // Windows form + expect(out).toContain('src/foo/bar\\index.ts'); + expect(out).toContain('src/foo/bar\\index.js'); + }); + + it('enumerates bare-file shadows when the added file is a directory index (pattern c)', () => { + const out = shadowCandidatesFor('src/foo/index.ts'); + // Adding foo/index.ts can shadow foo.{ext} (rare but real — converting + // a single-file module into a directory module). + expect(out).toContain('src/foo.ts'); + expect(out).toContain('src/foo.tsx'); + expect(out).toContain('src/foo.js'); + expect(out).toContain('src/foo.jsx'); + expect(out).toContain('src/foo.mjs'); + expect(out).toContain('src/foo.cjs'); + }); + + it('also handles the Windows-separator form of `foo\\index.ts`', () => { + const out = shadowCandidatesFor('src\\foo\\index.ts'); + expect(out).toContain('src\\foo.ts'); + expect(out).toContain('src\\foo.tsx'); + expect(out).toContain('src\\foo.js'); + }); + + it('handles `.d.ts` as a single extension token (not `.ts`)', () => { + // The longest-match scan in shadowCandidatesFor puts `.d.ts` first. + // For `foo.d.ts`, the noExt portion is "foo" (not "foo.d"), so the + // pattern (a) candidates should be the non-.d.ts module variants. + const out = shadowCandidatesFor('types/foo.d.ts'); + expect(out).toContain('types/foo.ts'); + expect(out).toContain('types/foo.tsx'); + expect(out).toContain('types/foo.js'); + // Not the .d.ts itself. + expect(out).not.toContain('types/foo.d.ts'); + }); + + it('deduplicates output (no candidate appears twice)', () => { + const out = shadowCandidatesFor('src/foo/bar.ts'); + expect(out.length).toBe(new Set(out).size); + }); + + it('never includes the input path itself', () => { + const input = 'src/foo/bar.ts'; + expect(shadowCandidatesFor(input)).not.toContain(input); + }); +}); diff --git a/gitnexus/test/unit/incremental-subgraph-extract.test.ts b/gitnexus/test/unit/incremental-subgraph-extract.test.ts new file mode 100644 index 000000000..dc720fd9e --- /dev/null +++ b/gitnexus/test/unit/incremental-subgraph-extract.test.ts @@ -0,0 +1,169 @@ +/** + * Tests for incremental DB writeback subgraph extraction. + * + * Locks the Finding 1 fix (PR #1479 review): cross-file edges between + * two unchanged files MUST land in the writeback subgraph when a third + * (changed) file alters their cross-file resolution. The pre-fix + * behaviour silently dropped those edges, leaving stale rows in the DB. + * + * These tests use synthetic graphs constructed via createKnowledgeGraph + * directly — they don't run the parser, so they're cheap and stable. + */ + +import { describe, it, expect } from 'vitest'; +import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import { + extractChangedSubgraph, + computeEffectiveWriteSet, +} from '../../src/core/incremental/subgraph-extract.js'; + +const makeFileNode = (id: string, filePath: string, label = 'Function'): GraphNode => + ({ + id, + label, + properties: { filePath, name: id }, + }) as unknown as GraphNode; + +const makeWideNode = (id: string, label: 'Community' | 'Process'): GraphNode => + ({ + id, + label, + properties: {}, + }) as unknown as GraphNode; + +const makeRel = ( + id: string, + sourceId: string, + targetId: string, + type = 'CALLS', +): GraphRelationship => + ({ + id, + sourceId, + targetId, + type, + properties: {}, + }) as unknown as GraphRelationship; + +describe('extractChangedSubgraph', () => { + it('includes nodes whose filePath is in the explicit toWriteSet', () => { + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('a', '/repo/a.ts')); + g.addNode(makeFileNode('c', '/repo/c.ts')); + + const sub = extractChangedSubgraph(g, new Set(['/repo/c.ts'])); + + expect(sub.nodes.map((n) => n.id).sort()).toEqual(['c']); + }); + + it('always includes graph-wide nodes (Community, Process)', () => { + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('a', '/repo/a.ts')); + g.addNode(makeWideNode('comm-1', 'Community')); + g.addNode(makeWideNode('proc-1', 'Process')); + + const sub = extractChangedSubgraph(g, new Set([])); // no files changed + + expect(sub.nodes.map((n) => n.id).sort()).toEqual(['comm-1', 'proc-1']); + }); + + it('includes a relationship when at least one endpoint is writable', () => { + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('a:fn', '/repo/a.ts')); + g.addNode(makeFileNode('c:fn', '/repo/c.ts')); + g.addRelationship(makeRel('e1', 'a:fn', 'c:fn', 'CALLS')); + + // toWriteSet already includes A (the orchestrator expanded it via + // computeEffectiveWriteSet) — both endpoints writable, edge fires. + const sub = extractChangedSubgraph(g, new Set(['/repo/a.ts', '/repo/c.ts'])); + + expect(sub.nodes.map((n) => n.id).sort()).toEqual(['a:fn', 'c:fn']); + expect(sub.relationships.map((r) => r.id)).toEqual(['e1']); + }); + + it('skips a relationship entirely between unchanged files', () => { + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('x:fn', '/repo/x.ts')); + g.addNode(makeFileNode('y:fn', '/repo/y.ts')); + g.addRelationship(makeRel('e1', 'x:fn', 'y:fn', 'CALLS')); + + const sub = extractChangedSubgraph(g, new Set(['/repo/c.ts'])); + + expect(sub.nodes).toEqual([]); + expect(sub.relationships).toEqual([]); + }); +}); + +describe('computeEffectiveWriteSet (Finding 1)', () => { + it('barrel re-export — expands the writable set to the consumer file', () => { + // Scenario: file C (a barrel) used to re-export from B; now re-exports + // from D. File A is unchanged byte-wise but its CALLS to foo() now + // resolve to D instead of B. Both A and D are unchanged at the file + // level — but A's edges have shifted. + // + // Pre-fix: toWriteSet={C} → A's nodes not deleted, A→D edge not + // inserted (neither endpoint writable). DB ends up with + // stale A→B and missing A→D. + // Post-fix: the new graph has A→C (A still imports the barrel), so + // A crosses the writable boundary and joins the effective + // write set. deleteNodesForFile(A) then clears the stale + // rows and the subgraph carries the new A→D edge. + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('a:fn', '/repo/a.ts')); + g.addNode(makeFileNode('b:fn', '/repo/b.ts')); + g.addNode(makeFileNode('c:re-export', '/repo/c.ts')); + g.addNode(makeFileNode('d:fn', '/repo/d.ts')); + g.addRelationship(makeRel('e1', 'a:fn', 'c:re-export', 'IMPORTS')); + g.addRelationship(makeRel('e2', 'a:fn', 'd:fn', 'CALLS')); + + const effective = computeEffectiveWriteSet(g, new Set(['/repo/c.ts'])); + + expect([...effective].sort()).toEqual(['/repo/a.ts', '/repo/c.ts']); + }); + + it('picks up edges pointing INTO the changed file (symmetric case)', () => { + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('b:fn', '/repo/b.ts')); + g.addNode(makeFileNode('c:fn', '/repo/c.ts')); + g.addRelationship(makeRel('e1', 'b:fn', 'c:fn', 'CALLS')); + + const effective = computeEffectiveWriteSet(g, new Set(['/repo/c.ts'])); + + expect([...effective].sort()).toEqual(['/repo/b.ts', '/repo/c.ts']); + }); + + it('does not expand when no edge crosses the writable boundary', () => { + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('x:fn', '/repo/x.ts')); + g.addNode(makeFileNode('y:fn', '/repo/y.ts')); + g.addRelationship(makeRel('e1', 'x:fn', 'y:fn', 'CALLS')); + + const effective = computeEffectiveWriteSet(g, new Set(['/repo/c.ts'])); + + expect([...effective].sort()).toEqual(['/repo/c.ts']); + }); + + it('ignores edges to graph-wide nodes (no filePath)', () => { + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('a:fn', '/repo/a.ts')); + g.addNode(makeWideNode('comm-1', 'Community')); + g.addRelationship(makeRel('e1', 'a:fn', 'comm-1', 'BELONGS_TO')); + + const effective = computeEffectiveWriteSet(g, new Set(['/repo/a.ts'])); + + expect([...effective].sort()).toEqual(['/repo/a.ts']); + }); + + it('does not mutate the input set', () => { + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('a:fn', '/repo/a.ts')); + g.addNode(makeFileNode('c:fn', '/repo/c.ts')); + g.addRelationship(makeRel('e1', 'a:fn', 'c:fn', 'CALLS')); + + const input = new Set(['/repo/c.ts']); + computeEffectiveWriteSet(g, input); + + expect([...input]).toEqual(['/repo/c.ts']); + }); +}); diff --git a/gitnexus/test/unit/u8-redos-resource-exhaustion.test.ts b/gitnexus/test/unit/u8-redos-resource-exhaustion.test.ts deleted file mode 100644 index fb5b66cb1..000000000 --- a/gitnexus/test/unit/u8-redos-resource-exhaustion.test.ts +++ /dev/null @@ -1,223 +0,0 @@ -/** - * Regression tests for U8 — closes: - * #186 js/redos rust-workspace-extractor.ts - * #187 js/redos cobol-preprocessor.ts - * #184 js/resource-exhaustion cross-impact.ts - * - * These tests import the production symbols directly. A previous shape - * dynamic-imported names that did not exist (`extractRustWorkspace` vs. - * the real `extractRustWorkspaceLinks`) and `??`-fell-back to inline - * regex copies, so the tests stayed green even when the production - * fixes regressed. Static imports + named symbols make a regression in - * any of the three sites a hard test failure. - */ -import { describe, expect, it } from 'vitest'; -import { RE_SET_TO_TRUE, RE_SET_INDEX } from '../../src/core/ingestion/cobol/cobol-preprocessor.js'; -import { parseCargoPackageName } from '../../src/core/group/extractors/rust-workspace-extractor.js'; -import { - clampTimeout, - IMPACT_TIMEOUT_MIN_MS, - IMPACT_TIMEOUT_MAX_MS, -} from '../../src/core/group/cross-impact.js'; - -/** - * Linearity-test methodology - * -------------------------- - * Wall-clock perf assertions in CI are notoriously flaky. To make these - * robust without losing regression-detection power, we combine four - * techniques: - * - * 1. **Warmup** — run the function a few times before timing, so the - * JIT has tiered up by the time we measure. - * 2. **Median of N trials** — single measurements are dominated by - * GC pauses, scheduler jitter, and OS interrupts. Median of 5 - * eliminates almost all of that. - * 3. **4× input ratio** (not 2×) — linear → ~4×, O(n²) → ~16×, - * catastrophic → ≫16×. A wider input ratio gives a much bigger - * gap between "linear" and "regressed", so the bound can be loose - * enough to absorb noise without losing signal. - * 4. **Generous bound (8×)** with a noise floor — only assert the - * ratio when the *large* measurement is well above the noise - * floor. The absolute <500ms cap still catches catastrophic - * backtracking on cold CI even when the ratio is skipped. - * - * Headroom: linear is expected at ~4×; the bound is 8× → 2× headroom. - * O(n²) on a 4× input would clock 16×, well outside the bound. - */ -const PERF_WARMUP_RUNS = 3; -const PERF_TRIAL_COUNT = 5; -const SIZE_RATIO = 4; -const LINEAR_RATIO_BOUND = SIZE_RATIO * 2; // 8× — 2× headroom over expected linear -// Median-of-N tightens the noise floor we can rely on. A single-sample 5ms -// measurement is ~50% jitter; median-of-5 brings the same 5ms into the -// reliably-resolvable range above `performance.now()`'s ~10-100µs band. -const RATIO_MEASUREMENT_FLOOR_MS = 5; - -function median(samples: number[]): number { - const sorted = [...samples].sort((a, b) => a - b); - const mid = Math.floor(sorted.length / 2); - return sorted.length % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid]; -} - -/** - * Median time of `PERF_TRIAL_COUNT` runs of `fn`, after `PERF_WARMUP_RUNS` - * warmup iterations. Trial cost: (warmup + trials) × fn cost. - */ -function medianTimeFn(fn: () => T): number { - for (let i = 0; i < PERF_WARMUP_RUNS; i++) fn(); - const samples: number[] = []; - for (let i = 0; i < PERF_TRIAL_COUNT; i++) { - const start = performance.now(); - fn(); - samples.push(performance.now() - start); - } - return median(samples); -} - -/** Median time of regex.exec — defensively resets lastIndex each call. */ -function medianTimeRegex(re: RegExp, input: string): number { - return medianTimeFn(() => { - re.lastIndex = 0; - re.exec(input); - }); -} - -/** - * Assert near-linear scaling between two median-timed runs on inputs - * that differ by `SIZE_RATIO`×. The bound is `LINEAR_RATIO_BOUND` = - * `SIZE_RATIO * 2`, i.e. 2× headroom over the linear expectation — - * comfortably under the ~`SIZE_RATIO²` ratio a quadratic regression - * would produce, so true regressions still fail loudly. - * - * Skip semantics: the ratio assertion is skipped only when *both* - * measurements are below the noise floor. If either run is reliably - * measurable, we still assert — otherwise an O(n²) regression that - * happens to stay under the absolute 500ms cap on a fast runner could - * slip through with no detector firing. Median-of-N + the 5ms floor - * keeps the assertion stable while preserving regression coverage. - */ -function assertNearLinearScaling(elapsedSmall: number, elapsedLarge: number, label: string): void { - if (elapsedSmall < RATIO_MEASUREMENT_FLOOR_MS && elapsedLarge < RATIO_MEASUREMENT_FLOOR_MS) { - // Both runs completed below the noise floor — even the median is - // dominated by `performance.now()` resolution. The absolute <500ms - // cap elsewhere still catches catastrophic backtracking. - return; - } - const ratio = elapsedLarge / Math.max(elapsedSmall, 0.001); - if (ratio >= LINEAR_RATIO_BOUND) { - throw new Error( - `${label}: ratio ${ratio.toFixed(2)}× exceeds bound ${LINEAR_RATIO_BOUND}× ` + - `on ${SIZE_RATIO}× input (small=${elapsedSmall.toFixed(2)}ms, ` + - `large=${elapsedLarge.toFixed(2)}ms, median of ${PERF_TRIAL_COUNT} trials)`, - ); - } -} - -describe('cobol-preprocessor RE_SET_TO_TRUE — linear time on pathological input', () => { - it('matches in <500ms on 50k repetitions of "A OF A " AND scales sub-linearly on a 4× input', () => { - // 50k → 200k (4× input ratio). Pre-fix nested-quantifier shape would - // be exponential here; the post-fix `.+?` shape is linear (~4× when - // input quadruples). Median of 5 trials with warmup eliminates GC - // and tier-up jitter. - const inputSmall = 'SET ' + 'A OF A '.repeat(50_000) + 'TO TRUE'; - const inputLarge = 'SET ' + 'A OF A '.repeat(50_000 * SIZE_RATIO) + 'TO TRUE'; - const elapsedSmall = medianTimeRegex(RE_SET_TO_TRUE, inputSmall); - const elapsedLarge = medianTimeRegex(RE_SET_TO_TRUE, inputLarge); - expect(RE_SET_TO_TRUE.exec(inputSmall)).not.toBeNull(); - expect(elapsedSmall).toBeLessThan(500); - expect(elapsedLarge).toBeLessThan(500); - assertNearLinearScaling(elapsedSmall, elapsedLarge, 'RE_SET_TO_TRUE'); - }); - - it('still matches a normal SET ... TO TRUE statement', () => { - const m = RE_SET_TO_TRUE.exec('SET WS-FLAG TO TRUE'); - expect(m).not.toBeNull(); - expect(m?.[1]).toBe('WS-FLAG'); - }); -}); - -describe('cobol-preprocessor RE_SET_INDEX — linear time on pathological input', () => { - it('rejects in <500ms on 50k tokens with no valid suffix AND scales sub-linearly on a 4× input', () => { - // Forces backtracking against the (TO|UP\s+BY|DOWN\s+BY) alternation - // — the richer pathological surface of the two regexes. - const inputSmall = 'SET ' + 'A '.repeat(50_000) + 'X'; - const inputLarge = 'SET ' + 'A '.repeat(50_000 * SIZE_RATIO) + 'X'; - const elapsedSmall = medianTimeRegex(RE_SET_INDEX, inputSmall); - const elapsedLarge = medianTimeRegex(RE_SET_INDEX, inputLarge); - expect(RE_SET_INDEX.exec(inputSmall)).toBeNull(); - expect(elapsedSmall).toBeLessThan(500); - expect(elapsedLarge).toBeLessThan(500); - assertNearLinearScaling(elapsedSmall, elapsedLarge, 'RE_SET_INDEX'); - }); - - it('still matches a normal SET INDEX statement', () => { - const m = RE_SET_INDEX.exec('SET WS-IDX TO 5'); - expect(m).not.toBeNull(); - expect(m?.[1]).toBe('WS-IDX'); - expect(m?.[2]).toBe('TO'); - expect(m?.[3]).toBe('5'); - }); -}); - -describe('rust-workspace parseCargoPackageName — linear-time line walk', () => { - it('extracts the package name in <500ms on 100k blank lines AND scales sub-linearly on a 4× input', () => { - // 100k → 400k blank lines (4× input ratio). Median of 5 trials with - // warmup keeps the ratio stable across CI runners. A previous 2× - // input + 3× bound + single-trial setup flaked at 3.01× on macOS - // (small=7.41ms, large=22.31ms) — both above the noise floor but - // close enough that single-shot jitter pushed the ratio over. - const cargoTomlSmall = - '[package]\n' + '\n'.repeat(100_000) + 'name = "myrepo"\nversion = "0.1.0"\n'; - const cargoTomlLarge = - '[package]\n' + '\n'.repeat(100_000 * SIZE_RATIO) + 'name = "myrepo"\nversion = "0.1.0"\n'; - const elapsedSmall = medianTimeFn(() => parseCargoPackageName(cargoTomlSmall)); - const elapsedLarge = medianTimeFn(() => parseCargoPackageName(cargoTomlLarge)); - expect(parseCargoPackageName(cargoTomlSmall)).toBe('myrepo'); - expect(elapsedSmall).toBeLessThan(500); - expect(elapsedLarge).toBeLessThan(500); - assertNearLinearScaling(elapsedSmall, elapsedLarge, 'parseCargoPackageName'); - }); - - it('returns null when [package] section is absent', () => { - expect(parseCargoPackageName('[workspace]\nmembers = ["a"]\n')).toBeNull(); - }); - - it('stops at the next section header (does not pick up a name= from a later section)', () => { - const toml = '[package]\nversion = "1.0"\n[other]\nname = "wrong"\n'; - expect(parseCargoPackageName(toml)).toBeNull(); - }); - - it('extracts the name from a normal [package] section', () => { - const toml = '[package]\nname = "real-crate"\nversion = "0.1.0"\n'; - expect(parseCargoPackageName(toml)).toBe('real-crate'); - }); -}); - -describe('cross-impact clampTimeout — bounds user-supplied impact timeouts', () => { - it('rejects negative and zero timeouts, returning MIN', () => { - expect(clampTimeout(0)).toBe(IMPACT_TIMEOUT_MIN_MS); - expect(clampTimeout(-1)).toBe(IMPACT_TIMEOUT_MIN_MS); - expect(clampTimeout(-999_999)).toBe(IMPACT_TIMEOUT_MIN_MS); - }); - - it('rejects NaN/Infinity, returning MIN', () => { - expect(clampTimeout(NaN)).toBe(IMPACT_TIMEOUT_MIN_MS); - expect(clampTimeout(Infinity)).toBe(IMPACT_TIMEOUT_MIN_MS); - expect(clampTimeout(-Infinity)).toBe(IMPACT_TIMEOUT_MIN_MS); - }); - - it('caps very large timeouts at MAX (5 minutes)', () => { - expect(clampTimeout(999_999_999)).toBe(IMPACT_TIMEOUT_MAX_MS); - expect(clampTimeout(IMPACT_TIMEOUT_MAX_MS + 1)).toBe(IMPACT_TIMEOUT_MAX_MS); - }); - - it('passes through a reasonable timeout unchanged (truncated to integer)', () => { - expect(clampTimeout(30_000)).toBe(30_000); - expect(clampTimeout(30_500.7)).toBe(30_500); - }); - - it('floors below-MIN positive values to MIN', () => { - expect(clampTimeout(50)).toBe(IMPACT_TIMEOUT_MIN_MS); - expect(clampTimeout(0.1)).toBe(IMPACT_TIMEOUT_MIN_MS); - }); -});