From 3b43eb8b474d560c80b1318f755be5498b42ff03 Mon Sep 17 00:00:00 2001 From: evolution Date: Sat, 6 Jun 2026 11:47:17 +0800 Subject: [PATCH 01/17] fix(go): capture multi-name declarations (#2032) --- .../ingestion/field-extractors/configs/go.ts | 48 ++++++++- .../ingestion/field-extractors/generic.ts | 15 ++- .../src/core/ingestion/languages/go/query.ts | 8 +- .../src/core/ingestion/parsing-processor.ts | 36 ++++++- .../src/core/ingestion/tree-sitter-queries.ts | 5 +- .../variable-extractors/configs/go.ts | 85 ++++++++++++++-- .../ingestion/variable-extractors/generic.ts | 22 +++-- gitnexus/src/core/ingestion/variable-types.ts | 9 ++ .../core/ingestion/workers/parse-worker.ts | 57 +++++++---- .../go-multi-name-sequential-metadata.test.ts | 83 ++++++++++++++++ .../go-multi-name-worker-metadata.test.ts | 97 +++++++++++++++++++ .../integration/tree-sitter-languages.test.ts | 34 +++++++ gitnexus/test/unit/field-extraction.test.ts | 42 ++++++++ .../go/go-captures-smoke.test.ts | 32 ++++++ .../test/unit/variable-extraction.test.ts | 70 +++++++++++++ 15 files changed, 597 insertions(+), 46 deletions(-) create mode 100644 gitnexus/test/integration/go-multi-name-sequential-metadata.test.ts create mode 100644 gitnexus/test/integration/go-multi-name-worker-metadata.test.ts diff --git a/gitnexus/src/core/ingestion/field-extractors/configs/go.ts b/gitnexus/src/core/ingestion/field-extractors/configs/go.ts index b51f1f046..0e37b89c4 100644 --- a/gitnexus/src/core/ingestion/field-extractors/configs/go.ts +++ b/gitnexus/src/core/ingestion/field-extractors/configs/go.ts @@ -3,6 +3,8 @@ import { SupportedLanguages } from 'gitnexus-shared'; import type { FieldExtractionConfig } from '../generic.js'; import { extractSimpleTypeName } from '../../type-extractors/shared.js'; +import type { FieldVisibility } from '../../field-types.js'; +import type { SyntaxNode } from '../../utils/ast-helpers.js'; /** * Go field extraction config. @@ -13,14 +15,52 @@ import { extractSimpleTypeName } from '../../type-extractors/shared.js'; * Visibility in Go is based on the first character: uppercase = exported (public), * lowercase = unexported (package). */ +function goVisibilityForName(name: string): FieldVisibility { + const first = name.charAt(0); + return first === first.toUpperCase() && first !== first.toLowerCase() ? 'public' : 'package'; +} + +function extractGoFieldNames(node: SyntaxNode): string[] { + const names: string[] = []; + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child?.type === 'field_identifier') names.push(child.text); + } + return names; +} + export const goConfig: FieldExtractionConfig = { language: SupportedLanguages.Go, - typeDeclarationNodes: ['type_declaration'], + typeDeclarationNodes: ['type_declaration', 'struct_type'], fieldNodeTypes: ['field_declaration'], bodyNodeTypes: ['field_declaration_list'], defaultVisibility: 'package', + extractOwnerName(node) { + if (node.type === 'struct_type') { + return node.parent?.type === 'type_spec' + ? node.parent.childForFieldName('name')?.text + : undefined; + } + const typeSpec = node.namedChildren.find((child) => child.type === 'type_spec'); + return typeSpec?.childForFieldName('name')?.text; + }, + + findBodyNodes(node) { + if (node.type === 'struct_type') { + const body = node.namedChildren.find((child) => child.type === 'field_declaration_list'); + return body ? [body] : []; + } + const typeSpec = node.namedChildren.find((child) => child.type === 'type_spec'); + const typeNode = typeSpec?.childForFieldName('type'); + const body = typeNode?.namedChildren.find((child) => child.type === 'field_declaration_list'); + return body ? [body] : []; + }, + extractName(node) { + const firstName = extractGoFieldNames(node)[0]; + if (firstName) return firstName; + // field_declaration > name:(field_identifier) const name = node.childForFieldName('name'); if (name) return name.text; @@ -32,6 +72,8 @@ export const goConfig: FieldExtractionConfig = { return undefined; }, + extractNames: extractGoFieldNames, + extractType(node) { // field_declaration > type:(type_identifier | pointer_type | ...) const typeNode = node.childForFieldName('type'); @@ -54,6 +96,10 @@ export const goConfig: FieldExtractionConfig = { return 'package'; }, + extractVisibilityForName(_node, name) { + return goVisibilityForName(name); + }, + isStatic(_node) { return false; // Go has no static fields }, diff --git a/gitnexus/src/core/ingestion/field-extractors/generic.ts b/gitnexus/src/core/ingestion/field-extractors/generic.ts index 4cb4a5b1b..68f77dc8b 100644 --- a/gitnexus/src/core/ingestion/field-extractors/generic.ts +++ b/gitnexus/src/core/ingestion/field-extractors/generic.ts @@ -33,6 +33,10 @@ export interface FieldExtractionConfig { bodyNodeTypes: string[]; /** Default visibility when no modifier is present */ defaultVisibility: FieldVisibility; + /** Extract owner type name from a type declaration node. */ + extractOwnerName?: (node: SyntaxNode) => string | undefined; + /** Find body nodes inside a type declaration node. */ + findBodyNodes?: (node: SyntaxNode) => SyntaxNode[]; /** * Extract field name from a field declaration node. * Use this for nodes that declare exactly one field. @@ -49,6 +53,8 @@ export interface FieldExtractionConfig { extractType: (node: SyntaxNode) => string | undefined; /** Extract visibility from a field declaration node */ extractVisibility: (node: SyntaxNode) => FieldVisibility; + /** Extract visibility for one field name from a multi-name declaration. */ + extractVisibilityForName?: (node: SyntaxNode, name: string) => FieldVisibility; /** Check if a field is static */ isStatic: (node: SyntaxNode) => boolean; /** Check if a field is readonly/final/const */ @@ -84,10 +90,9 @@ export function createFieldExtractor(config: FieldExtractionConfig): FieldExtrac extract(node: SyntaxNode, context: FieldExtractorContext): ExtractedFields | null { if (!this.isTypeDeclaration(node)) return null; - const nameNode = node.childForFieldName('name'); - if (!nameNode) return null; + const ownerFqn = config.extractOwnerName?.(node) ?? node.childForFieldName('name')?.text; + if (!ownerFqn) return null; - const ownerFqn = nameNode.text; const fields: FieldInfo[] = []; // Find body container(s) @@ -110,6 +115,8 @@ export function createFieldExtractor(config: FieldExtractionConfig): FieldExtrac // ------------------------------------------------------------------ private findBodies(node: SyntaxNode): SyntaxNode[] { + if (config.findBodyNodes) return config.findBodyNodes(node); + const result: SyntaxNode[] = []; // Try named 'body' field first const bodyField = node.childForFieldName('body'); @@ -179,7 +186,7 @@ export function createFieldExtractor(config: FieldExtractionConfig): FieldExtrac return { name, type, - visibility: config.extractVisibility(node), + visibility: config.extractVisibilityForName?.(node, name) ?? config.extractVisibility(node), isStatic: config.isStatic(node), isReadonly: config.isReadonly(node), sourceFile: context.filePath, diff --git a/gitnexus/src/core/ingestion/languages/go/query.ts b/gitnexus/src/core/ingestion/languages/go/query.ts index 48387582f..4246ee3b4 100644 --- a/gitnexus/src/core/ingestion/languages/go/query.ts +++ b/gitnexus/src/core/ingestion/languages/go/query.ts @@ -53,11 +53,15 @@ const GO_SCOPE_QUERY = ` ;; Declarations — variables (var_declaration (var_spec - name: (identifier) @declaration.name)) @declaration.variable + (identifier) @declaration.name)) @declaration.variable +(var_declaration + (var_spec_list + (var_spec + (identifier) @declaration.name))) @declaration.variable (const_declaration (const_spec - name: (identifier) @declaration.name)) @declaration.const + (identifier) @declaration.name)) @declaration.const (short_var_declaration left: (expression_list (identifier) @declaration.name)) @declaration.variable diff --git a/gitnexus/src/core/ingestion/parsing-processor.ts b/gitnexus/src/core/ingestion/parsing-processor.ts index 367fc5550..3a9bb6526 100644 --- a/gitnexus/src/core/ingestion/parsing-processor.ts +++ b/gitnexus/src/core/ingestion/parsing-processor.ts @@ -27,6 +27,7 @@ import { import { detectFrameworkFromAST } from './framework-detection.js'; import { buildTypeEnv } from './type-env.js'; import type { FieldInfo, FieldExtractorContext } from './field-types.js'; +import type { VariableExtractorContext, VariableInfo } from './variable-types.js'; import type { MethodInfo } from './method-types.js'; import { buildMethodProps, @@ -387,7 +388,7 @@ function seqGetFieldInfo( return cached; } -const processParsingSequential = async ( +export const processParsingSequential = async ( graph: KnowledgeGraph, files: { path: string; content: string }[], symbolTable: SymbolTableWriter, @@ -409,6 +410,7 @@ const processParsingSequential = async ( seqFieldInfoCache.clear(); seqMethodExtractCache.clear(); seqMethodMapCache.clear(); + const seqVariableInfoCache = new Map>(); onFileProgress?.(i + 1, total, file.path); @@ -917,11 +919,43 @@ const processParsingSequential = async ( // All 15 tree-sitter languages register a FieldExtractor — no fallback needed. } + if ( + (nodeLabel === 'Const' || nodeLabel === 'Static' || nodeLabel === 'Variable') && + definitionNode && + provider.variableExtractor + ) { + let variableInfoByName = seqVariableInfoCache.get(definitionNode.startIndex); + if (!variableInfoByName) { + const varCtx: VariableExtractorContext = { + filePath: file.path, + language, + }; + variableInfoByName = new Map( + provider.variableExtractor + .extractAll(definitionNode, varCtx) + .map((info) => [info.name, info]), + ); + seqVariableInfoCache.set(definitionNode.startIndex, variableInfoByName); + } + const varInfo = variableInfoByName.get(nodeName); + if (varInfo) { + if (varInfo.type) declaredType = varInfo.type; + seqVisibility = varInfo.visibility; + seqIsStatic = varInfo.isStatic; + methodProps.isConst = varInfo.isConst; + methodProps.isMutable = varInfo.isMutable; + methodProps.scope = varInfo.scope; + } + } + // Apply field metadata to the graph node retroactively if (seqVisibility !== undefined) node.properties.visibility = seqVisibility; if (seqIsStatic !== undefined) node.properties.isStatic = seqIsStatic; if (seqIsReadonly !== undefined) node.properties.isReadonly = seqIsReadonly; if (declaredType !== undefined) node.properties.declaredType = declaredType; + if (methodProps.isConst !== undefined) node.properties.isConst = methodProps.isConst; + if (methodProps.isMutable !== undefined) node.properties.isMutable = methodProps.isMutable; + if (methodProps.scope !== undefined) node.properties.scope = methodProps.scope; symbolTable.add(file.path, nodeName, nodeId, nodeLabel, { parameterCount: methodProps.parameterCount as number | undefined, diff --git a/gitnexus/src/core/ingestion/tree-sitter-queries.ts b/gitnexus/src/core/ingestion/tree-sitter-queries.ts index 472003210..d6e25bd57 100644 --- a/gitnexus/src/core/ingestion/tree-sitter-queries.ts +++ b/gitnexus/src/core/ingestion/tree-sitter-queries.ts @@ -835,8 +835,9 @@ export const GO_QUERIES = ` (call_expression function: (selector_expression field: (field_identifier) @call.name)) @call ; Const/var declarations -(const_declaration (const_spec name: (identifier) @name)) @definition.const -(var_declaration (var_spec name: (identifier) @name)) @definition.variable +(const_declaration (const_spec (identifier) @name)) @definition.const +(var_declaration (var_spec (identifier) @name)) @definition.variable +(var_declaration (var_spec_list (var_spec (identifier) @name))) @definition.variable ; Short variable declaration: x := 5 (short_var_declaration left: (expression_list (identifier) @name)) @definition.variable diff --git a/gitnexus/src/core/ingestion/variable-extractors/configs/go.ts b/gitnexus/src/core/ingestion/variable-extractors/configs/go.ts index 2277aecfc..06c18d396 100644 --- a/gitnexus/src/core/ingestion/variable-extractors/configs/go.ts +++ b/gitnexus/src/core/ingestion/variable-extractors/configs/go.ts @@ -21,7 +21,72 @@ import type { SyntaxNode } from '../../utils/ast-helpers.js'; * Visibility: uppercase first letter = exported (public), lowercase = unexported (package). */ +function goVisibilityForName(name: string): VariableVisibility { + const firstChar = name.charAt(0); + return firstChar === firstChar.toUpperCase() && firstChar !== firstChar.toLowerCase() + ? 'public' + : 'package'; +} + +function collectGoSpecNames(spec: SyntaxNode): string[] { + const names: string[] = []; + for (let i = 0; i < spec.namedChildCount; i++) { + const child = spec.namedChild(i); + if (!child) continue; + if (child.type === 'identifier') { + names.push(child.text); + continue; + } + break; + } + return names; +} + +function collectGoSpecs(node: SyntaxNode): SyntaxNode[] { + const specs: SyntaxNode[] = []; + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child?.type === 'var_spec' || child?.type === 'const_spec') { + specs.push(child); + continue; + } + if (child?.type === 'var_spec_list') { + specs.push(...collectGoSpecs(child)); + } + } + return specs; +} + +function collectGoDeclarationNames(node: SyntaxNode): string[] { + if (node.type === 'short_var_declaration') { + const left = node.childForFieldName('left'); + if (left?.type !== 'expression_list') return []; + return left.namedChildren + .filter((child: SyntaxNode) => child.type === 'identifier') + .map((child: SyntaxNode) => child.text); + } + + const names: string[] = []; + for (const spec of collectGoSpecs(node)) names.push(...collectGoSpecNames(spec)); + return names; +} + +function findGoSpecForName(node: SyntaxNode, name: string): SyntaxNode | undefined { + for (const spec of collectGoSpecs(node)) { + if (collectGoSpecNames(spec).includes(name)) return spec; + } + return undefined; +} + +function extractGoSpecType(spec: SyntaxNode): string | undefined { + const typeNode = spec.childForFieldName('type'); + return typeNode ? (extractSimpleTypeName(typeNode) ?? typeNode.text?.trim()) : undefined; +} + function extractGoVarName(node: SyntaxNode): string | undefined { + const firstName = collectGoDeclarationNames(node)[0]; + if (firstName) return firstName; + // var_declaration/const_declaration → var_spec/const_spec → identifier for (let i = 0; i < node.namedChildCount; i++) { const child = node.namedChild(i); @@ -47,12 +112,9 @@ function extractGoVarName(node: SyntaxNode): string | undefined { } function extractGoVarType(node: SyntaxNode): string | undefined { - for (let i = 0; i < node.namedChildCount; i++) { - const child = node.namedChild(i); - if (child?.type === 'var_spec' || child?.type === 'const_spec') { - const typeNode = child.childForFieldName('type'); - if (typeNode) return extractSimpleTypeName(typeNode) ?? typeNode.text?.trim(); - } + for (const spec of collectGoSpecs(node)) { + const typeName = extractGoSpecType(spec); + if (typeName) return typeName; } return undefined; } @@ -66,6 +128,11 @@ export const goVariableConfig: VariableExtractionConfig = { extractName: extractGoVarName, extractType: extractGoVarType, + extractTypeForName(node, name) { + const spec = findGoSpecForName(node, name); + return spec ? extractGoSpecType(spec) : undefined; + }, + extractVisibility(node): VariableVisibility { const name = extractGoVarName(node); if (!name) return 'package'; @@ -76,6 +143,12 @@ export const goVariableConfig: VariableExtractionConfig = { : 'package'; }, + extractNames: collectGoDeclarationNames, + + extractVisibilityForName(_node, name): VariableVisibility { + return goVisibilityForName(name); + }, + isConst(node) { return node.type === 'const_declaration'; }, diff --git a/gitnexus/src/core/ingestion/variable-extractors/generic.ts b/gitnexus/src/core/ingestion/variable-extractors/generic.ts index 1128805bd..e3eefeae6 100644 --- a/gitnexus/src/core/ingestion/variable-extractors/generic.ts +++ b/gitnexus/src/core/ingestion/variable-extractors/generic.ts @@ -77,13 +77,17 @@ export function createVariableExtractor(config: VariableExtractionConfig): Varia }, extract(node: SyntaxNode, context: VariableExtractorContext): VariableInfo | null { - if (!allNodeTypes.has(node.type)) return null; + return this.extractAll(node, context)[0] ?? null; + }, - const name = config.extractName(node); - if (!name) return null; + extractAll(node: SyntaxNode, context: VariableExtractorContext): VariableInfo[] { + if (!allNodeTypes.has(node.type)) return []; + + const names = config.extractNames + ? config.extractNames(node) + : [config.extractName(node)].filter((name): name is string => Boolean(name)); + if (names.length === 0) return []; - const type = config.extractType(node) ?? null; - const visibility = config.extractVisibility(node); // isConst/isStatic: node type membership is a hint, but config.isConst/isStatic // has final say. For languages where const and non-const share a node type // (e.g., TS lexical_declaration for both const and let), config.isConst disambiguates. @@ -92,17 +96,17 @@ export function createVariableExtractor(config: VariableExtractionConfig): Varia const isMutable = config.isMutable(node); const scope = determineScope(node); - return { + return names.map((name) => ({ name, - type, - visibility, + type: config.extractTypeForName?.(node, name) ?? config.extractType(node) ?? null, + visibility: config.extractVisibilityForName?.(node, name) ?? config.extractVisibility(node), isConst, isStatic, isMutable, scope, sourceFile: context.filePath, line: node.startPosition.row + 1, - }; + })); }, }; } diff --git a/gitnexus/src/core/ingestion/variable-types.ts b/gitnexus/src/core/ingestion/variable-types.ts index a2006080b..1bfef90aa 100644 --- a/gitnexus/src/core/ingestion/variable-types.ts +++ b/gitnexus/src/core/ingestion/variable-types.ts @@ -59,6 +59,9 @@ export interface VariableExtractor { /** Extract variable metadata from a declaration node. * Returns null if the node is not a recognized variable declaration. */ extract(node: SyntaxNode, context: VariableExtractorContext): VariableInfo | null; + /** Extract every variable metadata entry from a declaration node. + * Returns [] if the node is not a recognized variable declaration. */ + extractAll(node: SyntaxNode, context: VariableExtractorContext): VariableInfo[]; /** Check if a node is a recognized variable declaration type. */ isVariableDeclaration(node: SyntaxNode): boolean; } @@ -78,10 +81,16 @@ export interface VariableExtractionConfig { variableNodeTypes: string[]; /** Extract the variable name from a declaration node */ extractName: (node: SyntaxNode) => string | undefined; + /** Extract multiple variable names from a declaration node. */ + extractNames?: (node: SyntaxNode) => string[]; /** Extract type annotation from a declaration node */ extractType: (node: SyntaxNode) => string | undefined; + /** Extract type annotation for one variable name from a multi-name declaration. */ + extractTypeForName?: (node: SyntaxNode, name: string) => string | undefined; /** Extract visibility from a declaration node */ extractVisibility: (node: SyntaxNode) => VariableVisibility; + /** Extract visibility for one variable name from a multi-name declaration. */ + extractVisibilityForName?: (node: SyntaxNode, name: string) => VariableVisibility; /** Check if a declaration is const/immutable */ isConst: (node: SyntaxNode) => boolean; /** Check if a declaration is static */ diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index d1d63e832..b84777457 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -84,7 +84,7 @@ import { import type { NodeLabel, ParameterTypeClass } from 'gitnexus-shared'; import type { FieldInfo, FieldExtractorContext } from '../field-types.js'; import type { MethodInfo, MethodExtractorContext } from '../method-types.js'; -import type { VariableExtractorContext } from '../variable-types.js'; +import type { VariableExtractorContext, VariableInfo } from '../variable-types.js'; import { buildMethodProps, arityForIdFromInfo, @@ -1210,7 +1210,8 @@ const processFileGroup = ( // Track start indices of definition nodes already processed by higher-priority captures // (e.g. @definition.function) to avoid duplicate nodes when @definition.const/@definition.variable // patterns overlap with the same source range. - const processedDefinitionNodes = new Set(); + const processedDefinitionNodes = new Set(); + const variableInfoCache = new Map>(); for (const match of matches) { const captureMap: Record = {}; @@ -1651,20 +1652,6 @@ const processFileGroup = ( continue; } - // Dedup: variable captures (Const/Static/Variable) may overlap with higher-priority - // captures (e.g. `const fn = () => {}` matches both @definition.function and @definition.const). - // Skip variable captures whose definition node was already processed. - if ( - (nodeLabel === 'Const' || nodeLabel === 'Static' || nodeLabel === 'Variable') && - definitionNode && - processedDefinitionNodes.has(definitionNode.startIndex) - ) { - continue; - } - if (definitionNode) { - processedDefinitionNodes.add(definitionNode.startIndex); - } - const exportDefaultCall = nodeLabel === 'Function' && definitionNode?.type === 'export_statement' ? definitionNode.namedChildren.find((child) => child.type === 'call_expression') @@ -1706,6 +1693,25 @@ const processFileGroup = ( const nodeName = extractedClassSymbol?.name ?? defaultExportHocName ?? (nameNode ? nameNode.text : 'init'); + // Dedup: variable captures (Const/Static/Variable) may overlap with higher-priority + // captures (e.g. `const fn = () => {}` matches both @definition.function and @definition.const). + // Multi-name declarations share the same definition node, so include the emitted name. + if (definitionNode) { + const definitionBaseKey = `${definitionNode.startIndex}`; + if (nodeLabel === 'Const' || nodeLabel === 'Static' || nodeLabel === 'Variable') { + const definitionNameKey = `${definitionBaseKey}:${nodeName}`; + if ( + processedDefinitionNodes.has(definitionBaseKey) || + processedDefinitionNodes.has(definitionNameKey) + ) { + continue; + } + processedDefinitionNodes.add(definitionNameKey); + } else { + processedDefinitionNodes.add(definitionBaseKey); + } + } + const startLine = definitionNode ? definitionNode.startPosition.row + lineOffset : nameNode @@ -1980,11 +1986,20 @@ const processFileGroup = ( definitionNode && provider.variableExtractor ) { - const varCtx: VariableExtractorContext = { - filePath: file.path, - language, - }; - const varInfo = provider.variableExtractor.extract(definitionNode, varCtx); + let variableInfoByName = variableInfoCache.get(definitionNode.startIndex); + if (!variableInfoByName) { + const varCtx: VariableExtractorContext = { + filePath: file.path, + language, + }; + variableInfoByName = new Map( + provider.variableExtractor + .extractAll(definitionNode, varCtx) + .map((info) => [info.name, info]), + ); + variableInfoCache.set(definitionNode.startIndex, variableInfoByName); + } + const varInfo = variableInfoByName.get(nodeName); if (varInfo) { if (varInfo.type) declaredType = varInfo.type; methodProps.visibility = varInfo.visibility; diff --git a/gitnexus/test/integration/go-multi-name-sequential-metadata.test.ts b/gitnexus/test/integration/go-multi-name-sequential-metadata.test.ts new file mode 100644 index 000000000..1ddf16190 --- /dev/null +++ b/gitnexus/test/integration/go-multi-name-sequential-metadata.test.ts @@ -0,0 +1,83 @@ +import { describe, expect, it } from 'vitest'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import { createASTCache } from '../../src/core/ingestion/ast-cache.js'; +import { createSymbolTable } from '../../src/core/ingestion/model/index.js'; +import { processParsingSequential } from '../../src/core/ingestion/parsing-processor.js'; + +describe('Go multi-name declaration metadata in sequential parsing', () => { + it('enriches every const and var name when workers are skipped', async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'go-multi-name-seq-')); + fs.writeFileSync(path.join(dir, 'go.mod'), 'module example.com/multi\n\ngo 1.22\n'); + fs.writeFileSync( + path.join(dir, 'main.go'), + [ + 'package main', + '', + 'const X, Y int = 1, 2', + '', + 'var (', + ' a, b string', + ' c bool', + ')', + '', + 'func main() {}', + '', + ].join('\n'), + ); + + const graph = createKnowledgeGraph(); + const symbolTable = createSymbolTable(); + const astCache = createASTCache(); + const scopeTreeCache = createASTCache(); + await processParsingSequential( + graph, + [ + { + path: path.join(dir, 'main.go'), + content: fs.readFileSync(path.join(dir, 'main.go'), 'utf8'), + }, + ], + symbolTable, + astCache, + scopeTreeCache, + ); + + const metadata = new Map>(); + graph.forEachNode((node) => { + if (node.label === 'Const' || node.label === 'Variable') { + metadata.set(node.properties.name, node.properties); + } + }); + + expect(metadata.get('X')).toMatchObject({ + declaredType: 'int', + isConst: true, + isMutable: false, + scope: 'module', + }); + expect(metadata.get('Y')).toMatchObject({ + declaredType: 'int', + isConst: true, + isMutable: false, + scope: 'module', + }); + expect(metadata.get('a')).toMatchObject({ + declaredType: 'string', + isMutable: true, + scope: 'module', + }); + expect(metadata.get('b')).toMatchObject({ + declaredType: 'string', + isMutable: true, + scope: 'module', + }); + expect(metadata.get('c')).toMatchObject({ + declaredType: 'bool', + isMutable: true, + scope: 'module', + }); + }); +}); diff --git a/gitnexus/test/integration/go-multi-name-worker-metadata.test.ts b/gitnexus/test/integration/go-multi-name-worker-metadata.test.ts new file mode 100644 index 000000000..decef3271 --- /dev/null +++ b/gitnexus/test/integration/go-multi-name-worker-metadata.test.ts @@ -0,0 +1,97 @@ +import { describe, expect, it } from 'vitest'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { runPipelineFromRepo } from './resolvers/helpers.js'; +import type { PipelineResult } from '../../src/types/pipeline.js'; + +function createGoRepo(): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'go-multi-name-worker-')); + fs.writeFileSync(path.join(dir, 'go.mod'), 'module example.com/multi\n\ngo 1.22\n'); + fs.writeFileSync( + path.join(dir, 'main.go'), + [ + 'package main', + '', + 'const X, Y int = 1, 2', + 'var a, b string', + '', + 'var (', + ' c, d bool', + ')', + '', + 'type Point struct { Px, py int }', + '', + 'func main() {}', + '', + ].join('\n'), + ); + return dir; +} + +async function runMode(mode: 'worker' | 'sequential'): Promise { + return runPipelineFromRepo(createGoRepo(), () => {}, { + skipGraphPhases: true, + workerThresholdsForTest: { minFiles: 1, minBytes: 1 }, + ...(mode === 'worker' ? { workerPoolSize: 2 } : { skipWorkers: true }), + }); +} + +function collectMetadata(result: PipelineResult): Map> { + const metadata = new Map>(); + result.graph.forEachNode((node) => { + if (node.label === 'Const' || node.label === 'Variable' || node.label === 'Property') { + metadata.set(`${node.label}:${node.properties.name}`, node.properties); + } + }); + return metadata; +} + +describe('Go multi-name declaration metadata in worker parsing', () => { + it('emits every const and var name with metadata on the worker path', async () => { + const worker = await runMode('worker'); + const sequential = await runMode('sequential'); + + expect(worker.usedWorkerPool).toBe(true); + expect(sequential.usedWorkerPool).toBe(false); + + const workerMetadata = collectMetadata(worker); + const sequentialMetadata = collectMetadata(sequential); + + expect([...workerMetadata.keys()].sort()).toEqual([...sequentialMetadata.keys()].sort()); + expect(workerMetadata.get('Const:X')).toMatchObject({ + declaredType: 'int', + isConst: true, + isMutable: false, + scope: 'module', + }); + expect(workerMetadata.get('Const:Y')).toMatchObject({ + declaredType: 'int', + isConst: true, + isMutable: false, + scope: 'module', + }); + expect(workerMetadata.get('Variable:a')).toMatchObject({ + declaredType: 'string', + isMutable: true, + scope: 'module', + }); + expect(workerMetadata.get('Variable:b')).toMatchObject({ + declaredType: 'string', + isMutable: true, + scope: 'module', + }); + expect(workerMetadata.get('Variable:c')).toMatchObject({ + declaredType: 'bool', + isMutable: true, + scope: 'module', + }); + expect(workerMetadata.get('Variable:d')).toMatchObject({ + declaredType: 'bool', + isMutable: true, + scope: 'module', + }); + expect(workerMetadata.get('Property:Px')).toMatchObject({ declaredType: 'int' }); + expect(workerMetadata.get('Property:py')).toMatchObject({ declaredType: 'int' }); + }); +}); diff --git a/gitnexus/test/integration/tree-sitter-languages.test.ts b/gitnexus/test/integration/tree-sitter-languages.test.ts index c8349b440..a17241466 100644 --- a/gitnexus/test/integration/tree-sitter-languages.test.ts +++ b/gitnexus/test/integration/tree-sitter-languages.test.ts @@ -128,6 +128,40 @@ describe('Tree-sitter multi-language parsing', () => { const defTypes = defs.map((d) => d.type); expect(defTypes).toContain('definition.function'); }); + + it('captures every name in multi-name const, var, and field declarations', async () => { + await loadLanguage(SupportedLanguages.Go); + const provider = getProvider(SupportedLanguages.Go); + const code = ` + package main + const X, Y,Z,A,B = 1, 2,3,4,5 + var a, b int + var ( + c, d string + ) + const ( + C, D = 3, 4 + ) + type Point struct { X, y int } + `; + const { matches } = parseAndQuery(parser, code, provider.treeSitterQueries); + const defs = extractDefinitions(matches); + + const constNames = defs + .filter((def) => def.type === 'definition.const') + .map((def) => def.name); + expect(constNames.sort()).toEqual(['A', 'B', 'C', 'D', 'X', 'Y', 'Z']); + + const variableNames = defs + .filter((def) => def.type === 'definition.variable') + .map((def) => def.name); + expect(variableNames.sort()).toEqual(['a', 'b', 'c', 'd']); + + const propertyNames = defs + .filter((def) => def.type === 'definition.property') + .map((def) => def.name); + expect(propertyNames.sort()).toEqual(['X', 'y']); + }); }); describe('C', () => { diff --git a/gitnexus/test/unit/field-extraction.test.ts b/gitnexus/test/unit/field-extraction.test.ts index 2795f5833..3032eeded 100644 --- a/gitnexus/test/unit/field-extraction.test.ts +++ b/gitnexus/test/unit/field-extraction.test.ts @@ -703,6 +703,48 @@ describe('GenericFieldExtractor — Go', () => { expect(goConfig.extractType(xNode)).toBe('float64'); }); + it('extracts every name from Go multi-name struct fields', () => { + const { typeDecl } = findTypeSpec(`type Point struct {\n\tX, y int\n}`); + + const result = extractor.extract(typeDecl, mockContext); + expect(result).not.toBeNull(); + expect(result!.fields).toHaveLength(2); + + const xField = result!.fields.find((field) => field.name === 'X'); + expect(xField).toBeDefined(); + expect(xField!.type).toBe('int'); + expect(xField!.visibility).toBe('public'); + + const yField = result!.fields.find((field) => field.name === 'y'); + expect(yField).toBeDefined(); + expect(yField!.type).toBe('int'); + expect(yField!.visibility).toBe('package'); + }); + + it('extracts Go multi-name struct fields when called with the runtime struct_type owner', () => { + const { typeDecl } = findTypeSpec(`type Point struct {\n\tX, y int\n}`); + const typeSpec = typeDecl.namedChild(0)!; + const structType = typeSpec.namedChild(1)!; + + expect(structType.type).toBe('struct_type'); + expect(extractor.isTypeDeclaration(structType)).toBe(true); + + const result = extractor.extract(structType, mockContext); + expect(result).not.toBeNull(); + expect(result!.ownerFqn).toBe('Point'); + expect(result!.fields).toHaveLength(2); + + const xField = result!.fields.find((field) => field.name === 'X'); + expect(xField).toBeDefined(); + expect(xField!.type).toBe('int'); + expect(xField!.visibility).toBe('public'); + + const yField = result!.fields.find((field) => field.name === 'y'); + expect(yField).toBeDefined(); + expect(yField!.type).toBe('int'); + expect(yField!.visibility).toBe('package'); + }); + it('reports isStatic and isReadonly as false for all fields', () => { const { typeDecl } = findTypeSpec(`type S struct {\n\tX int\n}`); const typeSpec = typeDecl.namedChild(0)!; diff --git a/gitnexus/test/unit/scope-resolution/go/go-captures-smoke.test.ts b/gitnexus/test/unit/scope-resolution/go/go-captures-smoke.test.ts index 98bf4eaf0..337b040a5 100644 --- a/gitnexus/test/unit/scope-resolution/go/go-captures-smoke.test.ts +++ b/gitnexus/test/unit/scope-resolution/go/go-captures-smoke.test.ts @@ -63,6 +63,38 @@ func main() { expect(tags).toContain('@reference.write'); }); + it('emits every name from multi-name const, var, and field declarations', () => { + const src = ` +package main + +const X, Y,Z,A,B = 1, 2,3,4,5 +var a, b int +var ( + c, d string +) +const ( + C, D = 3, 4 +) +type Point struct { X, y int } +`; + const matches = emitGoScopeCaptures(src, 'main.go'); + + const constNames = matches + .filter((m) => m['@declaration.const'] !== undefined) + .map((m) => m['@declaration.name']!.text); + expect(constNames.sort()).toEqual(['A', 'B', 'C', 'D', 'X', 'Y', 'Z']); + + const variableNames = matches + .filter((m) => m['@declaration.variable'] !== undefined) + .map((m) => m['@declaration.name']!.text); + expect(variableNames.sort()).toEqual(['a', 'b', 'c', 'd']); + + const fieldNames = matches + .filter((m) => m['@declaration.field'] !== undefined) + .map((m) => m['@declaration.name']!.text); + expect(fieldNames.sort()).toEqual(['X', 'y']); + }); + // ── Edge shapes the #1915 captured-node refactor reasons about but no // lang-resolution fixture exercises (issue #1848 follow-up U2). ── diff --git a/gitnexus/test/unit/variable-extraction.test.ts b/gitnexus/test/unit/variable-extraction.test.ts index d40993e2c..62a5eab05 100644 --- a/gitnexus/test/unit/variable-extraction.test.ts +++ b/gitnexus/test/unit/variable-extraction.test.ts @@ -267,6 +267,76 @@ describe('VariableExtractor — Go', () => { expect(info!.type).toBe('int'); }); + it('extracts every name from multi-name const declarations', () => { + parser.setLanguage(Go); + const tree = parser.parse('package main\nconst X, Y,Z,A,B = 1, 2,3,4,5'); + const constNode = tree.rootNode.namedChildren.find( + (child) => child.type === 'const_declaration', + ); + expect(constNode).toBeDefined(); + + const infos = extractor.extractAll(constNode!, ctx); + expect(infos.map((info) => info.name)).toEqual(['X', 'Y', 'Z', 'A', 'B']); + for (const info of infos) { + expect(info.isConst).toBe(true); + expect(info.isMutable).toBe(false); + expect(info.visibility).toBe('public'); + expect(info.type).toBeNull(); + } + }); + + it('extracts every name from multi-name var declarations with a shared type', () => { + parser.setLanguage(Go); + const tree = parser.parse('package main\nvar a, b int'); + const varNode = tree.rootNode.namedChildren.find((child) => child.type === 'var_declaration'); + expect(varNode).toBeDefined(); + + const infos = extractor.extractAll(varNode!, ctx); + expect(infos.map((info) => info.name)).toEqual(['a', 'b']); + for (const info of infos) { + expect(info.isConst).toBe(false); + expect(info.isMutable).toBe(true); + expect(info.visibility).toBe('package'); + expect(info.type).toBe('int'); + } + }); + + it('preserves per-spec types in grouped multi-name var declarations', () => { + parser.setLanguage(Go); + const tree = parser.parse('package main\nvar (\n a, b int\n c string\n)'); + const varNode = tree.rootNode.namedChildren.find((child) => child.type === 'var_declaration'); + expect(varNode).toBeDefined(); + + const infos = extractor.extractAll(varNode!, ctx); + expect(infos.map((info) => [info.name, info.type])).toEqual([ + ['a', 'int'], + ['b', 'int'], + ['c', 'string'], + ]); + + const cInfo = infos.find((info) => info.name === 'c'); + expect(cInfo).toBeDefined(); + expect(cInfo!.isConst).toBe(false); + expect(cInfo!.isMutable).toBe(true); + expect(cInfo!.visibility).toBe('package'); + }); + + it('matches parse-worker nodeName selection for grouped multi-name vars', () => { + parser.setLanguage(Go); + const tree = parser.parse('package main\nvar (\n a, b int\n c string\n)'); + const varNode = tree.rootNode.namedChildren.find((child) => child.type === 'var_declaration'); + expect(varNode).toBeDefined(); + + const selected = extractor.extractAll(varNode!, ctx).find((info) => info.name === 'c'); + expect(selected).toMatchObject({ + name: 'c', + type: 'string', + visibility: 'package', + isConst: false, + isMutable: true, + }); + }); + it('detects lowercase as package-private', () => { parser.setLanguage(Go); const tree = parser.parse('package main\nconst maxSize = 100'); From 95f87fc12a44adc21b81a17085eb23079a23f21b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sat, 6 Jun 2026 22:46:34 +0100 Subject: [PATCH 02/17] =?UTF-8?q?perf(ingestion):=20Linux-kernel-scale=20a?= =?UTF-8?q?nalysis=20=E2=80=94=20worker-pool=20parse=20+=20finalize=20O(n?= =?UTF-8?q?=C2=B2)=20+=20scope-resolution=20memory=20wall=20(#1983)=20(#20?= =?UTF-8?q?38)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ingestion): reduce parse-phase memory for huge repos (#1983) Stop retaining full parse-cache chunks in RAM alongside the merged graph, slim on-disk shards, defer worker ParsedFile emission for scope-resolver languages, and add GITNEXUS_DEBUG_HEAP probes for OOM diagnosis. Co-authored-by: Cursor * fix(ingestion): address #2038 tri-review findings (parse-phase memory) Resolves the confirmed review findings on PR #2038: - P1: thread exportedTypeMap through the sequential parse path (processParsingSequential) so a no-worker run over a partially-warm cache no longer silently drops the sequential-miss files' exported types. Cache hits made exportedTypeMap.size > 0, suppressing the end-of-loop buildExportedTypeMapFromGraph rebuild, but the sequential path never populated the map. Regression test added (fails on the pre-fix tree, passes after) plus a fully-sequential differential oracle. - P2: saveParseCache builds its on-disk index from hashes actually written/copied (writtenKeys), never a usedKeys hash whose shard write or copy was skipped — no more phantom index entries. - P2: add a unit test asserting SCOPE_RESOLUTION_LANGUAGES stays in sync with SCOPE_RESOLVERS (asymmetric drift would lose a language's ParsedFile). - Backfill cache coverage: loadParseCacheChunk missing/corrupt -> undefined, pruneCache onDiskKeys branch, slim preserves nodes, saveParseCache copy-evicted-shard round-trip. - Cleanups: single-source heap-probe gating via isDebugHeapEnabled(); hoist the per-chunk mkdir in persistParseCacheChunk behind a process-scoped Set; gate COBOL's unused worker-side ParsedFile extraction (graph nodes still come from cobolPhase) while keeping fileCount/progress unconditional. Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(ingestion): remove dead worker-side ParsedFile extraction After #2038 gated worker `ParsedFile` emission behind `!isScopeResolutionLanguage(language)`, and with all 16 SupportedLanguages registered in SCOPE_RESOLVERS, that gate was structurally always true — the worker already produced no ParsedFiles and scope-resolution re-extracts each file from source on the main thread (run.ts). Remove the now-dead machinery: - Drop both worker `extractParsedFile` call-sites (tree-sitter processFileGroup + the standalone-provider branch) and the `result.parsedFiles.push`. The standalone branch keeps fileCount/onFileProcessed per file. `result.parsedFiles` stays declared but empty (field removal deferred). - Remove the now-orphaned `scopeSourceKind` var + `ScopeCaptureSourceKind`/`extractParsedFile`/`isScopeResolutionLanguage` imports. - Delete the consumerless `migrated-languages.ts` (isScopeResolutionLanguage + SCOPE_RESOLUTION_LANGUAGES) and its drift-guard test — parse-worker was their only importer. Also improves AGENTS.md "shared ingestion code must not name languages" compliance. `extractParsedFile` and the scope-extractor-bridge stay (scope-resolution/run.ts + Vue resolver use them). Behavior-preserving: worker-sequential-parity passes before and after; tsc/eslint clean; no baseline/golden drift. Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(ingestion): worker-pool-only parsing; remove sequential parser (#1983) Completes the #1983 huge-repo parse-OOM effort by making the worker pool GitNexus's sole parse path. Parallel serialization (the perf core): workers serialize their ParsedFiles to a disk store in parallel and stream them back to scope-resolution, so the main thread no longer re-parses every file (the tree-sitter native-memory leak that caused the OOM). Adds chunk merge-pipelining + work-proportional chunk sizing so the pool stays saturated. Remove the sequential parser: `--workers 0`, `GITNEXUS_WORKER_POOL_SIZE=0`, and `skipWorkers` now hard-error (no silent degrade — #1741); the small-repo threshold no longer selects an in-process path; pool creation stays lazy / cache-miss-gated so warm all-hit runs never spawn workers. Worker-path parity fixes — removing sequential surfaced two pre-existing gaps that tiny-fixture tests had masked by running below the worker threshold, both fixed by carrying per-file metadata as DATA across the worker boundary (never re-parsing on the main thread, preserving the OOM fix): - C++: templateConstraints wired into worker node identity (SFINAE overload disambiguation) + ADL / inline-namespace capture side-channel serialized onto the ParsedFile. - Kotlin: companion-scope side-channel serialized the same way (companion / static dispatch). Validation: tsc + build clean; full suite green (10,190 pass — the only deterministic failures were the now-fixed C++/Kotlin worker-path gaps; the 2 remaining full-run failures are pre-existing load flakiness, green in isolation); cpp-pipeline benchmark stays linear on a 1-worker pool. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(ingestion): wire C static-linkage side-channel + ADL O(1) collect + tri-review cleanups (#1983) Follow-up to the worker-pool-only refactor, from a tri-review of the parse path. - C static-linkage side-channel (P1): cProvider had no collect/applyCaptureSideChannel, so on the now-sole worker path C `static` file-local marks were lost across the worker boundary -> false cross-file CALLS edges + over-broad #include wildcard visibility on every C analysis (the Linux kernel is C). Mirror the C++/Kotlin wiring: serialize `staticNames` per file onto ParsedFile.captureSideChannel and restore it on the main thread (no re-parse). + a worker-path regression test (the existing c-static-isolation fixture passed vacuously — its collision resolves via #include before the global free-call fallback ever consults static-linkage). - captureSideChannel `kind` discriminant: add `kind:'cpp'`/`kind:'c'` tags + guards (Kotlin already had one) now that C/C++/Kotlin share the single generic field. - Perf: collectCppAdlSideChannel scanned the whole argInfoBySite/noAdlSites maps per file (O(F^2) per sub-batch, ~100M parseSiteKey calls at kernel scale). Add per-filePath lockstep indexes -> O(1) collect; serialized snapshot byte-identical. - Cleanups: inline the one-line processParsingWithWorkers wrapper into processParsing; drop the always-empty WorkerExtractedData.calls/assignments/constructorBindings fields; remove the voided astCache param from processParsing; refresh stale "sequential fallback" JSDoc. Validation: tsc + build clean; cpp 297/297, c 8/8 (incl. the new worker-path static-linkage guard), typescript + parsedfile-store green; cpp ADL benchmark stays linear. Co-Authored-By: Claude Opus 4.8 (1M context) * perf(scope-resolution): index C/C++ #include resolution in finalize (O(n²)→O(n)) Kernel-scale C/C++ analysis ground in finalizeScopeModel because three per-#include operations each did a full O(F) scan with no index — the finalize O(n²) that surfaced once the #1983 parse-phase OOM was fixed: - expand{C,Cpp}WildcardNames: parsedFiles.find() per wildcard edge → O(R·F) - resolveImportTarget: new Set(allFilePaths) rebuilt per #include - resolveCImportTarget: suffix-match scanned all workspace paths Each is replaced with a WeakMap-per-pass index keyed on the stable parsedFiles/allFilePaths references that scope-resolution run.ts passes once per pass: - Map for wildcard expansion (c/static-linkage.ts + cpp/file-local-linkage.ts) - memoized augmented header set (c/scope-resolver.ts + cpp/scope-resolver.ts) - basename-bucketed suffix index in resolveCImportTarget (c/import-target.ts), shared by C and C++ since resolveCppImportTarget delegates to it Collapses the C/C++ finalize from O(R·F) to O(R+F). Pure-perf, byte-identical edge output: 962 targeted tests green (490 C + 472 C/C++ scope-resolution); the basename index preserves the exact endsWith('/'+target) match and the fewest-path-components-then-lexicographic tie-break. The kernel's ~25-30k .h headers are classified C++, so both providers must be fixed. Proven on the Linux kernel: the C finalize completed (sr-post-finalize lang=c → sr-end lang=c), which the pre-fix run never reached in 16+ min of grinding. Build-independent follow-ups (separate from this finalize fix), documented for later: emitFreeCallFallback same-name buckets (emit phase), buildGraphNodeLookup + precount global setup, the ParsedFile store-load, the dart/go/ruby expand-wildcards .find siblings, and the ~26GB scope-resolution memory floor (full kernel completion needs >~40GB RAM). Co-Authored-By: Claude Opus 4.8 (1M context) * test(bench): regenerate C scope-capture baseline for the #1983 c-static-linkage-worker fixture bench/scope-capture/measure.mjs fingerprints emitCScopeCaptures over the lang-resolution/c-* fixture corpus. The #1983 PR added the c-static-linkage-worker fixture (caller.c/lib.c/lib.h/local.c — the worker-path static-linkage side-channel test) but did not regenerate the C baseline, so `--check` has been red on this branch (main, lacking the fixture, still matches 0de009b). Pure fixture-corpus drift — no c/captures.ts or query change branch-vs-main, existing fixtures' captures byte-identical (c-captures.test.ts 45/45), scaling stays linear (~0.97). Regenerated: 0de009b -> 39f3a83. Bench now PASS (14 languages). Unrelated to the finalize O(n²) fix. Co-Authored-By: Claude Opus 4.8 (1M context) * perf(scope-resolution): lower kernel-scale resident memory floor + setup cost Reduce the scope-resolution resident-memory floor and setup throughput on huge repos (Linux kernel), the wall that remains after #1983 (parse OOM) and the finalize O(n^2) fix (b71c77b8). Five units; all preserve byte-identical edge output (C fixture 177n/255e + c/cpp/cross-file/php/static-linkage suites green, 619 tests). U1 (src/cli/analyze.ts): RAM-aware auto heap-cap. Replace the hardcoded 16384MB cap with computeHeapCapMb = max(16384, floor(0.75*effectiveRAM)), where effectiveRAM = min(os.totalmem(), process.constrainedMemory()) with the unconstrained-sentinel guard. Add --max-semi-space-size=128 on the respawn. A user-supplied NODE_OPTIONS heap still wins (no re-exec). Verified: 23973MB on a 31964MB box, 16384 floor on small machines, cgroup-aware, sentinel safe. U2 (src/storage/parsedfile-store.ts, .../pipeline/phase.ts): export forceGc() and call it at the per-language eviction boundary, so a finished language's ParsedFiles are reclaimed before the next language's store-load instead of collected lazily under the next pass's allocation pressure (which at cap>=RAM degrades into swap-thrash). Measured on a real drivers/net/ethernet run: C 2113->894MB and C++ 1754->1057MB reclaimed at the boundary (no fragmentation defeat). Answers the plan's Open Question 1. U3 (src/storage/parsedfile-store.ts): intern def objects by nodeId in the load reviver so a SymbolDefinition's three serialized copies (localDefs / scope.ownedDefs / scope.bindings[].def) collapse to one shared object on load. Per-shard def pool (a def's copies are shard-local). Measured ~42% off the def-object retained heap (3->1; 1.8M->600k distinct objects on 600k defs). U4 (.../passes/free-call-fallback.ts): memoize pickUniqueGlobalCallable's post-filter candidate list per (name, callerFilePath), only when no per-caller visibility filter applies (the list is then a pure function of name+file), so repeated free calls of one name from a file reuse the same-name-bucket scan instead of re-walking a potentially huge bucket per site. The cached array is read-only-consumed by the .filter()-based arity/overload narrowers. Exported pickUniqueGlobalCallable + buildGlobalCallableIndex and added an equivalence test (memoized == un-memoized reference for every (name, file, arity), including warm-cache repeats and cross-file file-local exclusion). U5 (.../pipeline/phase.ts): replace the O(L*F) per-language precount + repeated scannedFiles.filter() with a single O(F) partition-by-language pass; bracket buildGraphNodeLookup with scope-setup-nodeLookup heap probes so the long setup is no longer silent. Plan: docs/plans/2026-06-06-001-perf-kernel-scope-resolution-memory-plan.md (U6 out-of-core global index deferred). Note: the kernel's full C++ pass floor (~20k headers + the 8.8GB graph) likely still exceeds 24GB by itself, which is why U6 remains the only unit that clears the wall. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(test): match OOM-guidance e2e assertions to the U1 reworded hint The analyze-heap-oom-e2e real-child-OOM test still asserted the pre-U1 wording ('...out of memory.' + a hardcoded 24576 cap). U1 reworded the hint to mention the auto heap-cap and use a placeholder, so the three toContain substrings no longer matched (the assertion at line 62 failed on all platforms). Update them to the current message. The unit twin (analyze-heap-respawn) was already updated in 85bfc216; this integration test was missed by the targeted local run. Co-Authored-By: Claude Opus 4.8 (1M context) * perf(lbug): U6a — deterministic id-sorted graph output behind GITNEXUS_SORT_GRAPH_OUTPUT First increment of U6 (out-of-core scope-resolution). Adds an optional deterministic ordering of node + relationship CSV rows by their unique graph id, behind GITNEXUS_SORT_GRAPH_OUTPUT (default OFF = today's graph-insertion order, byte-identical — the iterator is returned untouched). With the flag ON the CSV becomes a pure function of the node/edge SET rather than of emit order. This is the structural enabler for the windowed/out-of-core resolve (U6b-U6d): csv-generator.ts:518 currently iterates graph.iterRelationships() in insertion order with NO terminal sort, so any deviation from parsedFiles-order emit would change bytes. With U6a on, a windowed emit need only reproduce the same edge SET, not the global insertion order — removing the single largest byte-identical hazard from every later windowing step. Verified: default off keeps the existing csv-pipeline suite byte-identical; on, node rows are id-sorted and output is independent of graph insertion order (set-build) with the same node/edge set. Co-Authored-By: Claude Opus 4.8 (1M context) * perf(storage): U6d foundation — disk-backed scope store + lazy ScopeTree Adds scope-index-store.ts: persistScopeShards (per-file scope shards via the proven mapReplacer + def-interning reviver) + DiskBackedScopeTree, a lazy ScopeTree that serves getScope from a bounded LRU of decoded shards plus a small resident skeleton (scopeId -> {shard, childIds, parent}). Exports makeInterningReviver from parsedfile-store for reuse. This is the contained, highest-risk mechanism of U6d (out-of-core scope resolution): the emit passes reach the heavy per-Scope binding payload (~17-20GB on the kernel) ONLY through scopeTree.getScope (a point lookup) and getChildren — they never read parsed.scopes directly — so moving that payload to disk behind getScope is transparent. Every consumer reads a Scope BY VALUE, so a value-faithful disk round-trip is byte-identical to resolution. Proven in isolation: DiskBackedScopeTree is value-identical to buildScopeTree for getScope/getChildren/getParent/getAncestors/has/size across multiple files and after LRU eviction, and preserves the def-identity collapse (ownedDefs[i] === binding.def). Nothing wires it yet (the resolution-pipeline integration is the next increment) — zero production impact; default off. Co-Authored-By: Claude Opus 4.8 (1M context) * perf(scope-resolution): U6d integration — seal scopeTree to disk before emit (GITNEXUS_DISK_SCOPE_INDEX) Wires the U6d out-of-core scope index into the live pipeline behind GITNEXUS_DISK_SCOPE_INDEX (default OFF = byte-identical). When on: - finalize-orchestrator builds a TransitionalScopeTree (validated, fully resident) instead of buildScopeTree, so finalize/propagate/resolve are unchanged. - After resolve, before emit, run.ts seals it: persists the scopes to a file-sharded scope-index-store, swaps the model's scopeTree to disk-backed serving from the inside (the frozen bundle can't be reassigned, but the wrapper nulls its own resident backing), and drops the heavy Scope.bindings payload from all THREE holders — the model's tree (seal), the caller's preExtractedParsedFiles, and run.ts's own parsedFiles (scope-stripped copies for emit). Emit reads scopes only via scopeTree.getScope (a point lookup, now disk-backed + LRU) — verified it never reads parsed.scopes. Purpose: lower the per-language resident PEAK (kernel C pass ~20→~12 GB by moving the ~8-9 GB scope payload to disk) so the analysis fits on smaller-RAM machines. At >=24 GB the full kernel already fits with U1-U5 (U2's 8.7 GB inter-language forceGc reclaim keeps each pass under cap) — empirically confirmed — so this is the sub-24 GB lever, not needed at 24 GB. Byte-identical evidence: DiskBackedScopeTree/TransitionalScopeTree return value-identical scopes vs buildScopeTree (getScope/getChildren/getParent/ getAncestors, across files + after LRU eviction + post-seal); emit reads only getScope + referenceSites; flag-off (394 tests) and flag-on-resident (91 tests) resolver suites stay green; an end-to-end A/B on a 212-file C+cpp+rust subset produced identical 17,444 nodes / 31,343 edges with the seal firing per language (c: 410→141 MB reclaimed). Kernel-scale peak-drop measurement pending the in-flight verdict run freeing memory. Co-Authored-By: Claude Opus 4.8 (1M context) * perf(scope-resolution): U6d — id-back workspaceIndex so the disk seal can reclaim scopes The kernel run revealed the contained scopeTree seal didn't lower the heap: WorkspaceResolutionIndex held Scope OBJECTS (classScopeByDefId / moduleScopeByFile), built from every ParsedFile and live through emit, so the ~28k module + class scopes stayed pinned past the seal (sr-seal-pre 17,583 -> sr-seal-post 17,771 MB, no drop). It was the sole residual Scope-object holder (SemanticModel holds none). Fix: classScopeByDefId / moduleScopeByFile become id-backed ScopeByKeyView instances — a ReadonlyMap facade over a K->ScopeId map + the scopeTree, whose .get fetches via scopeTree.getScope(id). The index now pins only ids, so once the tree seals to disk the scopes become collectible. Byte-identical: the view returns the same Scope the resident tree holds (or a value-identical revived one in disk mode), and iteration keeps the old insertion order. buildWorkspace ResolutionIndex takes an optional scopeTree (live pipeline passes it); without it (unit tests) the legacy direct Scope-object maps are returned unchanged. Verified byte-identical: 733 tests across workspace-index / imported-return-types / c / cpp / cross-file / go / java. Kernel peak-drop re-measurement to follow. Co-Authored-By: Claude Opus 4.8 (1M context) * perf(scope-resolution): U6d — precompute exportedCallableByName (fix disk-getScope thrash) The workspaceIndex id-backing freed the kernel scopes but exposed a throughput collapse: findExportedDefByName's workspace fallback (walkers.ts:1019) scanned EVERY module scope's bindings per unresolved free call, and under the U6d disk-backed scopeTree each module-scope access faulted a shard in from disk — lib ON went ~1min -> ~7.5min. Fix: precompute the fallback result once into WorkspaceResolutionIndex.exportedCallableByName (simpleName -> first module-local callable def, first-file-wins — the exact semantics the scan returned), built from the resident module-scope bindings at index-build time. findExportedDefByName now does an O(1) lookup with zero disk reads. Result: lib ON ~7.5min -> 21s (cache-warm), byte-identical 17,444/31,343; 758 tests green across workspace-index + c/cpp/cross-file/go/python. Co-Authored-By: Claude Opus 4.8 (1M context) * docs: rename cryptic U-unit codes to descriptive names in comments The plan-unit shorthand (U3/U4/U6a/U6d/...) was meaningless in the code. Renamed in comments + test descriptions (no behavior change, byte-identical): out-of-core scope index (was U6) deterministic output (was U6a) disk-backed scope seal (was U6d) def-object interning (was U3) free-call candidate cache (was U4) Also renamed throughout the PR title/summary. Pushed commit messages keep their original U-codes as historical record. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(ingestion): durable ParsedFile shards for warm-cache coverage (#2038) On a warm re-analyze where every chunk is a parse-cache HIT, no parse worker runs, the run-scoped ParsedFile store is cleared at parse start, and the cached ParseWorkerResult carries no ParsedFiles (the worker writes them to the store and empties them from the message). Scope-resolution then found an empty store and fell back to main-thread extractParsedFile — re-opening the #1983 tree-sitter native-leak OOM the disk store closes (abhigyanpatwari review on parse-cache.ts). Fix: workers ALSO write their ParsedFiles to a durable, content-addressed store (parsedfile-cache/) keyed by chunk hash, mirroring the parse cache's lifecycle (version-gated by PARSE_CACHE_VERSION, pruned in lockstep to the surviving keys). On a warm hit the chunk's durable shards are byte-COPIED into the run-scoped store (no re-parse, no re-serialize -> byte-identical), so scope-resolution streams them exactly as on a cold run. A coherence gate re-dispatches the worker whenever a cached chunk's durable shards are missing (migration / pruned / version-stale) -- never the main-thread extract. - worker-pool/parse-worker: thread chunkHash through dispatch->job->flush (incl. split/requeue) so the worker tags its durable shard by content - parsedfile-store: durable persist / restore / index / prune API (sibling dir, never cleared per run); content-addressing makes stale reuse impossible - parse-impl: load durable index, gate the cache hit on durable coverage, restore on hit, dispatch chunkHash on miss - run-analyze: prune+save the durable store to the parse cache's surviving keys - saveParseCache returns its written keys (the durable keepKeys) Verified on linux/lib: warm preExtractedHits = full coverage (520/207/1, zero main-thread re-parse), byte-identical cold==warm (17,456n/31,353e), warm 8.5x faster. New two-run + mixed-mode + coherence-gate regression test. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(ingestion): clear stale scope-index-store shards on each seal (#2038) The disk-backed scope index writes sequential s.json shards into a shared /scope-index-store/ dir, with the index resetting per persistScopeShards call. A seal that writes fewer shards than a previous one (a later language with fewer files, or a re-run of a shrunken repo) left stale tail shards on disk indefinitely -- never read by the disk-backed tree, but multi-GB on kernel-scale repos. Add clearScopeIndexStore() and clear at the start of persistScopeShards: the previously sealed language has finished emit and been released before the next seal runs, so its DiskBackedScopeTree never reads those shards again. Unit tests: a stale prior-run shard is removed, a fewer-files re-seal leaves no tail shards, and the helper is idempotent. Addresses abhigyanpatwari review on run.ts (disk hygiene for the GITNEXUS_DISK_SCOPE_INDEX path). Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: Cursor Co-authored-by: Claude Opus 4.8 (1M context) --- .../src/scope-resolution/parsed-file.ts | 24 + gitnexus/README.md | 2 +- gitnexus/bench/scope-capture/baselines.json | 5 +- gitnexus/src/cli/analyze.ts | 68 +- gitnexus/src/cli/i18n/en.ts | 2 +- gitnexus/src/cli/i18n/zh-CN.ts | 2 +- gitnexus/src/cli/index.ts | 2 +- gitnexus/src/core/ingestion/call-processor.ts | 52 +- .../core/ingestion/finalize-orchestrator.ts | 10 +- .../src/core/ingestion/language-provider.ts | 21 + .../src/core/ingestion/languages/c-cpp.ts | 16 + .../languages/c/capture-side-channel.ts | 80 ++ .../ingestion/languages/c/import-target.ts | 77 +- .../src/core/ingestion/languages/c/index.ts | 5 + .../ingestion/languages/c/scope-resolver.ts | 62 +- .../ingestion/languages/c/static-linkage.ts | 49 +- .../src/core/ingestion/languages/cpp/adl.ts | 122 +- .../languages/cpp/capture-side-channel.ts | 111 ++ .../languages/cpp/file-local-linkage.ts | 69 +- .../src/core/ingestion/languages/cpp/index.ts | 4 + .../languages/cpp/inline-namespaces.ts | 24 + .../ingestion/languages/cpp/scope-resolver.ts | 61 +- .../languages/cpp/two-phase-lookup.ts | 50 + .../ingestion/languages/csharp/captures.ts | 9 +- .../languages/csharp/namespace-siblings.ts | 5 +- .../src/core/ingestion/languages/kotlin.ts | 8 + .../languages/kotlin/capture-side-channel.ts | 75 ++ .../languages/kotlin/companion-scopes.ts | 11 + .../core/ingestion/languages/kotlin/index.ts | 5 + .../languages/kotlin/scope-resolver.ts | 18 + .../languages/typescript/captures.ts | 9 +- .../src/core/ingestion/parsing-processor.ts | 1043 ++--------------- .../ingestion/pipeline-phases/parse-impl.ts | 819 +++++++------ .../core/ingestion/pipeline-phases/parse.ts | 39 +- gitnexus/src/core/ingestion/pipeline.ts | 31 +- .../contract/scope-resolver.ts | 39 + .../passes/free-call-fallback.ts | 82 +- .../scope-resolution/pipeline/phase.ts | 145 ++- .../scope-resolution/pipeline/run.ts | 121 +- .../scope-resolution/scope/walkers.ts | 28 +- .../scope-resolution/workspace-index.ts | 109 +- .../src/core/ingestion/utils/heap-probe.ts | 43 + .../core/ingestion/workers/parse-worker.ts | 175 ++- .../src/core/ingestion/workers/worker-pool.ts | 103 +- gitnexus/src/core/lbug/csv-generator.ts | 33 +- gitnexus/src/core/run-analyze.ts | 29 +- gitnexus/src/storage/parse-cache.ts | 197 +++- gitnexus/src/storage/parsedfile-store.ts | 420 +++++++ gitnexus/src/storage/scope-index-store.ts | 274 +++++ .../c-static-linkage-worker/caller.c | 16 + .../c-static-linkage-worker/lib.c | 7 + .../c-static-linkage-worker/lib.h | 7 + .../c-static-linkage-worker/local.c | 13 + gitnexus/test/helpers/worker-parse.ts | 77 ++ .../integration/analyze-heap-oom-e2e.test.ts | 6 +- .../c-cpp-typedef-legacy-parse.test.ts | 15 +- .../integration/cpp-adl-benchmark.test.ts | 13 +- .../cpp-pipeline-benchmark.test.ts | 10 +- .../test/integration/csv-pipeline.test.ts | 87 +- .../fastapi-prefix-pipeline.test.ts | 4 +- .../go-multi-name-sequential-metadata.test.ts | 83 -- .../go-multi-name-worker-metadata.test.ts | 26 +- .../integration/ignore-and-skip-e2e.test.ts | 16 +- .../object-literal-method-exports.test.ts | 47 + .../object-literal-owner-resolution.test.ts | 179 ++- .../parse-impl-chunk-concurrency.test.ts | 6 +- .../parse-impl-env-reads.test.ts | 6 +- .../parse-impl-large-fixture.test.ts | 20 +- .../parse-impl-progress-monotonic.test.ts | 17 +- .../parse-impl-quarantine-cache-skip.test.ts | 3 - .../php-pipeline-benchmark.test.ts | 2 +- .../qualified-class-lookups.test.ts | 73 +- gitnexus/test/integration/resolvers/c.test.ts | 59 + .../test/integration/resolvers/cpp.test.ts | 8 +- .../test/integration/resolvers/csharp.test.ts | 4 +- .../test/integration/resolvers/go.test.ts | 2 +- .../test/integration/resolvers/java.test.ts | 10 +- .../laravel-route-resolution.test.ts | 8 +- .../test/integration/resolvers/python.test.ts | 4 +- .../test/integration/resolvers/ruby.test.ts | 11 +- .../test/integration/resolvers/rust.test.ts | 4 +- .../integration/resolvers/typescript.test.ts | 4 +- .../ruby-pipeline-benchmark.test.ts | 2 +- .../worker-sequential-parity.test.ts | 79 -- .../test/unit/analyze-heap-respawn.test.ts | 71 +- .../unit/analyze-worker-pool-size.test.ts | 17 +- .../c-static-linkage-side-channel.test.ts | 97 ++ .../test/unit/incremental-parse-cache.test.ts | 156 ++- .../unit/language-availability-skip.test.ts | 89 ++ ...mpl-warm-cache-parsedfile-coverage.test.ts | 364 ++++++ .../unit/parse-impl-worker-lazy-cache.test.ts | 44 +- .../parse-impl-worker-startup-gating.test.ts | 24 +- gitnexus/test/unit/parsedfile-store.test.ts | 373 ++++++ .../test/unit/parsing-worker-fallback.test.ts | 70 +- gitnexus/test/unit/scope-index-store.test.ts | 208 ++++ .../pick-unique-global-callable.test.ts | 220 ++++ .../sequential-language-availability.test.ts | 86 -- 97 files changed, 5304 insertions(+), 2231 deletions(-) create mode 100644 gitnexus/src/core/ingestion/languages/c/capture-side-channel.ts create mode 100644 gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts create mode 100644 gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts create mode 100644 gitnexus/src/core/ingestion/utils/heap-probe.ts create mode 100644 gitnexus/src/storage/parsedfile-store.ts create mode 100644 gitnexus/src/storage/scope-index-store.ts create mode 100644 gitnexus/test/fixtures/lang-resolution/c-static-linkage-worker/caller.c create mode 100644 gitnexus/test/fixtures/lang-resolution/c-static-linkage-worker/lib.c create mode 100644 gitnexus/test/fixtures/lang-resolution/c-static-linkage-worker/lib.h create mode 100644 gitnexus/test/fixtures/lang-resolution/c-static-linkage-worker/local.c create mode 100644 gitnexus/test/helpers/worker-parse.ts delete mode 100644 gitnexus/test/integration/go-multi-name-sequential-metadata.test.ts create mode 100644 gitnexus/test/integration/object-literal-method-exports.test.ts rename gitnexus/test/{unit => integration}/parse-impl-chunk-concurrency.test.ts (97%) rename gitnexus/test/{unit => integration}/parse-impl-env-reads.test.ts (95%) rename gitnexus/test/{unit => integration}/parse-impl-progress-monotonic.test.ts (88%) delete mode 100644 gitnexus/test/integration/worker-sequential-parity.test.ts create mode 100644 gitnexus/test/unit/c-static-linkage-side-channel.test.ts create mode 100644 gitnexus/test/unit/language-availability-skip.test.ts create mode 100644 gitnexus/test/unit/parse-impl-warm-cache-parsedfile-coverage.test.ts create mode 100644 gitnexus/test/unit/parsedfile-store.test.ts create mode 100644 gitnexus/test/unit/scope-index-store.test.ts create mode 100644 gitnexus/test/unit/scope-resolution/pick-unique-global-callable.test.ts delete mode 100644 gitnexus/test/unit/sequential-language-availability.test.ts diff --git a/gitnexus-shared/src/scope-resolution/parsed-file.ts b/gitnexus-shared/src/scope-resolution/parsed-file.ts index 50eb5a795..98a327ad4 100644 --- a/gitnexus-shared/src/scope-resolution/parsed-file.ts +++ b/gitnexus-shared/src/scope-resolution/parsed-file.ts @@ -74,4 +74,28 @@ export interface ParsedFile { */ readonly localDefs: readonly SymbolDefinition[]; readonly referenceSites: readonly ReferenceSite[]; + /** + * Opaque, language-private serialization of capture-time side-channel + * state that a provider's `emitScopeCaptures` populates into module-level + * maps as a SIDE EFFECT (not onto the scopes/defs of this `ParsedFile`). + * + * Such state is computed inside the parse worker (where `emitScopeCaptures` + * runs) and would otherwise be lost across the worker→main MessageChannel + * and the disk store, because scope-resolution reuses the serialized + * `ParsedFile` and SKIPS re-extraction on the main thread (#1983 — the + * whole point is to avoid a main-thread tree-sitter re-parse). Carrying the + * data here lets the main thread repopulate those maps WITHOUT re-parsing. + * + * Shared / ingestion code treats this as opaque (`unknown`) per AGENTS.md + * (no language names in shared code). The producing language fills it via + * the `LanguageProvider.collectCaptureSideChannel` hook (worker side) and + * consumes it via the `ScopeResolver.applyCaptureSideChannel` hook + * (main-thread resolution side). It MUST be plain JSON-serializable data + * (objects / arrays / primitives) so it round-trips through the disk-backed + * `parsedfile-store` (JSON.stringify + interning reviver). + * + * Optional: providers whose `emitScopeCaptures` is pure (no module-level + * side effects — the contract default) leave this undefined. + */ + readonly captureSideChannel?: unknown; } diff --git a/gitnexus/README.md b/gitnexus/README.md index 7544508ba..54c0f7c81 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -400,7 +400,7 @@ Values above **32768 KB (32 MB)** are clamped to the tree-sitter parser ceiling; ### Analyze reports a worker timeout -Worker parse timeouts are recoverable. GitNexus retries stalled worker jobs with backoff, splits large jobs to isolate slow files, and falls back to the sequential parser when needed. If a large repository needs more time per worker job, use either: +Worker parse timeouts are recoverable. GitNexus retries stalled worker jobs with backoff, splits large jobs to isolate slow files, and quarantines a file that repeatedly crashes its worker (respawning the slot so the pool keeps going). If a large repository needs more time per worker job, use either: ```bash # CLI flag, in seconds diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index 225238dd8..3c3e4e0c2 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -11,9 +11,10 @@ "_note": "Updated for F17-F23 fixes (P2: TIMES guard, ADD GIVING, SQL AS alias). See PR #1959." }, "c": { - "fingerprint": "0de009bdbfe095f530fa87eb32bce6ab83092c904f26b3c8fe8d8ab587cf6dc9", + "fingerprint": "39f3a8346bb58159e9d79e2db6e1dd34ec6d70028e507ffd39548667dc658aa2", "scaling_budget": 1.5, - "_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance \u2014 flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96." + "_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance \u2014 flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96.", + "_note": "#1983: + c-static-linkage-worker fixture (caller.c/lib.c/lib.h/local.c \u2014 worker-path static-linkage side-channel test). Pure fixture-corpus drift: no c/captures.ts or query change branch-vs-main, existing fixtures' captures byte-identical (c-captures.test.ts 45/45), scaling stays linear (~0.97). The baseline was missed when the fixture landed; regenerated here. fingerprint 0de009b->39f3a83." }, "cpp": { "fingerprint": "6d6207ae1df3943c5fae28983e0c294e55225456e7cf39af1d46fda21b6787c4", diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index 6801f3f77..70bc9a566 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -9,6 +9,7 @@ */ import path from 'path'; +import os from 'os'; import { spawn } from 'child_process'; import v8 from 'v8'; import cliProgress from 'cli-progress'; @@ -84,13 +85,45 @@ const installFatalHandlers = (): void => { }); }; -const HEAP_MB = 16384; +/** Historical floor for the re-exec heap cap — the auto-sizer never goes below + * this, so small boxes / CI never regress. */ +const DEFAULT_HEAP_MB = 16384; + +/** + * RAM-aware re-exec heap cap (MB): `0.75 × effective RAM`, clamped to + * `>= DEFAULT_HEAP_MB`. Kept BELOW physical RAM on purpose — a cap `>=` RAM makes + * V8 collect lazily and inflate the heap into swap-thrash (observed analyzing the + * Linux kernel at a 30GB cap on a 31GB box). `constrainedBytes` is the cgroup + * limit or `null`; it is honored only as a real, smaller-than-physical cap, because + * `process.constrainedMemory()` returns a huge sentinel when UNCONSTRAINED. + */ +export function computeHeapCapMb(totalBytes: number, constrainedBytes: number | null): number { + const effectiveBytes = + constrainedBytes !== null && constrainedBytes > 0 && constrainedBytes < totalBytes + ? constrainedBytes + : totalBytes; + const effectiveMb = Math.floor(effectiveBytes / (1024 * 1024)); + return Math.max(DEFAULT_HEAP_MB, Math.floor(0.75 * effectiveMb)); +} + +function readConstrainedBytes(): number | null { + if (typeof process.constrainedMemory !== 'function') return null; + const c = process.constrainedMemory(); + return typeof c === 'number' && c > 0 ? c : null; +} + +const HEAP_MB = computeHeapCapMb(os.totalmem(), readConstrainedBytes()); const TEST_RESPAWN_HEAP_MB = Number(process.env.GITNEXUS_TEST_RESPAWN_HEAP_MB); const RESPAWN_HEAP_MB = Number.isFinite(TEST_RESPAWN_HEAP_MB) && TEST_RESPAWN_HEAP_MB > 0 ? Math.floor(TEST_RESPAWN_HEAP_MB) : HEAP_MB; const HEAP_FLAG = `--max-old-space-size=${RESPAWN_HEAP_MB}`; +/** Larger semi-space (young-gen) cuts minor-GC frequency + promotion churn during + * the multi-million-node graph build/emit. Allowed in NODE_OPTIONS (unlike + * --stack-size), so it propagates to the re-exec env cleanly. */ +const SEMI_SPACE_MB = 128; +const SEMI_FLAG = `--max-semi-space-size=${SEMI_SPACE_MB}`; /** Increase default stack size (KB) to prevent stack overflow on deep class hierarchies. */ const STACK_KB = 4096; const STACK_FLAG = `--stack-size=${STACK_KB}`; @@ -440,7 +473,8 @@ const forceHeapOOMForTestIfEnabled = (): void => { // `gitnexus/src/core/lbug/lbug-config.ts` in sync with this value. const RECOMMENDED_WAL_CHECKPOINT_THRESHOLD = 64 * 1024 * 1024; -/** Re-exec the process with a 16GB heap and larger stack if we're currently below that. */ +/** Re-exec the process with the RAM-aware auto heap cap + larger semi-space/stack + * if we're currently below that. A user-supplied NODE_OPTIONS heap wins (no re-exec). */ async function ensureHeap(): Promise { const nodeOpts = process.env.NODE_OPTIONS || ''; if (nodeOpts.includes('--max-old-space-size')) return false; @@ -448,25 +482,26 @@ async function ensureHeap(): Promise { const v8Heap = v8.getHeapStatistics().heap_size_limit; if (v8Heap >= HEAP_MB * 1024 * 1024 * 0.9) return false; - // --stack-size is a V8 flag not allowed in NODE_OPTIONS on Node 24+, - // so pass it only as a direct CLI argument, not via the environment. - const cliFlags = [HEAP_FLAG]; + // --stack-size is a V8 flag not allowed in NODE_OPTIONS on Node 24+, so pass it + // only as a direct CLI argument. --max-semi-space-size IS allowed in NODE_OPTIONS. + const cliFlags = [HEAP_FLAG, SEMI_FLAG]; if (!nodeOpts.includes('--stack-size')) cliFlags.push(STACK_FLAG); const childArgs = [...cliFlags, ...process.argv.slice(1)]; const childEnv = { ...process.env, - NODE_OPTIONS: `${nodeOpts} ${HEAP_FLAG}`.trim(), + NODE_OPTIONS: `${nodeOpts} ${HEAP_FLAG} ${SEMI_FLAG}`.trim(), }; if (shouldBridgeRespawnProgressTty()) childEnv[RESPAWN_PROGRESS_ENV] = '1'; const childExit = await runRespawnedAnalyze(childArgs, childEnv); if (childExit.status !== 0 || childExit.signal) { if (childProcessLikelyOom(childExit)) { cliError( - ` Analysis likely ran out of memory.\n` + - ` Retry with a larger heap if your machine allows it:\n` + - ` NODE_OPTIONS="--max-old-space-size=24576" gitnexus analyze [your-args]\n` + - ` (Windows: set NODE_OPTIONS=--max-old-space-size=24576 && gitnexus analyze [your-args])\n` + + ` Analysis likely ran out of memory (heap cap auto-sized to ${RESPAWN_HEAP_MB}MB ≈ 0.75x RAM).\n` + + ` This repository's working set exceeds available RAM. Use a machine with more RAM,\n` + + ` or override the cap (a cap above physical RAM causes swap-thrash — use with care):\n` + + ` NODE_OPTIONS="--max-old-space-size=" gitnexus analyze [your-args]\n` + + ` (Windows: set NODE_OPTIONS=--max-old-space-size= && gitnexus analyze [your-args])\n` + ` If this persists, it may be a native crash unrelated to heap size.\n`, { recoveryHint: 'heap-oom-respawn' }, ); @@ -474,8 +509,7 @@ async function ensureHeap(): Promise { cliError( ` Analysis aborted in a native worker or native binding path.\n` + ` Try one of these recovery paths:\n` + - ` gitnexus analyze --workers 0\n` + - ` npm uninstall -g gitnexus && npm install -g gitnexus@latest\n` + + ` npm uninstall -g gitnexus && npm install -g gitnexus@latest (rebuilds native bindings)\n` + ` Use Node 22 LTS if you are on a newer non-LTS runtime.\n`, { recoveryHint: 'native-worker-abort' }, ); @@ -500,6 +534,7 @@ const ANALYZE_CLI_ENV_KEYS = [ 'GITNEXUS_VERBOSE', 'GITNEXUS_PROFILE_DEFERRED', 'GITNEXUS_PROFILE_DEFERRED_SLOW_MS', + 'GITNEXUS_DEBUG_HEAP', 'GITNEXUS_MAX_FILE_SIZE', 'GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS', 'GITNEXUS_WAL_CHECKPOINT_THRESHOLD', @@ -598,7 +633,7 @@ export interface AnalyzeOptions { workerTimeout?: string; /** Control LadybugDB WAL auto-checkpoint threshold during analyze. */ walCheckpointThreshold?: string; - /** Parse worker pool size; 0 disables workers (sequential fallback). */ + /** Parse worker pool size (>=1); 0 is rejected (no sequential mode). */ workers?: string; embeddingThreads?: string; embeddingBatchSize?: string; @@ -793,10 +828,11 @@ const analyzeCommandImpl = async ( let workerPoolSize: number | undefined; if (options.workers !== undefined) { const parsedWorkers = Number(options.workers); - if (!Number.isInteger(parsedWorkers) || parsedWorkers < 0) { + if (!Number.isInteger(parsedWorkers) || parsedWorkers < 1) { cliError( - ' --workers must be a non-negative integer. ' + - 'Pass 0 to disable the worker pool (sequential fallback).\n', + ' --workers must be a positive integer (>= 1). ' + + 'GitNexus parses through a worker pool only — there is no sequential ' + + 'mode, so 0 is not allowed. Omit --workers for an auto-sized pool.\n', ); process.exitCode = 1; return; diff --git a/gitnexus/src/cli/i18n/en.ts b/gitnexus/src/cli/i18n/en.ts index 6b57c2aa6..040008570 100644 --- a/gitnexus/src/cli/i18n/en.ts +++ b/gitnexus/src/cli/i18n/en.ts @@ -175,7 +175,7 @@ export const en = { 'help.option.analyze.walCheckpointThreshold': 'LadybugDB WAL auto-checkpoint threshold in bytes during analyze (integer >= -1; default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).', 'help.option.analyze.workers': - 'Parse worker pool size. Default: cores-1 capped at 16. Pass 0 to disable workers (sequential).', + 'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.', 'help.option.analyze.embeddingThreads': 'Limit local ONNX embedding CPU threads', 'help.option.analyze.embeddingBatchSize': 'Number of nodes per embedding batch', 'help.option.analyze.embeddingSubBatchSize': 'Number of chunks per embedding model call', diff --git a/gitnexus/src/cli/i18n/zh-CN.ts b/gitnexus/src/cli/i18n/zh-CN.ts index 0eec71c44..6d1efb77a 100644 --- a/gitnexus/src/cli/i18n/zh-CN.ts +++ b/gitnexus/src/cli/i18n/zh-CN.ts @@ -164,7 +164,7 @@ export const zhCN = { 'help.option.analyze.walCheckpointThreshold': 'analyze 期间 LadybugDB WAL 自动 checkpoint 阈值(字节,整数 >= -1;默认:67108864 = 64 MiB;-1 保持 Ladybug 默认约 16 MiB)。', 'help.option.analyze.workers': - '解析 worker 池大小。默认:cores-1,最多 16。传 0 禁用 worker(顺序执行)。', + '解析 worker 池大小(>=1)。默认:cores-1,最多 16,按仓库规模自适应。', 'help.option.analyze.embeddingThreads': '限制本地 ONNX 嵌入 CPU 线程数', 'help.option.analyze.embeddingBatchSize': '每个嵌入批次的节点数', 'help.option.analyze.embeddingSubBatchSize': '每次嵌入模型调用的分块数', diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index 8ae99455d..a2a88bd83 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -87,7 +87,7 @@ program ) .option( '--workers ', - 'Parse worker pool size. Default: cores-1 capped at 16. Pass 0 to disable workers (sequential).', + 'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.', ) .option('--embedding-threads ', 'Limit local ONNX embedding CPU threads') .option('--embedding-batch-size ', 'Number of nodes per embedding batch') diff --git a/gitnexus/src/core/ingestion/call-processor.ts b/gitnexus/src/core/ingestion/call-processor.ts index 4a9fba9af..5c8e447f6 100644 --- a/gitnexus/src/core/ingestion/call-processor.ts +++ b/gitnexus/src/core/ingestion/call-processor.ts @@ -39,6 +39,34 @@ const MAX_TYPE_NAME_LENGTH = 256; * Consumed by the cross-file re-resolution / enrichment pass. */ export type ExportedTypeMap = Map>; +/** Record one exported graph node into the incremental ExportedTypeMap. */ +export const accumulateExportedTypesFromParsedNode = ( + result: ExportedTypeMap, + node: { id: string; properties?: Record }, + symbolTable: SymbolTableReader, +): void => { + if (!node.properties?.isExported) return; + if (!node.properties?.filePath || !node.properties?.name) return; + const filePath = node.properties.filePath as string; + const name = node.properties.name as string; + if (!name || name.length > MAX_TYPE_NAME_LENGTH) return; + const defs = symbolTable.lookupExactAll(filePath, name); + const def = defs.find((d) => d.nodeId === node.id) ?? defs[0]; + if (!def) return; + const typeName = def.returnType ?? def.declaredType; + if (!typeName || typeName.length > MAX_TYPE_NAME_LENGTH) return; + const simpleType = extractReturnTypeName(typeName) ?? typeName; + if (!simpleType) return; + let fileExports = result.get(filePath); + if (!fileExports) { + fileExports = new Map(); + result.set(filePath, fileExports); + } + if (fileExports.size < MAX_EXPORTS_PER_FILE) { + fileExports.set(name, simpleType); + } +}; + /** Build ExportedTypeMap from graph nodes — used for the worker path where the * sequential TypeEnv is not available in the main thread. Collects * returnType/declaredType from exported symbols with known types. */ @@ -48,29 +76,7 @@ export function buildExportedTypeMapFromGraph( ): ExportedTypeMap { const result: ExportedTypeMap = new Map(); graph.forEachNode((node) => { - if (!node.properties?.isExported) return; - if (!node.properties?.filePath || !node.properties?.name) return; - const filePath = node.properties.filePath as string; - const name = node.properties.name as string; - if (!name || name.length > MAX_TYPE_NAME_LENGTH) return; - // For callable symbols, use returnType; for properties/variables, use declaredType. - // Use lookupExactAll + nodeId match to handle same-name methods in different classes. - const defs = symbolTable.lookupExactAll(filePath, name); - const def = defs.find((d) => d.nodeId === node.id) ?? defs[0]; - if (!def) return; - const typeName = def.returnType ?? def.declaredType; - if (!typeName || typeName.length > MAX_TYPE_NAME_LENGTH) return; - // Extract simple type name (strip Promise<>, etc.) — reuse shared utility - const simpleType = extractReturnTypeName(typeName) ?? typeName; - if (!simpleType) return; - let fileExports = result.get(filePath); - if (!fileExports) { - fileExports = new Map(); - result.set(filePath, fileExports); - } - if (fileExports.size < MAX_EXPORTS_PER_FILE) { - fileExports.set(name, simpleType); - } + accumulateExportedTypesFromParsedNode(result, node, symbolTable); }); return result; } diff --git a/gitnexus/src/core/ingestion/finalize-orchestrator.ts b/gitnexus/src/core/ingestion/finalize-orchestrator.ts index 558672d7d..02717a0b9 100644 --- a/gitnexus/src/core/ingestion/finalize-orchestrator.ts +++ b/gitnexus/src/core/ingestion/finalize-orchestrator.ts @@ -51,6 +51,8 @@ import { finalize, } from 'gitnexus-shared'; import type { ScopeResolutionIndexes } from './model/scope-resolution-indexes.js'; +import { parseTruthyEnv } from './utils/env.js'; +import { TransitionalScopeTree } from '../../storage/scope-index-store.js'; // ─── Public entry point ───────────────────────────────────────────────────── @@ -114,7 +116,13 @@ export function finalizeScopeModel( moduleEntries.push({ filePath: file.filePath, moduleScopeId: file.moduleScope }); } - const scopeTree = buildScopeTree(allScopes); + // Out-of-core scope index: when enabled, build a TransitionalScopeTree + // (validated + fully resident now; sealed to disk by run.ts just before emit so + // the heavy Scope.bindings payload is reclaimed). Default off → the in-heap + // buildScopeTree result exactly, byte-identical. + const scopeTree = parseTruthyEnv(process.env.GITNEXUS_DISK_SCOPE_INDEX) + ? new TransitionalScopeTree(allScopes) + : buildScopeTree(allScopes); const defs = buildDefIndex(allDefs); const qualifiedNames = buildQualifiedNameIndex(allDefs); const moduleScopes = buildModuleScopeIndex(moduleEntries); diff --git a/gitnexus/src/core/ingestion/language-provider.ts b/gitnexus/src/core/ingestion/language-provider.ts index c979102e5..b375050fa 100644 --- a/gitnexus/src/core/ingestion/language-provider.ts +++ b/gitnexus/src/core/ingestion/language-provider.ts @@ -311,6 +311,27 @@ interface LanguageProviderConfig { }, ) => readonly CaptureMatch[]; + /** + * Snapshot the capture-time side-channel state that this provider's + * `emitScopeCaptures` just populated for `filePath` into module-level maps, + * returning a plain JSON-serializable value (or `undefined` when there is + * nothing to carry). + * + * Called in the parse worker IMMEDIATELY after `emitScopeCaptures` runs for + * a file (see `parse-worker.ts`), and the result is stored on the produced + * `ParsedFile.captureSideChannel`. Scope-resolution on the main thread reuses + * that serialized `ParsedFile` and skips re-extraction (#1983), so this hook + * is how the worker-computed marks survive the worker→main boundary and the + * disk store WITHOUT a main-thread re-parse. The main thread restores them + * via the matching `ScopeResolver.applyCaptureSideChannel` hook. + * + * MUST return plain data (objects / arrays / primitives) so it round-trips + * through `JSON.stringify` + the parsedfile-store interning reviver. + * + * Default: undefined (provider has no capture-time module-level side effects). + */ + readonly collectCaptureSideChannel?: (filePath: string) => unknown; + /** * Interpret a raw `@import.statement` capture group into a `ParsedImport`. * The central finalize algorithm resolves `ParsedImport.targetRaw` to a diff --git a/gitnexus/src/core/ingestion/languages/c-cpp.ts b/gitnexus/src/core/ingestion/languages/c-cpp.ts index 4ce17a9c5..84523fc70 100644 --- a/gitnexus/src/core/ingestion/languages/c-cpp.ts +++ b/gitnexus/src/core/ingestion/languages/c-cpp.ts @@ -53,6 +53,7 @@ import { cBindingScopeFor, cImportOwningScope, cReceiverBinding, + collectCStaticLinkageSideChannel, } from './c/index.js'; import { emitCppScopeCaptures, @@ -62,6 +63,7 @@ import { cppBindingScopeFor, cppImportOwningScope, cppReceiverBinding, + collectCppCaptureSideChannel, } from './cpp/index.js'; import { extractCppTemplateConstraints } from './cpp/constraint-extractor.js'; @@ -395,6 +397,15 @@ export const cProvider = defineLanguage({ // ── RFC #909 Ring 3: scope-based resolution hooks (RFC §5) ────────── emitScopeCaptures: emitCScopeCaptures, + // Worker-side: snapshot the module-level `static`-linkage marks + // `emitCScopeCaptures` just populated for this file (`markStaticName` → + // `staticNames`) into plain data on `ParsedFile.captureSideChannel`, so the + // main thread can restore them via `applyCaptureSideChannel` WITHOUT a + // re-parse (#1983 — the worker is the sole parse path). Without this, C + // `static` functions look non-file-local on the main thread and leak into + // cross-file global free-call resolution / wildcard imports. See + // `c/capture-side-channel.ts`. + collectCaptureSideChannel: collectCStaticLinkageSideChannel, interpretImport: interpretCImport, interpretTypeBinding: interpretCTypeBinding, bindingScopeFor: cBindingScopeFor, @@ -465,6 +476,11 @@ export const cppProvider = defineLanguage({ // ── RFC #909 Ring 3: scope-based resolution hooks (RFC §5) ────────── emitScopeCaptures: emitCppScopeCaptures, + // Worker-side: snapshot the module-level capture marks `emitCppScopeCaptures` + // just populated for this file into plain data on `ParsedFile.captureSideChannel`, + // so the main thread can restore them via `applyCaptureSideChannel` WITHOUT a + // re-parse (#1983). See `cpp/capture-side-channel.ts`. + collectCaptureSideChannel: collectCppCaptureSideChannel, interpretImport: interpretCppImport, interpretTypeBinding: interpretCppTypeBinding, bindingScopeFor: cppBindingScopeFor, diff --git a/gitnexus/src/core/ingestion/languages/c/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/c/capture-side-channel.ts new file mode 100644 index 000000000..2f619e5f3 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/c/capture-side-channel.ts @@ -0,0 +1,80 @@ +/** + * C capture-time side-channel serialization (#1983). + * + * `emitCScopeCaptures` populates one MODULE-LEVEL, per-file map as a side + * effect that is NOT part of the returned `ParsedFile`'s scopes/defs: + * + * - `staticNames` (static-linkage.ts) — the simple names of functions + * declared with `static` storage class (file-local / translation-unit + * linkage in C), recorded via `markStaticName` from the + * `@declaration.name` capture when the function node has a `static` + * storage-class specifier. + * + * On the worker path that map is filled in the WORKER process and lost across + * the worker→main MessageChannel (and the disk-backed parsedfile-store), + * because scope-resolution reuses the serialized `ParsedFile` and SKIPS the + * main-thread re-extraction (the #1983 fix that avoids a main-thread + * tree-sitter re-parse / OOM on huge repos — e.g. the Linux kernel). The main + * thread then reads the map empty in `isStaticName` (consulted by + * `isFileLocalDef` in `c/scope-resolver.ts` and by `expandCWildcardNames` in + * static-linkage.ts) — so file-local `static` functions become eligible for + * cross-file global free-call resolution (false CALLS edges) and `#include` + * wildcard imports over-expose them. + * + * This module snapshots the per-file slice of that map into a plain, + * JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and + * restores it on the main thread WITHOUT any parse. It mirrors the C++ pattern + * in `cpp/capture-side-channel.ts` and the Kotlin pattern in + * `kotlin/capture-side-channel.ts`. + * + * The single generic `ParsedFile.captureSideChannel` field is shared with C++ + * and Kotlin, which is safe because each file is one language (a `.c` file uses + * the C provider). The payload is self-describing (`{ kind: 'c', staticNames }`) + * so `applyCStaticLinkageSideChannel` only restores C state and ignores a + * foreign-shaped snapshot. + */ + +import type { ParsedFile } from 'gitnexus-shared'; +import { getStaticNamesForFile, markStaticName } from './static-linkage.js'; + +/** + * Plain JSON-serializable snapshot of the per-file C capture-time + * side-channel. Carried opaquely on `ParsedFile.captureSideChannel`. The + * `kind` tag makes the payload self-describing so `apply` can distinguish a C + * snapshot from another language's (C++ and Kotlin share the same field). + */ +export interface CCaptureSideChannel { + readonly kind: 'c'; + /** Simple names of `static` (file-local linkage) functions in this file. */ + readonly staticNames: readonly string[]; +} + +/** + * `LanguageProvider.collectCaptureSideChannel` implementation for C. + * Returns `undefined` when this file recorded no static names at all, so the + * produced `ParsedFile` carries the field only when there's data to ship. + */ +export function collectCStaticLinkageSideChannel( + filePath: string, +): CCaptureSideChannel | undefined { + const staticNames = getStaticNamesForFile(filePath); + if (staticNames.length === 0) return undefined; + return { kind: 'c', staticNames }; +} + +/** + * `ScopeResolver.applyCaptureSideChannel` implementation for C. Reads the + * worker-serialized snapshot from `parsed.captureSideChannel` and re-populates + * the module-level static-linkage map via `markStaticName`. Tolerant of + * `undefined` (file carried no data) and of an unexpected / foreign shape + * (defensive — the `kind` tag guards against restoring a non-C payload). + * Does NO tree-sitter parse. + */ +export function applyCStaticLinkageSideChannel(parsed: ParsedFile): void { + const data = parsed.captureSideChannel as CCaptureSideChannel | undefined; + if (data === undefined || data === null || typeof data !== 'object') return; + if (data.kind !== 'c' || !Array.isArray(data.staticNames)) return; + for (const name of data.staticNames) { + markStaticName(parsed.filePath, name); + } +} diff --git a/gitnexus/src/core/ingestion/languages/c/import-target.ts b/gitnexus/src/core/ingestion/languages/c/import-target.ts index 0cb9c2fb4..495846030 100644 --- a/gitnexus/src/core/ingestion/languages/c/import-target.ts +++ b/gitnexus/src/core/ingestion/languages/c/import-target.ts @@ -1,5 +1,54 @@ import { dirname, join } from 'path'; +/** + * A workspace file path pre-decomposed for the suffix-match fallback: + * `original` is returned verbatim (preserving the prior `bestMatch = filePath` + * contract); `normalized` and `depth` are precomputed so the hot path does no + * per-element regex/`split`. + */ +interface CSuffixCandidate { + original: string; + normalized: string; + depth: number; +} + +/** + * Per-pass memo: workspace paths bucketed by basename (last path segment), + * keyed on the `allFilePaths` set identity. + * + * `resolveCImportTarget` is called once per (quoted) C/C++ `#include` with the + * same `allFilePaths` set per pass (the augmented set is itself memoized in + * the C resolver). The old suffix-match fallback scanned ALL workspace paths + * per include — with a per-element `.replace`/`.split` and no early exit + * (the fewest-path-components tie-break forces a full scan) — i.e. + * O(R_suffix × (F+H)). A path can satisfy `endsWith('/'+target)` (or equal + * the target) ONLY IF its basename equals the target's last segment, so we + * pre-bucket by basename once (O(F+H), `normalized`/`depth` precomputed) and + * the fallback inspects a single small bucket → O(F+H) build + ~O(1)/include. + * `WeakMap`-keyed so it is reclaimed with the pass (no cross-pass staleness). + * Shared by C and C++ (`resolveCppImportTarget` delegates here). + */ +const suffixIndexByPaths = new WeakMap, Map>(); + +function suffixIndex(allFilePaths: ReadonlySet): Map { + let index = suffixIndexByPaths.get(allFilePaths); + if (index === undefined) { + index = new Map(); + for (const original of allFilePaths) { + const normalized = original.replace(/\\/g, '/'); + const basename = normalized.slice(normalized.lastIndexOf('/') + 1); + let bucket = index.get(basename); + if (bucket === undefined) { + bucket = []; + index.set(basename, bucket); + } + bucket.push({ original, normalized, depth: normalized.split('/').length }); + } + suffixIndexByPaths.set(allFilePaths, index); + } + return index; +} + /** * Resolve a C #include path to a file in the workspace. * @@ -41,21 +90,31 @@ export function resolveCImportTarget( // Exact match (path as-is in the workspace) if (allFilePaths.has(normalizedTarget)) return normalizedTarget; - // Suffix match: find files ending with /targetRaw or equal to targetRaw + // Suffix match: find files ending with /targetRaw or equal to targetRaw. + // A path can only match `=== normalizedTarget` or `endsWith('/'+target)` if + // its basename equals the target's last segment, so we inspect only that + // basename bucket (built once per pass) instead of scanning every workspace + // path. Match condition + tie-break (fewest path components, then + // lexicographic on the normalized path) are byte-identical to the prior scan. const suffix = '/' + normalizedTarget; + const targetBasename = normalizedTarget.slice(normalizedTarget.lastIndexOf('/') + 1); + const bucket = suffixIndex(allFilePaths).get(targetBasename); + if (bucket === undefined) return null; + let bestMatch: string | null = null; let bestDepth = Infinity; let bestNormalized = ''; - for (const filePath of allFilePaths) { - const normalized = filePath.replace(/\\/g, '/'); - if (normalized === normalizedTarget || normalized.endsWith(suffix)) { + for (const cand of bucket) { + if (cand.normalized === normalizedTarget || cand.normalized.endsWith(suffix)) { // Prefer shortest path (closest match) - const depth = normalized.split('/').length; - if (depth < bestDepth || (depth === bestDepth && normalized < bestNormalized)) { - bestDepth = depth; - bestMatch = filePath; - bestNormalized = normalized; + if ( + cand.depth < bestDepth || + (cand.depth === bestDepth && cand.normalized < bestNormalized) + ) { + bestDepth = cand.depth; + bestMatch = cand.original; + bestNormalized = cand.normalized; } } } diff --git a/gitnexus/src/core/ingestion/languages/c/index.ts b/gitnexus/src/core/ingestion/languages/c/index.ts index c6900ecba..8ebc2c98e 100644 --- a/gitnexus/src/core/ingestion/languages/c/index.ts +++ b/gitnexus/src/core/ingestion/languages/c/index.ts @@ -13,4 +13,9 @@ export { isStaticName, clearStaticNames, expandCWildcardNames, + getStaticNamesForFile, } from './static-linkage.js'; +export { + collectCStaticLinkageSideChannel, + applyCStaticLinkageSideChannel, +} from './capture-side-channel.js'; diff --git a/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts index 90d54e072..cb4a6424f 100644 --- a/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/c/scope-resolver.ts @@ -7,6 +7,43 @@ import { cProvider } from '../c-cpp.js'; import { cArityCompatibility, cMergeBindings, resolveCImportTarget } from './index.js'; import { scanHeaderFiles } from './header-scan.js'; import { expandCWildcardNames, isStaticName, clearStaticNames } from './static-linkage.js'; +import { applyCStaticLinkageSideChannel } from './capture-side-channel.js'; + +/** + * Per-pass memo of the augmented `#include`-resolution file set + * (`allFilePaths` ∪ header `.h` paths), keyed on the two stable source sets. + * + * `resolveImportTarget` is called once per C `#include`; the old code rebuilt + * a fresh ~F-entry `Set` on EVERY call (O(R × (F+H)) inserts + GC churn) and, + * worse, defeated `resolveCImportTarget`'s own per-set suffix-index memo by + * handing it a new set identity each time. Both `allFilePaths` (built once in + * scope-resolution `run.ts`) and the header set (`loadResolutionConfig` + * result) are stable per pass, so the union is built once and reused. + * `WeakMap`-keyed → reclaimed with the pass (no cross-pass staleness). + */ +const augmentedPathsByPass = new WeakMap< + ReadonlySet, + WeakMap, ReadonlySet> +>(); + +function augmentedFilePaths( + allFilePaths: ReadonlySet, + headerPaths: ReadonlySet, +): ReadonlySet { + let byHeaders = augmentedPathsByPass.get(allFilePaths); + if (byHeaders === undefined) { + byHeaders = new WeakMap(); + augmentedPathsByPass.set(allFilePaths, byHeaders); + } + let augmented = byHeaders.get(headerPaths); + if (augmented === undefined) { + const set = new Set(allFilePaths); + for (const h of headerPaths) set.add(h); + augmented = set; + byHeaders.set(headerPaths, augmented); + } + return augmented; +} /** * C `ScopeResolver` registered in `SCOPE_RESOLVERS` and consumed by @@ -31,15 +68,34 @@ export const cScopeResolver: ScopeResolver = { return scanHeaderFiles(repoPath); }, + // Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`). + // `emitCScopeCaptures` records per-file `static`-linkage names + // (`markStaticName` → `staticNames`) as a SIDE EFFECT — that state is NOT + // serialized onto the returned ParsedFile's scopes/defs. On the worker path + // those marks are populated in the worker process and lost across the + // MessageChannel / disk store; the main thread reuses the serialized + // ParsedFile and skips `extractParsedFile`, so `isStaticName` (read by + // `isFileLocalDef` and `expandCWildcardNames`) sees an empty map and C + // `static` functions leak into cross-file global free-call resolution + // (false CALLS edges) and `#include` wildcard imports. The worker stashed a + // plain-data snapshot on `parsed.captureSideChannel` via + // `cProvider.collectCaptureSideChannel`; this restores it into the module + // map WITHOUT any tree-sitter re-parse (the #1983 fix). The + // freshly-extracted leg never calls this — its marks were just populated in + // this process. Runs BEFORE `populateOwners`. + applyCaptureSideChannel: applyCStaticLinkageSideChannel, + resolveImportTarget: (targetRaw, fromFile, allFilePaths, resolutionConfig) => { // Augment allFilePaths with .h files discovered via loadResolutionConfig // since the phase only passes .c files to the C resolver but #include // targets .h files classified as C++ in language detection. const headerPaths = resolutionConfig as ReadonlySet | undefined; if (headerPaths !== undefined && headerPaths.size > 0) { - const augmented = new Set(allFilePaths); - for (const h of headerPaths) augmented.add(h); - return resolveCImportTarget(targetRaw, fromFile, augmented); + return resolveCImportTarget( + targetRaw, + fromFile, + augmentedFilePaths(allFilePaths, headerPaths), + ); } return resolveCImportTarget(targetRaw, fromFile, allFilePaths); }, diff --git a/gitnexus/src/core/ingestion/languages/c/static-linkage.ts b/gitnexus/src/core/ingestion/languages/c/static-linkage.ts index 5cfd166b1..2cc195205 100644 --- a/gitnexus/src/core/ingestion/languages/c/static-linkage.ts +++ b/gitnexus/src/core/ingestion/languages/c/static-linkage.ts @@ -29,11 +29,58 @@ export function isStaticName(filePath: string, name: string): boolean { return staticNames.get(filePath)?.has(name) ?? false; } +/** + * Return the `static` (file-local) names recorded for the given file as a + * plain array (empty when none). Used to snapshot the per-file slice of the + * module-level `staticNames` map into `ParsedFile.captureSideChannel` so it + * survives the worker→main boundary (#1983 — the worker is the sole parse + * path). See `c/capture-side-channel.ts`. + */ +export function getStaticNamesForFile(filePath: string): string[] { + const names = staticNames.get(filePath); + return names === undefined ? [] : [...names]; +} + /** Clear tracked static names (for testing). */ export function clearStaticNames(): void { staticNames.clear(); } +/** + * Per-pass memo: `moduleScope` → owning `ParsedFile`, keyed on the + * `parsedFiles` array identity. + * + * The shared finalize Phase-4 loop calls `expandsWildcardTo` + * (→ `expandCWildcardNames`) ONCE PER RESOLVED `#include` edge, every time + * with the SAME `parsedFiles` reference (wired at scope-resolution + * `run.ts` — `allFilePaths`/`parsedFiles` are built once per pass). The old + * `parsedFiles.find(...)` therefore did a full O(F) scan per edge → + * O(R_include × F) overall; at Linux-kernel scale (F ≈ 63k C files, tens of + * thousands of resolved includes) that is ~10^10+ comparisons on a single + * thread — the dominant term in the scope-resolution finalize grind. + * + * Building the lookup once collapses it to O(R_include + F). `WeakMap`-keyed + * on the array so the index is reclaimed with the pass — no cross-pass + * staleness (mirrors the {@link clearStaticNames} discipline for server-mode + * / multi-repo reuse), and a fresh array transparently rebuilds. + */ +const moduleScopeIndexByPass = new WeakMap>(); + +function moduleScopeIndex(parsedFiles: readonly ParsedFile[]): Map { + let index = moduleScopeIndexByPass.get(parsedFiles); + if (index === undefined) { + index = new Map(); + // First-wins to preserve `Array.find` semantics (returns the first match). + // `moduleScope` is unique per file in practice, so collisions are absent; + // the guard only formalises identical behaviour to the prior `.find`. + for (const p of parsedFiles) { + if (!index.has(p.moduleScope)) index.set(p.moduleScope, p); + } + moduleScopeIndexByPass.set(parsedFiles, index); + } + return index; +} + /** * Return the names visible through a C wildcard import (`#include`). * All module-scope defs from the target file are visible EXCEPT those @@ -43,7 +90,7 @@ export function expandCWildcardNames( targetModuleScope: ScopeId, parsedFiles: readonly ParsedFile[], ): readonly string[] { - const target = parsedFiles.find((p) => p.moduleScope === targetModuleScope); + const target = moduleScopeIndex(parsedFiles).get(targetModuleScope); if (target === undefined) return []; const seen = new Set(); diff --git a/gitnexus/src/core/ingestion/languages/cpp/adl.ts b/gitnexus/src/core/ingestion/languages/cpp/adl.ts index 565c23125..7a1bd9f66 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/adl.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/adl.ts @@ -108,6 +108,38 @@ const argInfoBySite = new Map(); const noAdlSites = new Set(); const classToNamespaceQualifiedName = new Map(); +/** + * Per-`filePath` index of the site keys this file contributed to + * `argInfoBySite` / `noAdlSites`, kept in **strict lockstep** with those two + * maps (#1983 perf). Without it, `collectCppAdlSideChannel(filePath)` had to + * scan the ENTIRE module-level maps (every site of every file the worker + * parsed in the current sub-batch) and `parseSiteKey` each entry just to pick + * out one file's slice — O(F²) per sub-batch (~100M `parseSiteKey` calls + * across the Linux kernel). These indexes turn collect into + * O(entries-for-this-file). + * + * Lockstep invariant: a key is pushed here at most once, exactly when it is + * first inserted into the corresponding map, and both indexes are cleared + * wherever `argInfoBySite` / `noAdlSites` are cleared (`clearCppAdlState` and + * the per-file restore in `applyCppAdlSideChannel`). The "first insert only" + * guard mirrors the maps' own de-dup (`Map.set` / `Set.add` are idempotent on + * the key), so iterating an index yields each of this file's keys exactly once + * — byte-identical to the old filtered full scan. + */ +const argInfoSiteKeysByFile = new Map(); +const noAdlSiteKeysByFile = new Map(); + +/** Push `key` into the per-file index `idx[filePath]` (creating the bucket on + * first use). Callers guard against duplicate keys so each key appears once. */ +function pushFileSiteKey(idx: Map, filePath: string, key: string): void { + let keys = idx.get(filePath); + if (keys === undefined) { + keys = []; + idx.set(filePath, keys); + } + keys.push(key); +} + /** * ADL candidate index — built **once** per pipeline run from * `(scopes, parsedFiles)` and reused by every call site. @@ -370,13 +402,94 @@ export function markCppAdlSiteArgs( col: number, args: readonly CppAdlArgInfo[], ): void { - argInfoBySite.set(siteKey(filePath, line, col), args); + const key = siteKey(filePath, line, col); + // Lockstep with `argInfoSiteKeysByFile`: index the key only on first insert + // (a re-mark overwrites the value but must NOT duplicate the index entry). + if (!argInfoBySite.has(key)) pushFileSiteKey(argInfoSiteKeysByFile, filePath, key); + argInfoBySite.set(key, args); } /** Mark a call site as ADL-suppressed (function child wrapped in * `parenthesized_expression`, e.g. `(f)(s)`). */ export function markCppAdlSiteNoAdl(filePath: string, line: number, col: number): void { - noAdlSites.add(siteKey(filePath, line, col)); + const key = siteKey(filePath, line, col); + // Lockstep with `noAdlSiteKeysByFile`: index the key only on first insert. + if (!noAdlSites.has(key)) pushFileSiteKey(noAdlSiteKeysByFile, filePath, key); + noAdlSites.add(key); +} + +/** + * Plain-data, JSON-serializable snapshot of the per-file ADL capture state + * (`argInfoBySite` entries for this file + `noAdlSites` keys for this file). + * Carried on `ParsedFile.captureSideChannel` across the worker→main boundary + * (#1983); the call-site key's `line:col` are stored per-entry so the full + * `filePath:line:col` key can be reconstructed without parsing. + */ +export interface CppAdlSideChannel { + /** Per-call-site arg info: `[line, col, args]` for sites in this file. */ + readonly argInfoBySite: readonly [number, number, readonly CppAdlArgInfo[]][]; + /** ADL-suppressed sites in this file: `[line, col]`. */ + readonly noAdlSites: readonly [number, number][]; +} + +const SITE_KEY_RE = /^(.*):(\d+):(\d+)$/; + +/** Split a `filePath:line:col` site key, tolerating colons in the path. */ +function parseSiteKey(key: string): { filePath: string; line: number; col: number } | undefined { + const m = SITE_KEY_RE.exec(key); + if (m === null) return undefined; + return { filePath: m[1], line: Number(m[2]), col: Number(m[3]) }; +} + +/** + * Snapshot this file's ADL capture state for the worker→main side-channel. + * + * Uses the per-file `argInfoSiteKeysByFile` / `noAdlSiteKeysByFile` indexes to + * touch only THIS file's entries — O(entries-for-this-file) — instead of the + * old O(all-entries) full scan over `argInfoBySite` / `noAdlSites` (#1983). + * The output order, and therefore the serialized JSON shape, is byte-identical + * to the old filtered scan: the index records keys in the same insertion order + * the maps' own iteration would have yielded for this file, and each key is + * indexed exactly once (mark guards on first insert), so the same per-file + * subsequence is produced. + * + * `parseSiteKey` is still used to recover `line:col` from each key, but now + * only for this file's keys (a bounded handful), never for the whole batch. + */ +export function collectCppAdlSideChannel(filePath: string): CppAdlSideChannel { + const args: [number, number, readonly CppAdlArgInfo[]][] = []; + for (const key of argInfoSiteKeysByFile.get(filePath) ?? []) { + const value = argInfoBySite.get(key); + const parsed = parseSiteKey(key); + if (value !== undefined && parsed !== undefined) { + args.push([parsed.line, parsed.col, value]); + } + } + const noAdl: [number, number][] = []; + for (const key of noAdlSiteKeysByFile.get(filePath) ?? []) { + const parsed = parseSiteKey(key); + if (parsed !== undefined) { + noAdl.push([parsed.line, parsed.col]); + } + } + return { argInfoBySite: args, noAdlSites: noAdl }; +} + +/** Restore this file's ADL capture state from the side-channel (no parse). + * Keeps the per-file site-key indexes in lockstep with `argInfoBySite` / + * `noAdlSites` (first-insert-only) so a later `collectCppAdlSideChannel` on + * the same process would still produce a correct, duplicate-free snapshot. */ +export function applyCppAdlSideChannel(filePath: string, data: CppAdlSideChannel): void { + for (const [line, col, value] of data.argInfoBySite) { + const key = siteKey(filePath, line, col); + if (!argInfoBySite.has(key)) pushFileSiteKey(argInfoSiteKeysByFile, filePath, key); + argInfoBySite.set(key, value); + } + for (const [line, col] of data.noAdlSites) { + const key = siteKey(filePath, line, col); + if (!noAdlSites.has(key)) pushFileSiteKey(noAdlSiteKeysByFile, filePath, key); + noAdlSites.add(key); + } } /** Clear ADL state. Called from `cppScopeResolver.loadResolutionConfig` @@ -385,6 +498,11 @@ export function markCppAdlSiteNoAdl(filePath: string, line: number, col: number) export function clearCppAdlState(): void { argInfoBySite.clear(); noAdlSites.clear(); + // Lockstep: the per-file site-key indexes mirror argInfoBySite/noAdlSites and + // MUST be cleared together — a stale index would resurrect a prior pass's + // (or prior file's, after a re-key) keys into the next snapshot. + argInfoSiteKeysByFile.clear(); + noAdlSiteKeysByFile.clear(); classToNamespaceQualifiedName.clear(); adlIndex = undefined; adlIndexSource = undefined; diff --git a/gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts new file mode 100644 index 000000000..74f432b69 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts @@ -0,0 +1,111 @@ +/** + * C++ capture-time side-channel serialization (#1983). + * + * `emitCppScopeCaptures` populates several MODULE-LEVEL maps as a side effect + * that are NOT part of the returned `ParsedFile`'s scopes/defs: + * + * - `argInfoBySite` / `noAdlSites` (adl.ts) + * - `inlineNamespaceRangesByFile` (inline-namespaces.ts) + * - `fileLocalNames` / `anonymousNamespaceRangesByFile` (file-local-linkage.ts) + * - `dependentBasesByFile` / `dependentPackBaseClassesByFile` (two-phase-lookup.ts) + * + * On the worker path those maps are filled in the WORKER process and lost + * across the worker→main MessageChannel (and the disk-backed parsedfile-store), + * because scope-resolution reuses the serialized `ParsedFile` and SKIPS the + * main-thread re-extraction — the entire point of #1983 is to avoid a + * main-thread tree-sitter re-parse on huge `.h`/`.cpp` repos (the OOM). + * + * This module snapshots the per-file slice of those maps into a plain, + * JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and + * restores it on the main thread WITHOUT any parse. It is the data-only + * replacement for the removed re-parse `replayCaptureSideChannel` hook. + * + * The derived state each `populateOwners` / `populateWorkspaceOwners` pass + * builds (resolved scope-id Sets, `dependentBaseNodeIds`, etc.) is recomputed + * on the main thread from these restored capture-time maps, so only the + * capture-time maps need to cross the boundary. + */ + +import type { ParsedFile } from 'gitnexus-shared'; +import { collectCppAdlSideChannel, applyCppAdlSideChannel, type CppAdlSideChannel } from './adl.js'; +import { + collectCppInlineNamespaceSideChannel, + applyCppInlineNamespaceSideChannel, +} from './inline-namespaces.js'; +import { + collectCppFileLocalSideChannel, + applyCppFileLocalSideChannel, + type CppFileLocalSideChannel, +} from './file-local-linkage.js'; +import { + collectCppTwoPhaseSideChannel, + applyCppTwoPhaseSideChannel, + type CppTwoPhaseSideChannel, +} from './two-phase-lookup.js'; + +/** + * Plain JSON-serializable composite of every C++ capture-time side-channel + * slice for one file. Carried opaquely on `ParsedFile.captureSideChannel`. + */ +export interface CppCaptureSideChannel { + /** + * Discriminant tag — the single generic `ParsedFile.captureSideChannel` + * field is shared with C (`{ kind: 'c' }`) and Kotlin (`{ kind: 'kotlin' }`). + * `applyCppCaptureSideChannel` checks this first so a foreign-language + * payload reaching the C++ apply (or vice-versa) is cleanly ignored. In + * practice apply only runs for the matching provider (one language per file), + * but the tag makes it robust and consistent with the C/Kotlin snapshots. + */ + readonly kind: 'cpp'; + readonly adl: CppAdlSideChannel; + /** Inline-namespace source-range keys recorded for this file. */ + readonly inlineNamespaceRanges: readonly string[]; + readonly fileLocal: CppFileLocalSideChannel; + readonly twoPhase: CppTwoPhaseSideChannel; +} + +/** + * `LanguageProvider.collectCaptureSideChannel` implementation for C++. + * Returns `undefined` when this file recorded no side-channel state at all, so + * the produced `ParsedFile` carries the field only when there's data to ship. + */ +export function collectCppCaptureSideChannel(filePath: string): CppCaptureSideChannel | undefined { + const adl = collectCppAdlSideChannel(filePath); + const inlineNamespaceRanges = collectCppInlineNamespaceSideChannel(filePath); + const fileLocal = collectCppFileLocalSideChannel(filePath); + const twoPhase = collectCppTwoPhaseSideChannel(filePath); + + const isEmpty = + adl.argInfoBySite.length === 0 && + adl.noAdlSites.length === 0 && + inlineNamespaceRanges.length === 0 && + fileLocal.fileLocalNames.length === 0 && + fileLocal.anonymousNamespaceRanges.length === 0 && + twoPhase.dependentBases.length === 0 && + twoPhase.dependentPackBaseClasses.length === 0; + if (isEmpty) return undefined; + + return { kind: 'cpp', adl, inlineNamespaceRanges, fileLocal, twoPhase }; +} + +/** + * `ScopeResolver.applyCaptureSideChannel` implementation for C++. Reads the + * worker-serialized snapshot from `parsed.captureSideChannel` and writes it + * back into the module-level maps. Tolerant of `undefined` (file carried no + * data) and of an unexpected shape (defensive — never throws on a malformed + * snapshot). Does NO tree-sitter parse. + */ +export function applyCppCaptureSideChannel(parsed: ParsedFile): void { + const data = parsed.captureSideChannel as CppCaptureSideChannel | undefined; + if (data === undefined || data === null || typeof data !== 'object') return; + // Discriminant guard — the generic `captureSideChannel` field is shared + // with C (`{ kind: 'c' }`) and Kotlin (`{ kind: 'kotlin' }`); cleanly + // ignore a non-C++ payload rather than mis-applying it. + if (data.kind !== 'cpp') return; + if (data.adl !== undefined) applyCppAdlSideChannel(parsed.filePath, data.adl); + if (data.inlineNamespaceRanges !== undefined) { + applyCppInlineNamespaceSideChannel(parsed.filePath, data.inlineNamespaceRanges); + } + if (data.fileLocal !== undefined) applyCppFileLocalSideChannel(parsed.filePath, data.fileLocal); + if (data.twoPhase !== undefined) applyCppTwoPhaseSideChannel(parsed.filePath, data.twoPhase); +} diff --git a/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts b/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts index dd5fc8c0a..c6b383bf0 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/file-local-linkage.ts @@ -95,6 +95,47 @@ export function isCppAnonymousNamespaceScope(scopeId: ScopeId): boolean { return anonymousNamespaceScopeIds.has(scopeId); } +/** + * Plain-data, JSON-serializable snapshot of the per-file capture-time + * file-local-linkage state. Carried on `ParsedFile.captureSideChannel` across + * the worker→main boundary (#1983). The derived sets (`nonGloballyVisibleNodeIds`, + * `anonymousNamespaceScopeIds`) are recomputed by `populateCppNonGloballyVisible` + * / `populateCppAnonymousNamespaceScopes` during `populateOwners`, so only the + * two capture-time maps cross the boundary. + */ +export interface CppFileLocalSideChannel { + /** File-local symbol names (static / anonymous-namespace) in this file. */ + readonly fileLocalNames: readonly string[]; + /** Anonymous-namespace source-range keys recorded for this file. */ + readonly anonymousNamespaceRanges: readonly string[]; +} + +/** Snapshot this file's file-local-linkage capture state for the side-channel. */ +export function collectCppFileLocalSideChannel(filePath: string): CppFileLocalSideChannel { + const names = fileLocalNames.get(filePath); + const anon = anonymousNamespaceRangesByFile.get(filePath); + return { + fileLocalNames: names === undefined ? [] : [...names], + anonymousNamespaceRanges: anon === undefined ? [] : [...anon], + }; +} + +/** Restore this file's file-local-linkage capture state from the side-channel. */ +export function applyCppFileLocalSideChannel( + filePath: string, + data: CppFileLocalSideChannel, +): void { + for (const name of data.fileLocalNames) markFileLocal(filePath, name); + if (data.anonymousNamespaceRanges.length > 0) { + let set = anonymousNamespaceRangesByFile.get(filePath); + if (set === undefined) { + set = new Set(); + anonymousNamespaceRangesByFile.set(filePath, set); + } + for (const r of data.anonymousNamespaceRanges) set.add(r); + } +} + /** Clear tracked file-local names (call at start of each resolution pass). */ export function clearFileLocalNames(): void { fileLocalNames.clear(); @@ -235,11 +276,37 @@ export function isCppDefGloballyVisible(filePath: string, nodeId: string): boole * does, mirror this filter or harden registration so class/namespace * members never enter `localDefs` unqualified. */ +/** + * Per-pass memo: `moduleScope` → owning `ParsedFile`, keyed on the + * `parsedFiles` array identity. The shared finalize Phase-4 loop calls + * `expandsWildcardTo` (→ this) ONCE PER RESOLVED `#include` edge with the same + * `parsedFiles` reference; the old `parsedFiles.find(...)` was therefore O(F) + * per edge → O(R·F) overall (at kernel scale the ~25–30k `.h` headers are + * classified C++, so this fires hard — the C twin in `c/static-linkage.ts`). + * Building the lookup once collapses it to O(R+F). `WeakMap`-keyed so it is + * reclaimed with the pass (no cross-pass staleness; mirrors + * {@link clearFileLocalNames}). + */ +const moduleScopeIndexByPass = new WeakMap>(); + +function moduleScopeIndex(parsedFiles: readonly ParsedFile[]): Map { + let index = moduleScopeIndexByPass.get(parsedFiles); + if (index === undefined) { + index = new Map(); + // First-wins to preserve `Array.find` semantics (returns the first match). + for (const p of parsedFiles) { + if (!index.has(p.moduleScope)) index.set(p.moduleScope, p); + } + moduleScopeIndexByPass.set(parsedFiles, index); + } + return index; +} + export function expandCppWildcardNames( targetModuleScope: ScopeId, parsedFiles: readonly ParsedFile[], ): readonly string[] { - const target = parsedFiles.find((p) => p.moduleScope === targetModuleScope); + const target = moduleScopeIndex(parsedFiles).get(targetModuleScope); if (target === undefined) return []; // Build nodeId → owning Scope map from the structural scope tree. diff --git a/gitnexus/src/core/ingestion/languages/cpp/index.ts b/gitnexus/src/core/ingestion/languages/cpp/index.ts index c4d208d76..4bc52dd32 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/index.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/index.ts @@ -14,3 +14,7 @@ export { clearFileLocalNames, expandCppWildcardNames, } from './file-local-linkage.js'; +export { + collectCppCaptureSideChannel, + applyCppCaptureSideChannel, +} from './capture-side-channel.js'; diff --git a/gitnexus/src/core/ingestion/languages/cpp/inline-namespaces.ts b/gitnexus/src/core/ingestion/languages/cpp/inline-namespaces.ts index eec44c68f..c7280bf46 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/inline-namespaces.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/inline-namespaces.ts @@ -61,6 +61,30 @@ export function markCppInlineNamespaceRange(filePath: string, range: RangeKey): set.add(rangeKey(range)); } +/** Snapshot this file's captured inline-namespace ranges for the worker→main + * side-channel (#1983). `populateCppInlineNamespaceScopes` (in `populateOwners`) + * later resolves these range keys to ScopeIds on the main thread, so only the + * capture-time ranges need to cross the boundary. Returns the rangeKey strings + * as a plain array (empty when this file recorded none). */ +export function collectCppInlineNamespaceSideChannel(filePath: string): readonly string[] { + const set = inlineNamespaceRangesByFile.get(filePath); + return set === undefined ? [] : [...set]; +} + +/** Restore this file's captured inline-namespace ranges from the side-channel. */ +export function applyCppInlineNamespaceSideChannel( + filePath: string, + ranges: readonly string[], +): void { + if (ranges.length === 0) return; + let set = inlineNamespaceRangesByFile.get(filePath); + if (set === undefined) { + set = new Set(); + inlineNamespaceRangesByFile.set(filePath, set); + } + for (const r of ranges) set.add(r); +} + /** Clear all inline-namespace state. Called from `clearFileLocalNames`. */ export function clearCppInlineNamespaces(): void { inlineNamespaceRangesByFile.clear(); diff --git a/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts index 459b313c6..5e24e292d 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts @@ -30,6 +30,7 @@ import { isCppDependentBaseMember, } from './two-phase-lookup.js'; import { populateCppAssociatedNamespaces, clearCppAdlState, pickCppAdlCandidates } from './adl.js'; +import { applyCppCaptureSideChannel } from './capture-side-channel.js'; import { clearCppInlineNamespaces, populateCppInlineNamespaceScopes, @@ -42,6 +43,40 @@ import { populateCppUserDefinedConversions, } from './user-defined-conversions.js'; +/** + * Per-pass memo of the augmented `#include`-resolution file set + * (`allFilePaths` ∪ header paths), keyed on the two stable source sets. + * `resolveImportTarget` is called once per C++ `#include`; the old code rebuilt + * a fresh ~F-entry `Set` on every call AND defeated the shared + * `resolveCImportTarget` suffix-index memo (in `c/import-target.ts`) by handing + * it a new set identity each time. Both inputs are stable per pass, so the + * union is built once and reused. `WeakMap`-keyed → reclaimed with the pass. + * (Twin of the C resolver's `augmentedFilePaths`.) + */ +const augmentedPathsByPass = new WeakMap< + ReadonlySet, + WeakMap, ReadonlySet> +>(); + +function augmentedFilePaths( + allFilePaths: ReadonlySet, + headerPaths: ReadonlySet, +): ReadonlySet { + let byHeaders = augmentedPathsByPass.get(allFilePaths); + if (byHeaders === undefined) { + byHeaders = new WeakMap(); + augmentedPathsByPass.set(allFilePaths, byHeaders); + } + let augmented = byHeaders.get(headerPaths); + if (augmented === undefined) { + const set = new Set(allFilePaths); + for (const h of headerPaths) set.add(h); + augmented = set; + byHeaders.set(headerPaths, augmented); + } + return augmented; +} + /** * C++ `ScopeResolver` registered in `SCOPE_RESOLVERS` and consumed by * the generic `runScopeResolution` orchestrator (RFC #909 Ring 3). @@ -78,9 +113,11 @@ export const cppScopeResolver: ScopeResolver = { // detection but are importable from .cpp files via #include. const headerPaths = resolutionConfig as ReadonlySet | undefined; if (headerPaths !== undefined && headerPaths.size > 0) { - const augmented = new Set(allFilePaths); - for (const h of headerPaths) augmented.add(h); - return resolveCppImportTarget(targetRaw, fromFile, augmented); + return resolveCppImportTarget( + targetRaw, + fromFile, + augmentedFilePaths(allFilePaths, headerPaths), + ); } return resolveCppImportTarget(targetRaw, fromFile, allFilePaths); }, @@ -103,6 +140,24 @@ export const cppScopeResolver: ScopeResolver = { buildMro: (graph, parsedFiles, nodeLookup) => buildMro(graph, parsedFiles, nodeLookup, defaultLinearize), + // Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`). + // `emitCppScopeCaptures` records per-file ADL call-site arg shapes + // (`markCppAdlSiteArgs`/`markCppAdlSiteNoAdl`), inline-/anonymous-namespace + // ranges (`markCppInlineNamespaceRange`/`markCppAnonymousNamespaceRange`), + // dependent-base names (`markCppDependentBase`/`markCppDependentPackBase`), + // and file-local linkage (`markFileLocal`) into module-level maps as a SIDE + // EFFECT — none of it is serialized onto the returned ParsedFile's scopes/defs. + // On the worker path those marks are populated in the worker process and lost + // across the MessageChannel / disk store; the main thread reuses the + // serialized ParsedFile and skips `extractParsedFile`, so `populateOwners` + + // the ADL / two-phase-lookup passes would see empty maps and emit zero edges. + // The worker stashed a plain-data snapshot on `parsed.captureSideChannel` via + // `cppProvider.collectCaptureSideChannel`; this restores it into the module + // maps WITHOUT any tree-sitter re-parse (the #1983 fix — the old re-parse + // replay re-OOM'd huge `.h`/`.cpp` repos). The freshly-extracted leg never + // calls this — its marks were just populated in this process. + applyCaptureSideChannel: applyCppCaptureSideChannel, + populateOwners: (parsed: ParsedFile) => { populateClassOwnedMembers(parsed); // #1982: tag namespace-nested defs with their enclosing-namespace prefix so diff --git a/gitnexus/src/core/ingestion/languages/cpp/two-phase-lookup.ts b/gitnexus/src/core/ingestion/languages/cpp/two-phase-lookup.ts index 7ec55bc9a..739cceeb5 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/two-phase-lookup.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/two-phase-lookup.ts @@ -104,6 +104,56 @@ export function markCppDependentPackBase(filePath: string, className: string): v perFile.add(className); } +/** + * Plain-data, JSON-serializable snapshot of the per-file capture-time + * two-phase-lookup state. Carried on `ParsedFile.captureSideChannel` across the + * worker→main boundary (#1983). The resolved `dependentBaseNodeIds` index is + * rebuilt by `populateCppDependentBases` (workspace pass) after all files have + * their `populateOwners` applied, so only the two capture-time maps cross. + * + * Nested `Map`/`Set` are flattened to arrays here so the snapshot stays plain + * JSON (avoids relying on the parsedfile-store's Map/Set replacer for nested + * structures): `dependentBases` is `[className, [baseName, qualifiers[]][]][]`. + */ +export interface CppTwoPhaseSideChannel { + readonly dependentBases: readonly [string, readonly [string, readonly string[]][]][]; + readonly dependentPackBaseClasses: readonly string[]; +} + +/** Snapshot this file's two-phase-lookup capture state for the side-channel. */ +export function collectCppTwoPhaseSideChannel(filePath: string): CppTwoPhaseSideChannel { + const perFile = dependentBasesByFile.get(filePath); + const dependentBases: [string, [string, string[]][]][] = []; + if (perFile !== undefined) { + for (const [className, bases] of perFile) { + const baseEntries: [string, string[]][] = []; + for (const [baseName, quals] of bases) { + baseEntries.push([baseName, [...quals]]); + } + dependentBases.push([className, baseEntries]); + } + } + const pack = dependentPackBaseClassesByFile.get(filePath); + return { + dependentBases, + dependentPackBaseClasses: pack === undefined ? [] : [...pack], + }; +} + +/** Restore this file's two-phase-lookup capture state from the side-channel. */ +export function applyCppTwoPhaseSideChannel(filePath: string, data: CppTwoPhaseSideChannel): void { + for (const [className, baseEntries] of data.dependentBases) { + for (const [baseName, quals] of baseEntries) { + for (const qualifier of quals) { + markCppDependentBase(filePath, className, baseName, qualifier); + } + } + } + for (const className of data.dependentPackBaseClasses) { + markCppDependentPackBase(filePath, className); + } +} + /** Clear two-phase-lookup state. Called from `clearFileLocalNames`. */ export function clearCppDependentBases(): void { dependentBasesByFile.clear(); diff --git a/gitnexus/src/core/ingestion/languages/csharp/captures.ts b/gitnexus/src/core/ingestion/languages/csharp/captures.ts index b3b720587..3f28e6be9 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/captures.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/captures.ts @@ -86,10 +86,11 @@ export function emitCsharpScopeCaptures( _filePath: string, cachedTree?: unknown, ): readonly CaptureMatch[] { - // Skip the parse when the caller (parse phase's scopeTreeCache) - // already produced a Tree for this source. Cache miss = re-parse, - // same as before. The cachedTree parameter is typed as `unknown` at - // the LanguageProvider contract layer; cast here at the use site. + // Reuse a pre-parsed Tree when the caller passes one via `cachedTree`; a + // miss re-parses. (The cache is currently always empty — its only producer, + // the sequential parser, was removed — so this re-parses in practice.) The + // cachedTree parameter is typed `unknown` at the LanguageProvider contract + // layer; cast here at the use site. let tree = cachedTree as ReturnType['parse']> | undefined; if (tree === undefined) { tree = parseSourceSafe(getCsharpParser(), sourceText, undefined, { diff --git a/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts b/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts index 634afdeec..5dab83afe 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/namespace-siblings.ts @@ -356,8 +356,9 @@ function extractFileStructure(content: string, cachedTree: unknown): CsharpFileS /** Content + (optional) pre-parsed tree-sitter trees keyed by filePath. * The orchestrator builds `fileContents` from the pipeline's file list; - * `treeCache` is the same `scopeTreeCache` already populated by the - * parse phase, so cache hits avoid a second `parser.parse()`. */ + * `treeCache` is currently always empty (its only producer, the sequential + * parser, was removed), so the providers re-parse. Kept as an extension + * point that would let cache hits avoid a second `parser.parse()`. */ export interface CsharpSiblingInputs { readonly fileContents: ReadonlyMap; readonly treeCache?: { get(filePath: string): unknown }; diff --git a/gitnexus/src/core/ingestion/languages/kotlin.ts b/gitnexus/src/core/ingestion/languages/kotlin.ts index 31116663e..6ee7b1dda 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin.ts @@ -28,6 +28,7 @@ import { kotlinMethodConfig } from '../method-extractors/configs/jvm.js'; import { createVariableExtractor } from '../variable-extractors/generic.js'; import { kotlinVariableConfig } from '../variable-extractors/configs/jvm.js'; import { + collectKotlinCaptureSideChannel, emitKotlinScopeCaptures, interpretKotlinImport, interpretKotlinTypeBinding, @@ -175,6 +176,13 @@ export const kotlinProvider = defineLanguage({ // ── RFC #909 Ring 3: scope-based resolution hooks ── emitScopeCaptures: emitKotlinScopeCaptures, + // Worker-side: snapshot the module-level companion-scope marks + // `emitKotlinScopeCaptures` just populated for this file (`markCompanionScope` + // → `companionScopesByFile`) into plain data on `ParsedFile.captureSideChannel`, + // so the main thread can restore them via `applyCaptureSideChannel` WITHOUT a + // re-parse (#1983). Without this, companion/static dispatch emits no CALLS + // edges on the worker path. See `kotlin/capture-side-channel.ts`. + collectCaptureSideChannel: collectKotlinCaptureSideChannel, interpretImport: interpretKotlinImport, interpretTypeBinding: interpretKotlinTypeBinding, bindingScopeFor: kotlinBindingScopeFor, diff --git a/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts new file mode 100644 index 000000000..375e0f679 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts @@ -0,0 +1,75 @@ +/** + * Kotlin capture-time side-channel serialization (#1983). + * + * `emitKotlinScopeCaptures` populates one MODULE-LEVEL, per-file map as a side + * effect that is NOT part of the returned `ParsedFile`'s scopes/defs: + * + * - `companionScopesByFile` (companion-scopes.ts) — the `ScopeId`s that came + * from a `companion_object` AST node, recorded via `markCompanionScope` + * from the `@scope.companion` marker capture. + * + * On the worker path that map is filled in the WORKER process and lost across + * the worker→main MessageChannel (and the disk-backed parsedfile-store), + * because scope-resolution reuses the serialized `ParsedFile` and SKIPS the + * main-thread re-extraction (the #1983 fix that avoids a main-thread + * tree-sitter re-parse / OOM on huge repos). The main thread then reads the map + * empty in `isKotlinStaticOnly` / `populateCompanionMembersOnEnclosingClass` + * (owners.ts) — so companion methods aren't identified as static and + * companion/static dispatch emits no CALLS edges. + * + * This module snapshots the per-file slice of that map into a plain, + * JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and + * restores it on the main thread WITHOUT any parse. It mirrors the C++ pattern + * in `cpp/capture-side-channel.ts`. + * + * The single generic `ParsedFile.captureSideChannel` field is shared with C++, + * which is safe because each file is one language (a `.kt` file uses the kotlin + * provider, a `.cpp` file the cpp provider). The payload is self-describing + * (`{ kind: 'kotlin', companionScopes }`) so `applyKotlinCaptureSideChannel` + * only restores kotlin state and ignores a foreign-shaped snapshot. + */ + +import type { ParsedFile, ScopeId } from 'gitnexus-shared'; +import { getCompanionScopesForFile, markCompanionScope } from './companion-scopes.js'; + +/** + * Plain JSON-serializable snapshot of the per-file Kotlin capture-time + * side-channel. Carried opaquely on `ParsedFile.captureSideChannel`. The + * `kind` tag makes the payload self-describing so `apply` can distinguish a + * kotlin snapshot from another language's (C++ shares the same field). + */ +export interface KotlinCaptureSideChannel { + readonly kind: 'kotlin'; + /** Companion-object scope ids recorded for this file. */ + readonly companionScopes: readonly ScopeId[]; +} + +/** + * `LanguageProvider.collectCaptureSideChannel` implementation for Kotlin. + * Returns `undefined` when this file recorded no companion scopes at all, so + * the produced `ParsedFile` carries the field only when there's data to ship. + */ +export function collectKotlinCaptureSideChannel( + filePath: string, +): KotlinCaptureSideChannel | undefined { + const companionScopes = getCompanionScopesForFile(filePath); + if (companionScopes.length === 0) return undefined; + return { kind: 'kotlin', companionScopes }; +} + +/** + * `ScopeResolver.applyCaptureSideChannel` implementation for Kotlin. Reads the + * worker-serialized snapshot from `parsed.captureSideChannel` and re-populates + * the module-level companion-scope map via `markCompanionScope`. Tolerant of + * `undefined` (file carried no data) and of an unexpected / foreign shape + * (defensive — the `kind` tag guards against restoring a non-kotlin payload). + * Does NO tree-sitter parse. + */ +export function applyKotlinCaptureSideChannel(parsed: ParsedFile): void { + const data = parsed.captureSideChannel as KotlinCaptureSideChannel | undefined; + if (data === undefined || data === null || typeof data !== 'object') return; + if (data.kind !== 'kotlin' || !Array.isArray(data.companionScopes)) return; + for (const scopeId of data.companionScopes) { + markCompanionScope(parsed.filePath, scopeId); + } +} diff --git a/gitnexus/src/core/ingestion/languages/kotlin/companion-scopes.ts b/gitnexus/src/core/ingestion/languages/kotlin/companion-scopes.ts index ac4034620..f1c45c65c 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/companion-scopes.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/companion-scopes.ts @@ -55,6 +55,17 @@ export function isCompanionScope(filePath: string, scopeId: ScopeId): boolean { return companionScopesByFile.get(filePath)?.has(scopeId) ?? false; } +/** + * Snapshot the companion-object scope ids recorded for `filePath` as a plain + * array (for the worker→main capture side-channel, #1983). Returns an empty + * array when the file recorded no companion scopes. See + * `capture-side-channel.ts`. + */ +export function getCompanionScopesForFile(filePath: string): ScopeId[] { + const scopes = companionScopesByFile.get(filePath); + return scopes === undefined ? [] : [...scopes]; +} + /** Clear all tracked companion scopes (for testing). */ export function clearCompanionScopes(): void { companionScopesByFile.clear(); diff --git a/gitnexus/src/core/ingestion/languages/kotlin/index.ts b/gitnexus/src/core/ingestion/languages/kotlin/index.ts index 206128e52..206254540 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/index.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/index.ts @@ -1,4 +1,9 @@ export { emitKotlinScopeCaptures } from './captures.js'; +export { + collectKotlinCaptureSideChannel, + applyKotlinCaptureSideChannel, + type KotlinCaptureSideChannel, +} from './capture-side-channel.js'; export { getKotlinCaptureCacheStats, resetKotlinCaptureCacheStats } from './cache-stats.js'; export { interpretKotlinImport, interpretKotlinTypeBinding } from './interpret.js'; export { kotlinArityCompatibility } from './arity.js'; diff --git a/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts index bd63c1e86..6bd48f0d1 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts @@ -14,6 +14,7 @@ import { type KotlinResolveContext, } from './index.js'; import { clearCompanionScopes } from './companion-scopes.js'; +import { applyKotlinCaptureSideChannel } from './capture-side-channel.js'; import { isKotlinStaticOnly } from './owners.js'; /** @@ -84,6 +85,23 @@ export const kotlinScopeResolver: ScopeResolver = { buildMro: (graph, parsedFiles, nodeLookup) => buildKotlinMro(graph, parsedFiles, nodeLookup), + // Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`). + // `emitKotlinScopeCaptures` records per-file companion-object scope ids + // (`markCompanionScope` → `companionScopesByFile`) as a SIDE EFFECT — that + // state is NOT serialized onto the returned ParsedFile's scopes/defs. On the + // worker path those marks are populated in the worker process and lost across + // the MessageChannel / disk store; the main thread reuses the serialized + // ParsedFile and skips `extractParsedFile`, so `isKotlinStaticOnly` and + // `populateCompanionMembersOnEnclosingClass` (owners.ts) would see an empty + // map and companion/static dispatch would emit zero CALLS edges. The worker + // stashed a plain-data snapshot on `parsed.captureSideChannel` via + // `kotlinProvider.collectCaptureSideChannel`; this restores it into the + // module map WITHOUT any tree-sitter re-parse (the #1983 fix). The + // freshly-extracted leg never calls this — its marks were just populated in + // this process. Runs BEFORE `populateOwners` so the restored companion map is + // visible to it. + applyCaptureSideChannel: applyKotlinCaptureSideChannel, + populateOwners: (parsed: ParsedFile) => populateKotlinOwners(parsed), isSuperReceiver: (text) => text.trim() === 'super', diff --git a/gitnexus/src/core/ingestion/languages/typescript/captures.ts b/gitnexus/src/core/ingestion/languages/typescript/captures.ts index b3e338242..9d466c3f9 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/captures.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/captures.ts @@ -163,10 +163,11 @@ export function emitTsScopeCaptures( filePath: string, cachedTree?: unknown, ): readonly CaptureMatch[] { - // Skip the parse when the caller (parse phase's scopeTreeCache) already - // produced a Tree for this source. Cache miss = re-parse, same as before. - // The cachedTree parameter is typed as `unknown` at the LanguageProvider - // contract layer; cast here at the use site. + // Reuse a pre-parsed Tree when the caller passes one via `cachedTree`; a + // miss re-parses. (The cache is currently always empty — its only producer, + // the sequential parser, was removed — so this re-parses in practice.) The + // cachedTree parameter is typed `unknown` at the LanguageProvider contract + // layer; cast here at the use site. // // Grammar selection: `.tsx` files are parsed with the TSX grammar, // `.ts` files with the TypeScript grammar. The two grammars have diff --git a/gitnexus/src/core/ingestion/parsing-processor.ts b/gitnexus/src/core/ingestion/parsing-processor.ts index 3a9bb6526..41c2b4d32 100644 --- a/gitnexus/src/core/ingestion/parsing-processor.ts +++ b/gitnexus/src/core/ingestion/parsing-processor.ts @@ -1,61 +1,20 @@ -import type { GraphNode, GraphRelationship, NodeLabel, ParameterTypeClass } from 'gitnexus-shared'; +import type { NodeLabel } from 'gitnexus-shared'; import { KnowledgeGraph } from '../graph/types.js'; -import Parser from 'tree-sitter'; -import { loadParser, loadLanguage, isLanguageAvailable } from '../tree-sitter/parser-loader.js'; -import { getProvider } from './languages/index.js'; -import { generateId } from '../../lib/utils.js'; -import type { SymbolTableReader, SymbolTableWriter } from './model/index.js'; -import { ASTCache } from './ast-cache.js'; -import { getLanguageFromFilename, SupportedLanguages } from 'gitnexus-shared'; -import { extractVueScript, isVueSetupTopLevel } from './vue-sfc-extractor.js'; -import { yieldToEventLoop } from './utils/event-loop.js'; -import { parseSourceSafe } from '../tree-sitter/safe-parse.js'; -import { isVerboseIngestionEnabled } from './utils/verbose.js'; -import { - buildConcreteTypedefDefinitionRanges, - getDefinitionNodeFromCaptures, - findEnclosingClassInfo, - findObjectLiteralBindingInfo, - getLabelFromCaptures, - isSuppressedConcreteTypedefDuplicate, - isQualifiableScopeLabel, - qualifyRustImplTargetByModScope, - CLASS_CONTAINER_TYPES, - type SyntaxNode, - type EnclosingClassInfo, -} from './utils/ast-helpers.js'; -import { detectFrameworkFromAST } from './framework-detection.js'; -import { buildTypeEnv } from './type-env.js'; -import type { FieldInfo, FieldExtractorContext } from './field-types.js'; -import type { VariableExtractorContext, VariableInfo } from './variable-types.js'; -import type { MethodInfo } from './method-types.js'; -import { - buildMethodProps, - arityForIdFromInfo, - typeTagForId, - constTagForId, - buildCollisionGroups, - parameterShapeIdTag, -} from './utils/method-props.js'; -import { - extractTemplateArguments, - templateArgumentsIdTag, - templateConstraintsIdTag, -} from './utils/template-arguments.js'; -import type { LanguageProvider } from './language-provider.js'; +import type { SymbolTableWriter } from './model/index.js'; +import { getLanguageFromFilename } from 'gitnexus-shared'; + +import { accumulateExportedTypesFromParsedNode, type ExportedTypeMap } from './call-processor.js'; + import type { ParsedFile } from 'gitnexus-shared'; import { WorkerPool } from './workers/worker-pool.js'; import { logger } from '../logger.js'; import type { ParseWorkerResult, ParseWorkerInput, - ExtractedCall, - ExtractedAssignment, ExtractedRoute, ExtractedFetchCall, ExtractedDecoratorRoute, ExtractedToolDef, - FileConstructorBindings, FileScopeBindings, ExtractedORMQuery, FetchWrapperDef, @@ -65,22 +24,10 @@ import type { ExtractedRouterInclude, ExtractedRouterModuleAlias, } from './route-extractors/fastapi-router-bindings.js'; -import { - getTreeSitterBufferSize, - getTreeSitterContentByteLength, - TREE_SITTER_MAX_BUFFER, -} from './constants.js'; -import { - ARRAY_METHOD_HOC_BLOCKLIST_SET, - DEFAULT_EXPORT_IDENTIFIER_BLOCKLIST_SET, - deriveDefaultExportHocName, -} from './ts-js-hoc-utils.js'; export type FileProgressCallback = (current: number, total: number, filePath: string) => void; export interface WorkerExtractedData { - calls: ExtractedCall[]; - assignments: ExtractedAssignment[]; routes: ExtractedRoute[]; fetchCalls: ExtractedFetchCall[]; fetchWrapperDefs: FetchWrapperDef[]; @@ -90,7 +37,6 @@ export interface WorkerExtractedData { routerModuleAliases: ExtractedRouterModuleAlias[]; toolDefs: ExtractedToolDef[]; ormQueries: ExtractedORMQuery[]; - constructorBindings: FileConstructorBindings[]; fileScopeBindings: FileScopeBindings[]; /** * Per-file `ParsedFile` artifacts from the new scope-based resolution @@ -110,7 +56,7 @@ export interface WorkerExtractedData { * Merge a list of `ParseWorkerResult`s into the running graph + symbol * table state and produce the chunk-aggregated `WorkerExtractedData`. * - * Extracted from `processParsingWithWorkers` so the same merge logic can + * Split out from the worker-parse path so the same merge logic can * be applied to both freshly-parsed worker output AND cached worker * output replayed during incremental analyze. Idempotent on the * accumulator fields (push-only); idempotent on graph if the caller @@ -121,9 +67,8 @@ export const mergeChunkResults = ( graph: KnowledgeGraph, symbolTable: SymbolTableWriter, chunkResults: readonly ParseWorkerResult[], + exportedTypeMap?: ExportedTypeMap, ): WorkerExtractedData => { - const allCalls: ExtractedCall[] = []; - const allAssignments: ExtractedAssignment[] = []; const allRoutes: ExtractedRoute[] = []; const allFetchCalls: ExtractedFetchCall[] = []; const allFetchWrapperDefs: FetchWrapperDef[] = []; @@ -133,7 +78,6 @@ export const mergeChunkResults = ( const allRouterModuleAliases: ExtractedRouterModuleAlias[] = []; const allToolDefs: ExtractedToolDef[] = []; const allORMQueries: ExtractedORMQuery[] = []; - const allConstructorBindings: FileConstructorBindings[] = []; const fileScopeBindingsByFile: FileScopeBindings[] = []; const allParsedFiles: ParsedFile[] = []; @@ -161,8 +105,11 @@ export const mergeChunkResults = ( qualifiedName: sym.qualifiedName, }); } - for (const item of result.calls) allCalls.push(item); - for (const item of result.assignments) allAssignments.push(item); + if (exportedTypeMap) { + for (const node of result.nodes) { + accumulateExportedTypesFromParsedNode(exportedTypeMap, node, symbolTable); + } + } for (const item of result.routes) allRoutes.push(item); for (const item of result.fetchCalls) allFetchCalls.push(item); for (const item of result.fetchWrapperDefs ?? []) allFetchWrapperDefs.push(item); @@ -172,15 +119,12 @@ export const mergeChunkResults = ( for (const item of result.routerModuleAliases ?? []) allRouterModuleAliases.push(item); for (const item of result.toolDefs) allToolDefs.push(item); if (result.ormQueries) for (const item of result.ormQueries) allORMQueries.push(item); - for (const item of result.constructorBindings) allConstructorBindings.push(item); if (result.fileScopeBindings) for (const item of result.fileScopeBindings) fileScopeBindingsByFile.push(item); if (result.parsedFiles) for (const item of result.parsedFiles) allParsedFiles.push(item); } return { - calls: allCalls, - assignments: allAssignments, routes: allRoutes, fetchCalls: allFetchCalls, fetchWrapperDefs: allFetchWrapperDefs, @@ -190,74 +134,55 @@ export const mergeChunkResults = ( routerModuleAliases: allRouterModuleAliases, toolDefs: allToolDefs, ormQueries: allORMQueries, - constructorBindings: allConstructorBindings, fileScopeBindings: fileScopeBindingsByFile, parsedFiles: allParsedFiles, }; }; -const processParsingWithWorkers = async ( - graph: KnowledgeGraph, +/** + * Dispatch a chunk's files to the worker pool and return the RAW per-worker + * results, WITHOUT merging them into the graph. Split out from + * {@link processParsing} so the parse loop can overlap one chunk's + * merge (main-thread, via {@link mergeChunkResults}) with the NEXT chunk's + * worker parse — the merge is the only remaining serial main-thread step once + * ParsedFile serialization moved into the workers (#worker-idle pipelining). + * Returns `[]` for an all-unparseable chunk (the caller merges `[]` → empty). + */ +export const dispatchChunkParse = async ( files: { path: string; content: string }[], - symbolTable: SymbolTableWriter, - astCache: ASTCache, workerPool: WorkerPool, onFileProgress?: FileProgressCallback, - /** - * When provided, populated with the raw worker results before merging. - * Used by the incremental-indexing parse cache to capture the per-chunk - * worker output for caching across runs. The mutation happens in-place - * so the caller (parse-impl) can keep a reference. See - * `gitnexus/src/storage/parse-cache.ts`. - */ + /** Populated in-place with the raw results (parse-cache capture). */ outRawResults?: ParseWorkerResult[], -): Promise => { - // Filter to parseable files only + /** + * Content hash of this parse chunk. When set, the workers tag their durable + * ParsedFile shards with it so a future warm cache hit can restore them + * (#2038). `undefined` ⇒ no durable write (tests / no-cache path). + */ + chunkHash?: string, +): Promise => { const parseableFiles: ParseWorkerInput[] = []; for (const file of files) { const lang = getLanguageFromFilename(file.path); if (lang) parseableFiles.push({ path: file.path, content: file.content }); } - - if (parseableFiles.length === 0) - return { - calls: [], - assignments: [], - routes: [], - fetchCalls: [], - fetchWrapperDefs: [], - decoratorRoutes: [], - routerIncludes: [], - routerImports: [], - routerModuleAliases: [], - toolDefs: [], - ormQueries: [], - constructorBindings: [], - fileScopeBindings: [], - parsedFiles: [], - }; + if (parseableFiles.length === 0) return []; const total = files.length; - - // Dispatch to worker pool — pool handles splitting into chunks and sub-batching const chunkResults = await workerPool.dispatch( parseableFiles, (filesProcessed) => { onFileProgress?.(Math.min(filesProcessed, total), total, 'Parsing...'); }, + chunkHash, ); - // Capture the raw chunk results for the incremental parse cache before - // merging — the cache stores the unmerged worker output so a future run - // can re-merge them into a fresh graph state. + // Capture raw results for the incremental parse cache before merging. if (outRawResults) { for (const r of chunkResults) outRawResults.push(r); } - // Merge results from all workers into graph and symbol table. - const merged = mergeChunkResults(graph, symbolTable, chunkResults); - - // Merge and log skipped languages from workers + // Skipped-language telemetry (worker output, independent of the merge). const skippedLanguages = new Map(); for (const result of chunkResults) { for (const [lang, count] of Object.entries(result.skippedLanguages)) { @@ -271,749 +196,8 @@ const processParsingWithWorkers = async ( logger.warn(` Skipped unsupported languages: ${summary}`); } - // Final progress onFileProgress?.(total, total, 'done'); - return merged; -}; - -// ============================================================================ -// Sequential fallback (original implementation) -// ============================================================================ - -// Inline caches to avoid repeated parent-walks per node (same pattern as parse-worker.ts). -// Keyed by tree-sitter node reference — cleared at the start of each file. -const classInfoCache = new Map(); -const exportCache = new Map(); - -const cachedFindEnclosingClassInfo = ( - node: SyntaxNode, - filePath: string, - resolveEnclosingOwner?: (node: SyntaxNode) => SyntaxNode | null, - getQualifiedOwnerName?: (node: SyntaxNode, simpleName: string) => string | null, -): EnclosingClassInfo | null => { - const cached = classInfoCache.get(node); - if (cached !== undefined) return cached; - const result = findEnclosingClassInfo( - node, - filePath, - resolveEnclosingOwner, - getQualifiedOwnerName, - ); - classInfoCache.set(node, result); - return result; -}; - -const cachedExportCheck = ( - checker: (node: SyntaxNode, name: string) => boolean, - node: SyntaxNode, - name: string, -): boolean => { - const cached = exportCache.get(node); - if (cached !== undefined) return cached; - const result = checker(node, name); - exportCache.set(node, result); - return result; -}; - -// FieldExtractor cache for sequential path — same pattern as parse-worker.ts -const seqFieldInfoCache = new Map>(); - -// MethodExtractor cache for sequential path — avoids re-traversing the same class -// body once per method. Keyed on classNode.id (tree-sitter node identity number). -const seqMethodExtractCache = new Map< - number, - { ownerName: string | undefined; methods: MethodInfo[] } | null ->(); -// Derived method map + collision groups cache — avoids rebuilding per method. -const seqMethodMapCache = new Map< - number, - { map: Map; groups: Map } ->(); - -/** Provider-aware enclosing container lookup. - * Walks up from `node` until a CLASS_CONTAINER_TYPES node is found. - * When `resolveEnclosingOwner` is provided, delegates language-specific - * container remapping (e.g., Ruby singleton_class → enclosing class). - * Without the hook, returns the first matching container directly (raw lookup). */ -function seqFindEnclosingOwnerNode( - node: SyntaxNode, - resolveEnclosingOwner?: (node: SyntaxNode) => SyntaxNode | null, -): SyntaxNode | null { - let current = node.parent; - while (current) { - if (CLASS_CONTAINER_TYPES.has(current.type)) { - if (resolveEnclosingOwner) { - const resolved = resolveEnclosingOwner(current); - if (resolved === null) { - // Provider says skip this container — keep walking up. - current = current.parent; - continue; - } - return resolved; - } - return current; - } - current = current.parent; - } - return null; -} - -/** Minimal no-op SymbolTable stub for sequential extractor contexts. The real - * SymbolTable is not fully populated yet at this stage, so use the stub for safety. - * Implements the full {@link SymbolTableReader} surface so future extractor additions - * don't silently fall off an `as unknown as` cast. */ -const NOOP_SYMBOL_TABLE_SEQ: SymbolTableReader = { - lookupExact: () => undefined, - lookupExactFull: () => undefined, - lookupExactAll: () => [], - lookupCallableByName: () => [], - getFiles: () => [][Symbol.iterator](), - getStats: () => ({ fileCount: 0 }), -}; - -function seqGetFieldInfo( - classNode: SyntaxNode, - provider: LanguageProvider, - context: FieldExtractorContext, -): Map | undefined { - if (!provider.fieldExtractor) return undefined; - const cacheKey = classNode.startIndex; - let cached = seqFieldInfoCache.get(cacheKey); - if (cached) return cached; - const extracted = provider.fieldExtractor.extract(classNode, context); - if (!extracted?.fields?.length) return undefined; - cached = new Map(); - for (const field of extracted.fields) cached.set(field.name, field); - seqFieldInfoCache.set(cacheKey, cached); - return cached; -} - -export const processParsingSequential = async ( - graph: KnowledgeGraph, - files: { path: string; content: string }[], - symbolTable: SymbolTableWriter, - astCache: ASTCache, - scopeTreeCache: ASTCache | undefined, - onFileProgress?: FileProgressCallback, -) => { - const parser = await loadParser(); - const total = files.length; - const logSkipped = isVerboseIngestionEnabled(); - const skippedByLang = logSkipped ? new Map() : null; - - for (let i = 0; i < files.length; i++) { - const file = files[i]; - - // Reset memoization before each new file (node refs are per-tree) - classInfoCache.clear(); - exportCache.clear(); - seqFieldInfoCache.clear(); - seqMethodExtractCache.clear(); - seqMethodMapCache.clear(); - const seqVariableInfoCache = new Map>(); - - onFileProgress?.(i + 1, total, file.path); - - if (i % 20 === 0) await yieldToEventLoop(); - - const language = getLanguageFromFilename(file.path); - - if (!language) continue; - if (!isLanguageAvailable(language)) { - if (skippedByLang) { - skippedByLang.set(language, (skippedByLang.get(language) ?? 0) + 1); - } - continue; - } - - // Skip files larger than the max tree-sitter buffer (32 MB) - if (getTreeSitterContentByteLength(file.content) > TREE_SITTER_MAX_BUFFER) continue; - - // Vue SFC preprocessing: extract + diff --git a/gitnexus/test/fixtures/vue-scope/vue-js-lang/App.vue b/gitnexus/test/fixtures/vue-scope/vue-js-lang/App.vue new file mode 100644 index 000000000..06c6e7026 --- /dev/null +++ b/gitnexus/test/fixtures/vue-scope/vue-js-lang/App.vue @@ -0,0 +1,8 @@ + + diff --git a/gitnexus/test/integration/resolvers/vue-scope.test.ts b/gitnexus/test/integration/resolvers/vue-scope.test.ts index 2a828316f..062e5ede0 100644 --- a/gitnexus/test/integration/resolvers/vue-scope.test.ts +++ b/gitnexus/test/integration/resolvers/vue-scope.test.ts @@ -473,3 +473,38 @@ describe('Vue cross-file composable and class resolution', () => { expect(files.some((f) => f.endsWith('models.ts'))).toBe(true); }); }); + +// --------------------------------------------------------------------------- +// F90 — dual-script merge ( - +`; + const result = extractVueScript(vue); + expect(result).not.toBeNull(); + expect(result!.lang).toBe('js'); + }); + + it('returns empty lang when no lang attribute', () => { + const vue = ` +`; + const result = extractVueScript(vue); + expect(result).not.toBeNull(); + expect(result!.lang).toBe(''); + }); + + it('jsx lang triggers JS grammar (maps to lang=js)', () => { + const vue = ` +`; + const result = extractVueScript(vue); + expect(result).not.toBeNull(); + expect(result!.lang).toBe('js'); + }); + + it('ts lang returns empty (only js/jsx triggers JS grammar)', () => { + const vue = ` +`; + const result = extractVueScript(vue); + expect(result).not.toBeNull(); + expect(result!.lang).toBe(''); + }); + + it('mixed js + ts blocks return empty lang (TypeScript wins)', () => { + const vue = ` + +`; + const result = extractVueScript(vue); + expect(result).not.toBeNull(); + expect(result!.lang).toBe(''); + }); + + it('isSetup is true when at least one block is setup', () => { + const vue = ` + +`; + const result = extractVueScript(vue); + expect(result).not.toBeNull(); + expect(result!.isSetup).toBe(true); + // ts blocks — lang should be empty + expect(result!.lang).toBe(''); }); it('returns null for .vue files with no - - diff --git a/gitnexus/src/core/ingestion/languages/csharp/index.ts b/gitnexus/src/core/ingestion/languages/csharp/index.ts index 9f06ce87d..5e8fa99ea 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/index.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/index.ts @@ -68,10 +68,9 @@ * `using static X = Y.Z;`, attributes, and preprocessor-gated * declarations are all recognized correctly. * - * Shadow-harness corpus parity is the authoritative signal for which - * of these matter in practice. The CI parity gate blocks any PR that - * regresses either the legacy or registry-primary run of - * `test/integration/resolvers/csharp.test.ts`. + * The `test/integration/resolvers/csharp.test.ts` resolver suite is the + * authoritative signal for which of these matter in practice; it runs in + * the standard CI test workflow, so a regression blocks the merge. */ export { emitCsharpScopeCaptures } from './captures.js'; diff --git a/gitnexus/src/core/ingestion/languages/php/index.ts b/gitnexus/src/core/ingestion/languages/php/index.ts index 9b549bc3d..de7b6345b 100644 --- a/gitnexus/src/core/ingestion/languages/php/index.ts +++ b/gitnexus/src/core/ingestion/languages/php/index.ts @@ -54,10 +54,9 @@ * 6. **Intersection types in parameters** — `T&U $param` takes the first * named part (`T`). This matches the legacy type-extractor's behavior. * - * Shadow-harness corpus parity is the authoritative signal for which of - * these matter in practice. The CI parity gate blocks any PR that regresses - * either the legacy or registry-primary run of - * `test/integration/resolvers/php.test.ts`. + * The `test/integration/resolvers/php.test.ts` resolver suite is the + * authoritative signal for which of these matter in practice; it runs in + * the standard CI test workflow, so a regression blocks the merge. */ export { emitPhpScopeCaptures } from './captures.js'; diff --git a/gitnexus/src/core/ingestion/languages/python/index.ts b/gitnexus/src/core/ingestion/languages/python/index.ts index 166c3a7ae..dc7e84f64 100644 --- a/gitnexus/src/core/ingestion/languages/python/index.ts +++ b/gitnexus/src/core/ingestion/languages/python/index.ts @@ -66,10 +66,9 @@ * site where the enclosing class can't be statically determined * is left unresolved. * - * Shadow-harness corpus parity is the authoritative signal for which - * of these matter in practice. The CI parity gate blocks any PR that - * regresses either the legacy or registry-primary run of - * `test/integration/resolvers/python.test.ts`. + * The `test/integration/resolvers/python.test.ts` resolver suite is the + * authoritative signal for which of these matter in practice; it runs in + * the standard CI test workflow, so a regression blocks the merge. */ export { emitPythonScopeCaptures } from './captures.js'; diff --git a/gitnexus/src/core/ingestion/languages/typescript/index.ts b/gitnexus/src/core/ingestion/languages/typescript/index.ts index cb4bd4023..120566f48 100644 --- a/gitnexus/src/core/ingestion/languages/typescript/index.ts +++ b/gitnexus/src/core/ingestion/languages/typescript/index.ts @@ -80,10 +80,9 @@ * identifiers are narrowed (`user instanceof User`). Member paths * such as `user.address instanceof Address` remain unresolved. * - * Shadow-harness corpus parity on `test/integration/resolvers/ - * typescript.test.ts` is the authoritative signal for which of these - * matter in practice. The CI parity gate blocks any PR that regresses - * either the legacy or registry-primary run. + * The `test/integration/resolvers/typescript.test.ts` resolver suite is + * the authoritative signal for which of these matter in practice; it runs + * in the standard CI test workflow, so a regression blocks the merge. */ export { emitTsScopeCaptures } from './captures.js'; diff --git a/gitnexus/test/unit/shadow/aggregate.test.ts b/gitnexus/test/unit/shadow/aggregate.test.ts deleted file mode 100644 index 00fb44652..000000000 --- a/gitnexus/test/unit/shadow/aggregate.test.ts +++ /dev/null @@ -1,234 +0,0 @@ -/** - * Unit tests for `aggregateDiffs` (RFC #909 Ring 2 SHARED #918). - * - * Covers bucketing by language, parity math (incl. zero-resolved edge), - * evidence-kind breakdown, and stable sort order on the output rows. - */ - -import { describe, it, expect } from 'vitest'; -import { - aggregateDiffs, - SupportedLanguages, - type LanguageParityRow, - type ResolutionEvidence, - type ShadowAgreement, - type ShadowDiff, -} from 'gitnexus-shared'; - -// ─── Fixtures ─────────────────────────────────────────────────────────────── - -const FIXED_NOW = new Date('2026-04-18T12:00:00.000Z'); - -const makeDiff = ( - agreement: ShadowAgreement, - evidenceKinds: readonly ResolutionEvidence['kind'][] = [], -): ShadowDiff => ({ - callsite: { filePath: 'src/x.ts', line: 1, col: 0, calledName: 'foo' }, - legacy: null, - newResult: null, - agreement, - evidenceDelta: evidenceKinds.map((kind) => ({ kind, weight: 0.3 })), -}); - -const entry = (language: SupportedLanguages, diff: ShadowDiff) => ({ language, diff }); - -const findRow = ( - rows: readonly LanguageParityRow[], - language: SupportedLanguages, -): LanguageParityRow => { - const row = rows.find((r) => r.language === language); - if (!row) throw new Error(`no row for ${language}`); - return row; -}; - -// ─── Empty input ──────────────────────────────────────────────────────────── - -describe('aggregateDiffs — empty input', () => { - it('returns empty perLanguage, zeroed overall, generatedAt populated', () => { - const report = aggregateDiffs([], FIXED_NOW); - expect(report.perLanguage).toEqual([]); - expect(report.overall).toEqual({ - totalCalls: 0, - bothAgree: 0, - onlyLegacy: 0, - onlyNew: 0, - bothDisagree: 0, - bothEmpty: 0, - parity: 0, - }); - expect(report.generatedAt).toBe('2026-04-18T12:00:00.000Z'); - }); -}); - -// ─── Single language, single outcome ──────────────────────────────────────── - -describe('aggregateDiffs — single language', () => { - it('all both-agree → parity = 1.0', () => { - const diffs = [ - entry(SupportedLanguages.Python, makeDiff('both-agree')), - entry(SupportedLanguages.Python, makeDiff('both-agree')), - entry(SupportedLanguages.Python, makeDiff('both-agree')), - ]; - const report = aggregateDiffs(diffs, FIXED_NOW); - expect(report.perLanguage).toHaveLength(1); - const row = findRow(report.perLanguage, SupportedLanguages.Python); - expect(row).toMatchObject({ - language: SupportedLanguages.Python, - totalCalls: 3, - bothAgree: 3, - onlyLegacy: 0, - onlyNew: 0, - bothDisagree: 0, - bothEmpty: 0, - parity: 1, - }); - }); - - it('mixed outcomes → parity excludes both-empty from denominator', () => { - const diffs = [ - entry(SupportedLanguages.TypeScript, makeDiff('both-agree')), - entry(SupportedLanguages.TypeScript, makeDiff('both-agree')), - entry(SupportedLanguages.TypeScript, makeDiff('only-legacy', ['global-name'])), - entry(SupportedLanguages.TypeScript, makeDiff('only-new', ['local'])), - entry(SupportedLanguages.TypeScript, makeDiff('both-disagree', ['import'])), - entry(SupportedLanguages.TypeScript, makeDiff('both-empty')), - entry(SupportedLanguages.TypeScript, makeDiff('both-empty')), - ]; - const report = aggregateDiffs(diffs, FIXED_NOW); - const row = findRow(report.perLanguage, SupportedLanguages.TypeScript); - expect(row.totalCalls).toBe(7); - expect(row.bothAgree).toBe(2); - expect(row.onlyLegacy).toBe(1); - expect(row.onlyNew).toBe(1); - expect(row.bothDisagree).toBe(1); - expect(row.bothEmpty).toBe(2); - // parity = bothAgree / (totalCalls - bothEmpty) = 2 / (7 - 2) = 0.4 - expect(row.parity).toBeCloseTo(0.4, 10); - }); - - it('all both-empty → parity = 0 (not NaN)', () => { - const diffs = [ - entry(SupportedLanguages.Java, makeDiff('both-empty')), - entry(SupportedLanguages.Java, makeDiff('both-empty')), - ]; - const report = aggregateDiffs(diffs, FIXED_NOW); - const row = findRow(report.perLanguage, SupportedLanguages.Java); - expect(row.totalCalls).toBe(2); - expect(row.bothEmpty).toBe(2); - expect(row.parity).toBe(0); - expect(Number.isNaN(row.parity)).toBe(false); - }); -}); - -// ─── Multi-language ───────────────────────────────────────────────────────── - -describe('aggregateDiffs — multiple languages', () => { - it('buckets rows by language and sums overall column-wise', () => { - const diffs = [ - entry(SupportedLanguages.Python, makeDiff('both-agree')), - entry(SupportedLanguages.Python, makeDiff('both-disagree', ['local'])), - entry(SupportedLanguages.Ruby, makeDiff('both-agree')), - entry(SupportedLanguages.Ruby, makeDiff('both-agree')), - entry(SupportedLanguages.Ruby, makeDiff('only-new', ['type-binding'])), - ]; - const report = aggregateDiffs(diffs, FIXED_NOW); - expect(report.perLanguage).toHaveLength(2); - - const python = findRow(report.perLanguage, SupportedLanguages.Python); - expect(python.totalCalls).toBe(2); - expect(python.bothAgree).toBe(1); - expect(python.bothDisagree).toBe(1); - expect(python.parity).toBe(0.5); - - const ruby = findRow(report.perLanguage, SupportedLanguages.Ruby); - expect(ruby.totalCalls).toBe(3); - expect(ruby.bothAgree).toBe(2); - expect(ruby.onlyNew).toBe(1); - expect(ruby.parity).toBeCloseTo(2 / 3, 10); - - expect(report.overall).toEqual({ - totalCalls: 5, - bothAgree: 3, - onlyLegacy: 0, - onlyNew: 1, - bothDisagree: 1, - bothEmpty: 0, - parity: 3 / 5, - }); - }); - - it('perLanguage rows are sorted alphabetically by language value for stable output', () => { - const diffs = [ - entry(SupportedLanguages.TypeScript, makeDiff('both-agree')), - entry(SupportedLanguages.C, makeDiff('both-agree')), - entry(SupportedLanguages.Python, makeDiff('both-agree')), - entry(SupportedLanguages.Java, makeDiff('both-agree')), - ]; - const report = aggregateDiffs(diffs, FIXED_NOW); - const ordered = report.perLanguage.map((r) => r.language); - // Alphabetical by enum VALUE: 'c' < 'java' < 'python' < 'typescript' - expect(ordered).toEqual([ - SupportedLanguages.C, - SupportedLanguages.Java, - SupportedLanguages.Python, - SupportedLanguages.TypeScript, - ]); - }); -}); - -// ─── Evidence breakdown ───────────────────────────────────────────────────── - -describe('aggregateDiffs — evidence breakdown', () => { - it('counts divergence evidence kinds across non-agreeing rows only', () => { - const diffs = [ - entry(SupportedLanguages.Go, makeDiff('both-disagree', ['import', 'owner-match'])), - entry(SupportedLanguages.Go, makeDiff('only-legacy', ['import', 'global-name'])), - entry(SupportedLanguages.Go, makeDiff('only-new', ['local'])), - // both-agree contributes 0 to evidence breakdown regardless of any attached evidence - entry(SupportedLanguages.Go, makeDiff('both-agree', ['import'])), - // both-empty also contributes 0 - entry(SupportedLanguages.Go, makeDiff('both-empty')), - ]; - const report = aggregateDiffs(diffs, FIXED_NOW); - const row = findRow(report.perLanguage, SupportedLanguages.Go); - expect(Array.from(row.evidenceBreakdown.entries())).toEqual([ - ['global-name', 1], - ['import', 2], - ['local', 1], - ['owner-match', 1], - ]); - }); - - it('emits empty evidenceBreakdown when all calls agree or are empty', () => { - const diffs = [ - entry(SupportedLanguages.Rust, makeDiff('both-agree')), - entry(SupportedLanguages.Rust, makeDiff('both-empty')), - ]; - const report = aggregateDiffs(diffs, FIXED_NOW); - const row = findRow(report.perLanguage, SupportedLanguages.Rust); - expect(row.evidenceBreakdown.size).toBe(0); - }); -}); - -// ─── Determinism ──────────────────────────────────────────────────────────── - -describe('aggregateDiffs — determinism', () => { - it('injected `now` is used verbatim for generatedAt', () => { - const t = new Date('2030-01-01T00:00:00.000Z'); - const report = aggregateDiffs([], t); - expect(report.generatedAt).toBe('2030-01-01T00:00:00.000Z'); - }); - - it('same input produces byte-identical JSON (stable keys + sort)', () => { - const diffs = [ - entry(SupportedLanguages.Python, makeDiff('both-disagree', ['local', 'import'])), - entry(SupportedLanguages.Java, makeDiff('both-agree')), - ]; - const a = aggregateDiffs(diffs, FIXED_NOW); - const b = aggregateDiffs(diffs, FIXED_NOW); - // Round-trip through JSON to drop Map identity and force structural comparison. - const toJson = (r: typeof a): string => - JSON.stringify(r, (_key, v: unknown) => (v instanceof Map ? Object.fromEntries(v) : v)); - expect(toJson(a)).toBe(toJson(b)); - }); -}); diff --git a/gitnexus/test/unit/shadow/diff.test.ts b/gitnexus/test/unit/shadow/diff.test.ts deleted file mode 100644 index e9baa606f..000000000 --- a/gitnexus/test/unit/shadow/diff.test.ts +++ /dev/null @@ -1,192 +0,0 @@ -/** - * Unit tests for `diffResolutions` (RFC #909 Ring 2 SHARED #918). - * - * Pins the 5 `ShadowAgreement` outcomes and the symmetric-by-kind evidence- - * delta contract. Inputs are pure data fixtures — no real pipeline state. - */ - -import { describe, it, expect } from 'vitest'; -import { - diffResolutions, - type Resolution, - type ResolutionEvidence, - type ShadowCallsite, - type SymbolDefinition, -} from 'gitnexus-shared'; - -// ─── Fixtures ─────────────────────────────────────────────────────────────── - -const callsite: ShadowCallsite = { - filePath: 'src/app.ts', - line: 42, - col: 8, - calledName: 'save', -}; - -const makeDef = (nodeId: string): SymbolDefinition => ({ - nodeId, - filePath: 'src/models.ts', - type: 'Method', -}); - -const makeEvidence = (kind: ResolutionEvidence['kind'], weight = 0.5): ResolutionEvidence => ({ - kind, - weight, -}); - -const makeResolution = ( - nodeId: string, - evidenceKinds: readonly ResolutionEvidence['kind'][], -): Resolution => ({ - def: makeDef(nodeId), - confidence: Math.min(1, evidenceKinds.length * 0.3), - evidence: evidenceKinds.map((k) => makeEvidence(k)), -}); - -// ─── Agreement outcomes ───────────────────────────────────────────────────── - -describe('diffResolutions — agreement outcomes', () => { - it("both arrays empty → 'both-empty' with no evidence delta", () => { - const result = diffResolutions(callsite, [], []); - expect(result.agreement).toBe('both-empty'); - expect(result.evidenceDelta).toEqual([]); - expect(result.legacy).toBeNull(); - expect(result.newResult).toBeNull(); - }); - - it("identical top DefIds → 'both-agree' with empty evidence delta", () => { - const legacy = [makeResolution('def:User.save', ['local', 'owner-match'])]; - const next = [makeResolution('def:User.save', ['local', 'kind-match'])]; - const result = diffResolutions(callsite, legacy, next); - expect(result.agreement).toBe('both-agree'); - expect(result.evidenceDelta).toEqual([]); - expect(result.legacy).toBe(legacy[0]); - expect(result.newResult).toBe(next[0]); - }); - - it("legacy empty, new non-empty → 'only-new' with new's evidence as delta", () => { - const next = [makeResolution('def:User.save', ['local', 'owner-match'])]; - const result = diffResolutions(callsite, [], next); - expect(result.agreement).toBe('only-new'); - expect(result.evidenceDelta).toEqual(next[0].evidence); - expect(result.legacy).toBeNull(); - expect(result.newResult).toBe(next[0]); - }); - - it("legacy non-empty, new empty → 'only-legacy' with legacy's evidence as delta", () => { - const legacy = [makeResolution('def:User.save', ['global-name'])]; - const result = diffResolutions(callsite, legacy, []); - expect(result.agreement).toBe('only-legacy'); - expect(result.evidenceDelta).toEqual(legacy[0].evidence); - expect(result.legacy).toBe(legacy[0]); - expect(result.newResult).toBeNull(); - }); - - it("different top DefIds → 'both-disagree'", () => { - const legacy = [makeResolution('def:ModelA.save', ['global-name'])]; - const next = [makeResolution('def:ModelB.save', ['local'])]; - const result = diffResolutions(callsite, legacy, next); - expect(result.agreement).toBe('both-disagree'); - expect(result.legacy).toBe(legacy[0]); - expect(result.newResult).toBe(next[0]); - }); -}); - -// ─── Evidence delta — symmetric difference by `kind` ──────────────────────── - -describe('diffResolutions — evidence delta (symmetric-by-kind)', () => { - it("'both-disagree' with disjoint evidence → delta contains both sides' kinds", () => { - const legacy = [makeResolution('def:A', ['global-name'])]; - const next = [makeResolution('def:B', ['local', 'owner-match'])]; - const result = diffResolutions(callsite, legacy, next); - expect(result.evidenceDelta.map((e) => e.kind)).toEqual([ - 'global-name', - 'local', - 'owner-match', - ]); - }); - - it("'both-disagree' with overlapping kinds → overlapping kinds removed from delta", () => { - const legacy = [makeResolution('def:A', ['local', 'scope-chain', 'global-name'])]; - const next = [makeResolution('def:B', ['local', 'import', 'owner-match'])]; - const result = diffResolutions(callsite, legacy, next); - // 'local' is on both sides → dropped - // Remaining: legacy-only ['scope-chain', 'global-name'], then new-only ['import', 'owner-match'] - expect(result.evidenceDelta.map((e) => e.kind)).toEqual([ - 'scope-chain', - 'global-name', - 'import', - 'owner-match', - ]); - }); - - it("'both-disagree' with fully overlapping kinds → empty evidence delta", () => { - const legacy = [makeResolution('def:A', ['local', 'owner-match'])]; - const next = [makeResolution('def:B', ['owner-match', 'local'])]; - const result = diffResolutions(callsite, legacy, next); - // Same kind set, different order → symmetric difference is empty - expect(result.evidenceDelta).toEqual([]); - expect(result.agreement).toBe('both-disagree'); // agreement still disagrees because nodeIds differ - }); - - it('differing weights on the same kind → NOT a delta (keyed on kind only)', () => { - const legacy = [ - { - def: makeDef('def:A'), - confidence: 0.9, - evidence: [{ kind: 'local' as const, weight: 0.55 }], - }, - ]; - const next = [ - { - def: makeDef('def:B'), - confidence: 0.1, - evidence: [{ kind: 'local' as const, weight: 0.25 }], - }, - ]; - const result = diffResolutions(callsite, legacy, next); - expect(result.agreement).toBe('both-disagree'); - expect(result.evidenceDelta).toEqual([]); - }); -}); - -// ─── Metadata + ordering ──────────────────────────────────────────────────── - -describe('diffResolutions — metadata + ordering', () => { - it('ignores resolutions beyond index 0 (top match only)', () => { - const legacy = [ - makeResolution('def:User.save', ['local']), - makeResolution('def:other', ['global-name']), - ]; - const next = [ - makeResolution('def:User.save', ['local']), - // The 2nd entry is here to verify index-0 isolation — the only kind - // requirement is that it be a valid `ResolutionEvidence.kind` so the - // fixture is type-correct. `'global-name'` is a real kind that - // `diffResolutions` never treats specially. - makeResolution('def:yet-another', ['global-name']), - ]; - const result = diffResolutions(callsite, legacy, next); - expect(result.agreement).toBe('both-agree'); - }); - - it('preserves callsite verbatim', () => { - const result = diffResolutions(callsite, [], []); - expect(result.callsite).toBe(callsite); - }); - - it("'both-disagree' delta order: legacy-only first (input order), then new-only", () => { - const legacy = [makeResolution('def:A', ['owner-match', 'scope-chain', 'kind-match'])]; - const next = [makeResolution('def:B', ['import', 'owner-match', 'arity-match'])]; - const result = diffResolutions(callsite, legacy, next); - // 'owner-match' overlaps → dropped - // legacy-only in original order: ['scope-chain', 'kind-match'] - // then new-only in original order: ['import', 'arity-match'] - expect(result.evidenceDelta.map((e) => e.kind)).toEqual([ - 'scope-chain', - 'kind-match', - 'import', - 'arity-match', - ]); - }); -}); From 3963c497dd2a150ff607122a250adcc809c4d49d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 8 Jun 2026 07:20:12 +0100 Subject: [PATCH 09/17] fix(parse): correct worker-pool docs drift + surface worker-side stack on crash (#2068) (#2070) --- ARCHITECTURE.md | 14 +- README.md | 4 +- gitnexus/bench/parse-throughput.md | 43 +++-- gitnexus/package-lock.json | 1 - gitnexus/package.json | 1 - gitnexus/src/cli/analyze.ts | 16 ++ gitnexus/src/core/ingestion/ast-cache.ts | 77 -------- gitnexus/src/core/ingestion/call-processor.ts | 88 +-------- .../src/core/ingestion/export-detection.ts | 3 +- .../src/core/ingestion/filesystem-walker.ts | 22 --- .../ingestion/languages/python/captures.ts | 7 +- .../pipeline-phases/orm-extraction.ts | 106 ----------- .../ingestion/pipeline-phases/parse-impl.ts | 11 -- .../core/ingestion/workers/parse-worker.ts | 18 +- .../src/core/ingestion/workers/worker-pool.ts | 38 +++- gitnexus/src/core/lbug/lbug-adapter.ts | 33 +++- gitnexus/src/core/lbug/pool-adapter.ts | 56 +++++- gitnexus/src/core/lbug/query-result-utils.ts | 31 +++ gitnexus/src/mcp/local/local-backend.ts | 15 +- gitnexus/src/types/pipeline.ts | 9 +- gitnexus/test/integration/lbug-pool.test.ts | 80 ++++++++ gitnexus/test/unit/ast-cache.test.ts | 95 ---------- .../test/unit/lbug-query-result-utils.test.ts | 45 +++++ .../test/unit/lbug-readonly-error.test.ts | 28 +++ .../test/unit/worker-pool-error-stack.test.ts | 176 ++++++++++++++++++ 25 files changed, 562 insertions(+), 455 deletions(-) delete mode 100644 gitnexus/src/core/ingestion/ast-cache.ts delete mode 100644 gitnexus/src/core/ingestion/pipeline-phases/orm-extraction.ts create mode 100644 gitnexus/src/core/lbug/query-result-utils.ts delete mode 100644 gitnexus/test/unit/ast-cache.test.ts create mode 100644 gitnexus/test/unit/lbug-query-result-utils.test.ts create mode 100644 gitnexus/test/unit/worker-pool-error-stack.test.ts diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index d90d132f6..01013c71c 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -15,7 +15,7 @@ Monorepo: **CLI/MCP** (`gitnexus/`) + **browser UI** (`gitnexus-web/`). ## End-to-end flow: index → graph → tools -1. **Ingestion** — `analyze.ts` → `runFullAnalysis` (`run-analyze.ts`) → `runPipelineFromRepo` (`pipeline.ts`). DAG of 12 phases builds a `KnowledgeGraph` in memory, then loads into LadybugDB under `.gitnexus/`. Repo registered in `~/.gitnexus/registry.json` for MCP discovery. +1. **Ingestion** — `analyze.ts` → `runFullAnalysis` (`run-analyze.ts`) → `runPipelineFromRepo` (`pipeline.ts`). DAG of 14 phases builds a `KnowledgeGraph` in memory, then loads into LadybugDB under `.gitnexus/`. Repo registered in `~/.gitnexus/registry.json` for MCP discovery. 2. **Persistence** — `repo-manager.ts` (paths, registry, KuzuDB cleanup). `lbug-adapter.ts` (graph load, queries, embedding batches). @@ -101,7 +101,7 @@ scan → structure → [markdown, cobol] → parse → [routes, tools, orm] | `communities` | `communities.ts` | `mro`, `pruneLocalSymbols`, `structure` | Community nodes + MEMBER_OF edges (Leiden algorithm) | | `processes` | `processes.ts` | `communities`, `routes`, `tools`, `pruneLocalSymbols`, `structure` | Process nodes + STEP_IN_PROCESS edges | -**Non-phase files in the same directory:** `parse-impl.ts`, `cross-file-impl.ts` (implementation), `wildcard-synthesis.ts` (whole-module import expansion), `orm-extraction.ts` (sequential ORM fallback), `types.ts`, `runner.ts`, `index.ts`. +**Non-phase files in the same directory:** `parse-impl.ts`, `cross-file-impl.ts` (implementation), `wildcard-synthesis.ts` (whole-module import expansion), `types.ts`, `runner.ts`, `index.ts`. ### DAG runner @@ -121,7 +121,7 @@ scan → structure → [markdown, cobol] → parse → [routes, tools, orm] - **Single graph accumulator** — all phases mutate the same `KnowledgeGraph` in `ctx`; the graph is the primary output. - **Typed phase access** — `getPhaseOutput(deps, 'name')` for type-safe upstream results. - **Binding accumulator lifecycle** — created in `parse`, disposed by `crossFile` (in `finally`). No other phase should take ownership. -- **Skippable phases** — `skipGraphPhases` omits MRO/communities/processes (faster tests); `pruneLocalSymbols` still runs (it is graph cleanup, not analysis). `skipWorkers` forces sequential parsing. +- **Skippable phases** — `skipGraphPhases` omits MRO/communities/processes (faster tests); `pruneLocalSymbols` still runs (it is graph cleanup, not analysis). `skipWorkers` is no longer a sequential escape hatch — it (like `--workers 0` / `GITNEXUS_WORKER_POOL_SIZE=0`) is rejected with an actionable error, since the worker pool is the sole parse path (§ Chunked parse-and-resolve). - **Local-symbol pruning** — `pruneLocalSymbols` removes inert block-local value symbols after scope resolution has consumed them. Opt out per-call with `PipelineOptions.keepLocalValueSymbols` or globally with the `GITNEXUS_KEEP_LOCAL_VALUE_SYMBOLS` env var. ### How to add a new phase @@ -202,7 +202,7 @@ Language-agnostic scope-resolution resolver. This is the resolution path for eve ``` Orchestrator: `runScopeResolution(input, provider)` in `scope-resolution/pipeline/run.ts`. -Pipeline phase: `scopeResolutionPhase` in `scope-resolution/pipeline/phase.ts` — iterates the registered `SCOPE_RESOLVERS`, reads per-file Trees from the parse phase's `scopeTreeCache`, disposes the cache at the end. +Pipeline phase: `scopeResolutionPhase` in `scope-resolution/pipeline/phase.ts` — iterates the registered `SCOPE_RESOLVERS` over the worker-serialized `ParsedFile`s. (Per-language `emitScopeCaptures` hooks may reuse a cached Tree via the orchestrator's `treeCache`, but in worker-pool runs that cache is empty — Trees can't cross MessageChannels — so they consume the pre-extracted `ParsedFile` instead; § Performance notes.) ### `ScopeResolver` contract @@ -251,7 +251,7 @@ CI auto-discovers the set via `tsx`. No workflow edit required. ### Performance notes -- **Cross-phase Tree cache**: parse phase writes Trees into `scopeTreeCache` (separate from the chunk-local `astCache`) ONLY for languages with `emitScopeCaptures`. Scope-resolution reads from it to skip the second parse. Cleared at end of the phase. Workers leave the cache empty — Trees can't cross MessageChannels; cache miss = fresh parse. `PROF_SCOPE_RESOLUTION=1` emits hit/miss counters and a worker-engaged warning. +- **Cross-phase Tree cache**: the orchestrator's `treeCache` (`RunScopeResolutionInput.treeCache`) lets a scope-resolution per-language hook (`emitScopeCaptures`) reuse a tree instead of re-parsing. Workers leave it empty — Trees can't cross MessageChannels — so in normal (worker-pool) runs scope-resolution does NOT rely on it: workers serialize each file's `ParsedFile` (+ capture side-channel) and stream them in, so scope-resolution consumes the pre-extracted artifact rather than re-parsing on the main thread (§ Chunked parse-and-resolve). `PROF_SCOPE_RESOLUTION=1` emits hit/miss counters and a worker-engaged warning. - **Typed relationship iteration**: heritage + MRO walk only the EXTENDS / IMPLEMENTS / HAS_METHOD edges via `iterRelationshipsByType`, not the full relationship map. - **Workspace-resolution-index**: O(1) `findOwnedMember` / `findExportedDef` / `classScopeByDefId` built once per run. - **SCC-ordered cross-file return-type propagation** (PR #1050): `propagateImportedReturnTypes` walks `indexes.sccs` in reverse-topological order (leaves first), so multi-hop alias chains like `models.User → service.user → app.user` collapse to the terminal class in a single linear pass. Within each importer, the source module's `typeBindings` is chain-followed BEFORE mirroring (so we mirror terminal types, not intermediate refs), and the importer's own `typeBindings` is chain-followed AFTER mirroring (so local `const x = importedFn()` resolves before downstream importers run). Cyclic SCCs reach a partial fixpoint within a single pass without iterating to convergence — see the `ts-circular` cross-file-binding fixture which only asserts pipeline-no-throw. PROF output (`PROF_SCOPE_RESOLUTION=1`) splits `finalize` from `propagate` so quadratic regressions in the chain-follow surface independently. @@ -314,7 +314,7 @@ Unified 3-tier algorithm (`model/resolution-context.ts`), per-language `importSe ### Chunked parse-and-resolve `parse` processes files in ~20 MB byte-budget chunks to bound memory. Per chunk: -1. Worker pool dispatches files (or sequential fallback via `skipWorkers`) +1. Worker pool dispatches files (the sole parse path — there is no sequential fallback; `skipWorkers`, `--workers 0`, and `GITNEXUS_WORKER_POOL_SIZE=0` are rejected with an actionable error) 2. Each worker: detect language → load grammar → run queries → return unified `ParseWorkerResult` 3. Synthesize wildcard bindings (`wildcard-synthesis.ts`) 4. Resolve imports @@ -324,6 +324,8 @@ Inheritance edges are emitted later, by the scope-resolution phase (`preEmitInhe Workers: `workers/worker-pool.ts`, `workers/parse-worker.ts`. +**Worker-serialized ParsedFiles (#2038).** To index very large repos (e.g. the Linux kernel) without OOM, the worker pool is the *sole* parse path and workers serialize each file's `ParsedFile` (plus its capture side-channel) in parallel, streaming them to scope-resolution through a disk-backed store. Scope-resolution consumes the pre-extracted artifact instead of re-parsing every file on the main thread — tree-sitter's native input buffers are not GC-reclaimable, so the former main-thread re-parse leaked native memory until the process died. Pool creation is lazy / cache-miss-gated, so a warm all-cache-hit run replays cached worker output without spawning a worker (hence `usedWorkerPool` can be false even when the repo has parseable files). + ### Inheritance and MRO Inheritance is captured by the `@reference.inherits` tag and emitted by the scope-resolution phase: `preEmitInheritanceEdges` resolves each base in scope, then `emitHeritageEdges` writes the `EXTENDS`/`IMPLEMENTS` edges. The phase then computes method resolution order via each `ScopeResolver`'s `buildMro` hook, feeding a `MethodDispatchIndex` used for owner-scoped lookups. Per-language strategy: diff --git a/README.md b/README.md index 28229c98c..2df823d9a 100644 --- a/README.md +++ b/README.md @@ -236,7 +236,7 @@ gitnexus analyze --embeddings [limit] # Enable embedding generation (slower, be gitnexus analyze --verbose # Log skipped files when parsers are unavailable gitnexus analyze --worker-timeout 60 # Increase worker idle timeout for slow parses gitnexus analyze --wal-checkpoint-threshold 67108864 # 64 MiB. Control LadybugDB WAL auto-checkpoint threshold (default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB) -gitnexus analyze --workers # Parse worker pool size (default: cores-1, capped at 16; 0 = sequential) +gitnexus analyze --workers # Parse worker pool size (>=1; default: cores-1, capped at 16, auto-sized to the repo). 0 is rejected — there is no sequential mode. gitnexus mcp # Start MCP server (stdio) — serves all indexed repos gitnexus serve # Start local HTTP server (multi-repo) for web UI connection gitnexus list # List all indexed repositories @@ -314,7 +314,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max | Variable | Default | Effect | Tune when… | | -------------------------------------- | ------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------- | -| `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size. `0` disables the pool (sequential fallback). Equivalent to `--workers `. | Constrained containers (cgroup CPU limits), CI runners with explicit quotas, or debugging a worker-only crash via `0`. | +| `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size (must be ≥ 1). Equivalent to `--workers `. The worker pool is the sole parse path — there is no sequential parser, so `0` is rejected with an actionable error (the pool self-heals via quarantine + respawn). | Constrained containers (cgroup CPU limits) or CI runners with explicit quotas. To narrow down a worker crash set `1` for a single-worker pool — not `0`. | | `GITNEXUS_PARSE_CHUNK_CONCURRENCY` | `2` | Number of chunks whose file contents may be read into memory in parallel while the pool dispatches the current chunk. Worker dispatch itself stays serial. | Repos large enough to chunk (multi-MB total source) where disk I/O is a measurable fraction of analyze wall-clock. | | `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. | | `GITNEXUS_PROFILE_DEFERRED` | unset | When `1`, emits `[deferred-profile]` timing/progress logs for the post-chunk deferred resolution band (imports → heritage → buildHeritageMap → legacy call resolution). Implied by `GITNEXUS_VERBOSE`. | Diagnosing analyze stalls in "Resolving calls (all chunks)" on large Java/Kotlin repos (issue #1741) without the full verbose ingestion noise. | diff --git a/gitnexus/bench/parse-throughput.md b/gitnexus/bench/parse-throughput.md index 24a7c98b1..4d53cd2dc 100644 --- a/gitnexus/bench/parse-throughput.md +++ b/gitnexus/bench/parse-throughput.md @@ -70,10 +70,17 @@ this doc, run it under instrumentation: ```bash # From the gitnexus/ subdir: cd gitnexus -# Single-threaded baseline (sequential fallback): -npx vitest run test/integration/parse-impl-large-fixture.test.ts --reporter=verbose +# The worker pool is the sole parse path, so every run needs the dist worker +# (`npm run build`) and a pool size pinned via GITNEXUS_WORKER_POOL_SIZE. -# Worker-pool path (requires built dist/ — pre-built by `npm run build`): +# Single-worker-pool baseline (closest analog to the old single-threaded run — +# sequential parsing was removed, so a 1-worker pool is the floor): +npm run build && \ + GITNEXUS_WORKER_POOL_SIZE=1 \ + GITNEXUS_VERBOSE=1 \ + npx vitest run test/integration/parse-impl-large-fixture.test.ts --reporter=verbose + +# Multi-worker path: npm run build && \ GITNEXUS_WORKER_POOL_SIZE=4 \ GITNEXUS_PARSE_CHUNK_CONCURRENCY=2 \ @@ -97,22 +104,26 @@ node --inspect=0 \ ## Latest measurement > _No measurement data has been collected yet — this file is the -> methodology + harness scaffold. The single recorded data point is the -> U6 wall-clock smoke baseline below; the worker-pool rows are -> placeholders for future bench-pass output._ +> methodology + harness scaffold. The U6 smoke test confirms the +> worker-pool path stays well within its wall-clock budget, but every +> throughput/heap cell below is a `_TBD_` placeholder for a future +> bench-pass._ The U6 integration test (`gitnexus/test/integration/parse-impl-large-fixture.test.ts`) -was observed completing the synthetic fixture in **~6 seconds** under -the sequential path (`skipWorkers: true`) on the development machine, -well under the 30 s `Promise.race` wall-clock budget. That number is a -smoke baseline only — recorded here for reference, not as a regression -target. +runs the worker pool — the sole parse path now that sequential parsing +has been removed (disabling the pool on a repo with parseable files +raises a hard `WorkerPoolDisabledError`). It completes the synthetic +fixture well within the 30 s `Promise.race` wall-clock budget on the +development machine, but no worker-pool throughput/heap numbers have been +captured yet, so the rows below are all `_TBD_`. (An earlier ~6 s figure +recorded here was measured on the now-removed sequential path; it has +been dropped rather than relabelled as a worker-pool baseline, since the +two paths are not comparable.) -| Path | files/s | wall-clock | peak heap | chunks | quarantined | -| ------------------------------------------ | ------- | -------------------- | --------- | ------ | ----------- | -| Sequential (`skipWorkers: true`, U6 smoke) | _TBD_ | ~6 s _(observation)_ | _TBD_ | 17 | 0 | -| Worker pool, `--workers 4`, concurrency 2 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 | -| Worker pool, `--workers 1`, concurrency 1 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 | +| Path | files/s | wall-clock | peak heap | chunks | quarantined | +| ------------------------------------------------------------------------- | ------- | ---------- | --------- | ------ | ----------- | +| Worker pool, `--workers 1` (`GITNEXUS_WORKER_POOL_SIZE=1`), concurrency 1 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 | +| Worker pool, `--workers 4`, concurrency 2 | _TBD_ | _TBD_ | _TBD_ | _TBD_ | 0 | **Hardware:** _TBD — record OS, CPU, RAM, Node version, gitnexus SHA at the time of the bench-pass that populates the table above._ diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 9ec0598f9..83534ed77 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -26,7 +26,6 @@ "ignore": "^7.0.5", "js-yaml": "^4.1.1", "jsonc-parser": "^3.3.1", - "lru-cache": "^11.0.0", "mnemonist": "^0.40.3", "onnxruntime-node": "^1.24.0", "pandemonium": "^2.4.0", diff --git a/gitnexus/package.json b/gitnexus/package.json index 54c0b5d5f..1dbb1ed80 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -70,7 +70,6 @@ "ignore": "^7.0.5", "js-yaml": "^4.1.1", "jsonc-parser": "^3.3.1", - "lru-cache": "^11.0.0", "mnemonist": "^0.40.3", "onnxruntime-node": "^1.24.0", "pandemonium": "^2.4.0", diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index 70bc9a566..3f196b11d 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -60,6 +60,22 @@ const writeFatalToStderr = (label: string, err: unknown): void => { const message = isErr ? err.message : String(err); realStderrWrite(`\n ${label}: ${message}\n`); if (isErr && err.stack) realStderrWrite(`${err.stack}\n`); + // Walk and print the `cause` chain. The phase runner wraps the underlying + // failure as `new Error("Phase 'X' failed: …", { cause })`, so the original + // error (e.g. a WorkerPoolDispatchError carrying the worker-side stack from + // #2068) is only reachable via `.cause`. Without this the user sees the + // wrapper's main-thread stack and never the real frame. `cause.stack` already + // begins with the cause's message, so we print the stack alone (not message + + // stack) to avoid repeating it. Depth-bounded so a cyclic `cause` can't loop + // (the phase runner wraps one level; the bound leaves headroom for future + // nesting); uses realStderrWrite so the redirected console.error's ANSI + // clear-line wrapping can't erase it (#1169). + const MAX_CAUSE_DEPTH = 5; + let cause: unknown = isErr ? (err as { cause?: unknown }).cause : undefined; + for (let depth = 0; depth < MAX_CAUSE_DEPTH && cause instanceof Error; depth++) { + realStderrWrite(`\n Caused by: ${cause.stack ?? cause.message}\n`); + cause = (cause as { cause?: unknown }).cause; + } }; let fatalHandlersInstalled = false; diff --git a/gitnexus/src/core/ingestion/ast-cache.ts b/gitnexus/src/core/ingestion/ast-cache.ts deleted file mode 100644 index 454c60df2..000000000 --- a/gitnexus/src/core/ingestion/ast-cache.ts +++ /dev/null @@ -1,77 +0,0 @@ -import { LRUCache } from 'lru-cache'; -import Parser from 'tree-sitter'; - -import { logger } from '../logger.js'; -/** - * Minimal structural shape consumers need when reading Trees back - * through a phase-dependency boundary. Declared here so phases that - * receive ASTCache via `getPhaseOutput<...>` don't hand-roll their - * own inline structural types that silently drift when ASTCache's - * contract changes. - * - * Typed as `unknown` at the Tree boundary because consumers on the - * other side of the phase-output map don't share tree-sitter's type - * graph (e.g. COBOL's standalone processor). - */ -export interface ASTCacheReader { - get(filePath: string): unknown; - clear(): void; -} - -// Define the interface for the Cache -export interface ASTCache extends ASTCacheReader { - get: (filePath: string) => Parser.Tree | undefined; - set: (filePath: string, tree: Parser.Tree) => void; - clear: () => void; - stats: () => { size: number; maxSize: number }; -} - -export const createASTCache = (maxSize: number = 50): ASTCache => { - const effectiveMax = Math.max(maxSize, 1); - // Initialize the cache with a 'dispose' handler - // This is the magic: When an item is evicted (dropped), this runs automatically. - const cache = new LRUCache({ - max: effectiveMax, - dispose: (tree) => { - try { - // NOTE: web-tree-sitter has tree.delete(); native tree-sitter - // trees are GC-managed and .delete is absent (no-op here). - // - // Single-owner invariant (load-bearing under WASM): a given - // Parser.Tree reference must live in AT MOST ONE ASTCache - // that disposes. The parse-phase chunk-local cache clears - // between chunks; the cross-phase `scopeTreeCache` (also an - // ASTCache today) holds the same Tree by reference. Under - // native tree-sitter this is benign (dispose is a no-op). - // If/when GitNexus adopts web-tree-sitter for sequential - // parsing, the cross-phase cache must either (a) skip - // writing Trees that are already owned by a disposing cache, - // or (b) use tree.copy() per entry. Failing to pick one - // will hand freed memory to scope-resolution. - (tree as unknown as { delete?: () => void }).delete?.(); - } catch (e) { - logger.warn({ e }, 'Failed to delete tree from WASM memory'); - } - }, - }); - - return { - get: (filePath: string) => { - const tree = cache.get(filePath); - return tree; // Returns undefined if not found - }, - - set: (filePath: string, tree: Parser.Tree) => { - cache.set(filePath, tree); - }, - - clear: () => { - cache.clear(); - }, - - stats: () => ({ - size: cache.size, - maxSize: effectiveMax, - }), - }; -}; diff --git a/gitnexus/src/core/ingestion/call-processor.ts b/gitnexus/src/core/ingestion/call-processor.ts index 5c8e447f6..b5b258dcd 100644 --- a/gitnexus/src/core/ingestion/call-processor.ts +++ b/gitnexus/src/core/ingestion/call-processor.ts @@ -9,25 +9,17 @@ * * - `processRoutesFromExtracted` — CALLS edges from framework routes * (e.g. Laravel) to their controller methods. - * - `processNextjsFetchRoutes` / `extractFetchCallsFromFiles` / - * `extractConsumerAccessedKeys` — FETCHES edges from `fetch()` calls to - * Next.js Route nodes. + * - `processNextjsFetchRoutes` / `extractConsumerAccessedKeys` — FETCHES edges + * from `fetch()` calls to Next.js Route nodes. * - `buildExportedTypeMapFromGraph` — exported symbol → return/declared type * map, consumed by the cross-file enrichment pass. */ -import Parser from 'tree-sitter'; import { KnowledgeGraph } from '../graph/types.js'; -import { ASTCache } from './ast-cache.js'; import type { SemanticModel, SymbolTableReader } from './model/index.js'; -import { isLanguageAvailable, loadParser, loadLanguage } from '../tree-sitter/parser-loader.js'; -import { getProvider } from './languages/index.js'; import { generateId } from '../../lib/utils.js'; -import { getLanguageFromFilename } from 'gitnexus-shared'; import type { SymbolDefinition } from 'gitnexus-shared'; import { yieldToEventLoop } from './utils/event-loop.js'; -import { parseSourceSafe } from '../tree-sitter/safe-parse.js'; -import { getTreeSitterBufferSize } from './constants.js'; import type { ExtractedRoute, ExtractedFetchCall } from './workers/parse-worker.js'; import { normalizeFetchURL, routeMatches } from './route-extractors/nextjs.js'; import { extractReturnTypeName } from './type-extractors/shared.js'; @@ -454,79 +446,3 @@ export const processNextjsFetchRoutes = ( } } }; - -/** - * Extract fetch() calls from source files (sequential path). - * Workers handle this via tree-sitter captures in parse-worker; this function - * provides the same extraction for the sequential fallback path. - */ -export const extractFetchCallsFromFiles = async ( - files: { path: string; content: string }[], - astCache: ASTCache, -): Promise => { - const parser = await loadParser(); - const result: ExtractedFetchCall[] = []; - - for (const file of files) { - const language = getLanguageFromFilename(file.path); - if (!language) continue; - if (!isLanguageAvailable(language)) continue; - - const provider = getProvider(language); - const queryStr = provider.treeSitterQueries; - if (!queryStr) continue; - - await loadLanguage(language, file.path); - - let tree = astCache.get(file.path); - if (!tree) { - const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content; - try { - tree = parseSourceSafe(parser, parseContent, undefined, { - bufferSize: getTreeSitterBufferSize(parseContent), - }); - } catch { - continue; - } - astCache.set(file.path, tree); - } - - let matches; - try { - const lang = parser.getLanguage(); - const query = new Parser.Query(lang, queryStr); - matches = query.matches(tree.rootNode); - } catch { - continue; - } - - for (const match of matches) { - const captureMap: Record = {}; - match.captures.forEach((c) => (captureMap[c.name] = c.node)); - - if (captureMap['route.fetch']) { - const urlNode = captureMap['route.url'] ?? captureMap['route.template_url']; - if (urlNode) { - result.push({ - filePath: file.path, - fetchURL: urlNode.text, - lineNumber: captureMap['route.fetch'].startPosition.row, - }); - } - } else if (captureMap['http_client'] && captureMap['http_client.url']) { - const method = captureMap['http_client.method']?.text; - const url = captureMap['http_client.url'].text; - const HTTP_CLIENT_ONLY = new Set(['head', 'options', 'request', 'ajax']); - if (method && HTTP_CLIENT_ONLY.has(method) && url.startsWith('/')) { - result.push({ - filePath: file.path, - fetchURL: url, - lineNumber: captureMap['http_client'].startPosition.row, - }); - } - } - } - } - - return result; -}; diff --git a/gitnexus/src/core/ingestion/export-detection.ts b/gitnexus/src/core/ingestion/export-detection.ts index 31d0722f4..17494e7bb 100644 --- a/gitnexus/src/core/ingestion/export-detection.ts +++ b/gitnexus/src/core/ingestion/export-detection.ts @@ -4,7 +4,8 @@ * Determines whether a symbol (function, class, etc.) is exported/public * in its language. This is a pure function — safe for use in worker threads. * - * Shared between parse-worker.ts (worker pool) and parsing-processor.ts (sequential fallback). + * Used by the language providers during worker parsing (parse-worker.ts) — the + * sole parse path. (Sequential parsing was removed.) */ import { findSiblingChild, type SyntaxNode } from './utils/ast-helpers.js'; diff --git a/gitnexus/src/core/ingestion/filesystem-walker.ts b/gitnexus/src/core/ingestion/filesystem-walker.ts index 9ba959ea6..0af28a958 100644 --- a/gitnexus/src/core/ingestion/filesystem-walker.ts +++ b/gitnexus/src/core/ingestion/filesystem-walker.ts @@ -6,10 +6,6 @@ import { glob } from 'glob'; import { createIgnoreFilter } from '../../config/ignore-service.js'; import { logger } from '../logger.js'; -export interface FileEntry { - path: string; - content: string; -} /** Lightweight entry — path + size from stat, no content in memory */ export interface ScannedFile { @@ -153,21 +149,3 @@ export const readFileContents = async ( return contents; }; - -/** - * Legacy API — scans and reads everything into memory. - * Used by sequential fallback path only. - */ -export const walkRepository = async ( - repoPath: string, - onProgress?: (current: number, total: number, filePath: string) => void, -): Promise => { - const scanned = await walkRepositoryPaths(repoPath, onProgress); - const contents = await readFileContents( - repoPath, - scanned.map((f) => f.path), - ); - return scanned - .filter((f) => contents.has(f.path)) - .map((f) => ({ path: f.path, content: contents.get(f.path)! })); -}; diff --git a/gitnexus/src/core/ingestion/languages/python/captures.ts b/gitnexus/src/core/ingestion/languages/python/captures.ts index 761404ebb..756066382 100644 --- a/gitnexus/src/core/ingestion/languages/python/captures.ts +++ b/gitnexus/src/core/ingestion/languages/python/captures.ts @@ -38,9 +38,10 @@ export function emitPythonScopeCaptures( _filePath: string, cachedTree?: unknown, ): readonly CaptureMatch[] { - // Skip the parse when the caller (parse phase's ASTCache) already - // produced a Tree for this source. Cache miss = re-parse, same as - // before. The cachedTree parameter is typed as `unknown` at the + // Skip the parse when the caller (the scope-resolution orchestrator's + // `treeCache`) already produced a Tree for this source — empty under + // worker-pool runs, so cache miss = re-parse. The cachedTree parameter + // is typed as `unknown` at the // contract layer (see `LanguageProvider.emitScopeCaptures`); cast // here at the use site. let tree = cachedTree as ReturnType['parse']> | undefined; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/orm-extraction.ts b/gitnexus/src/core/ingestion/pipeline-phases/orm-extraction.ts deleted file mode 100644 index c6b46610e..000000000 --- a/gitnexus/src/core/ingestion/pipeline-phases/orm-extraction.ts +++ /dev/null @@ -1,106 +0,0 @@ -/** - * Inline ORM query extraction (sequential fallback path). - * - * Extracts Prisma and Supabase query calls from source content using - * regex patterns. Used by the sequential parse path when workers are - * not available — the worker path extracts ORM queries via tree-sitter - * queries instead. - * - * @module - */ - -import type { ExtractedORMQuery } from '../workers/parse-worker.js'; - -// ── Regex patterns ───────────────────────────────────────────────────────── - -/** Matches Prisma client method calls: `prisma.user.findMany(...)` */ -const PRISMA_QUERY_RE = - /\bprisma\.(\w+)\.(findMany|findFirst|findUnique|findUniqueOrThrow|findFirstOrThrow|create|createMany|update|updateMany|delete|deleteMany|upsert|count|aggregate|groupBy)\s*\(/g; - -/** Matches Supabase client method calls: `supabase.from('users').select(...)` */ -const SUPABASE_QUERY_RE = - /\bsupabase\.from\s*\(\s*['"](\w+)['"]\s*\)\s*\.(select|insert|update|delete|upsert)\s*\(/g; - -// ── Extraction function ─────────────────────────────────────────────────── - -/** - * Extract ORM query calls from file content using regex. - * - * Fast-path: skips files that don't contain `prisma.` or `supabase.from`. - * Results are appended to the `out` array (push pattern avoids allocation). - * - * @param filePath Relative path of the source file - * @param content File content string - * @param out Output array to append extracted queries to - */ -export function extractORMQueriesInline( - filePath: string, - content: string, - out: ExtractedORMQuery[], -): void { - const hasPrisma = content.includes('prisma.'); - const hasSupabase = content.includes('supabase.from'); - if (!hasPrisma && !hasSupabase) return; - - // Pre-compute line number offsets to avoid O(n²) substring+split per match - const lineOffsets = buildLineOffsets(content); - - if (hasPrisma) { - PRISMA_QUERY_RE.lastIndex = 0; - let m; - while ((m = PRISMA_QUERY_RE.exec(content)) !== null) { - const model = m[1]; - if (model.startsWith('$')) continue; - out.push({ - filePath, - orm: 'prisma', - model, - method: m[2], - lineNumber: lineNumberAtOffset(lineOffsets, m.index), - }); - } - } - - if (hasSupabase) { - SUPABASE_QUERY_RE.lastIndex = 0; - let m; - while ((m = SUPABASE_QUERY_RE.exec(content)) !== null) { - out.push({ - filePath, - orm: 'supabase', - model: m[1], - method: m[2], - lineNumber: lineNumberAtOffset(lineOffsets, m.index), - }); - } - } -} - -// ── Line offset helpers ─────────────────────────────────────────────────── - -/** Build an array of byte offsets where each newline occurs (O(n) once). */ -function buildLineOffsets(content: string): number[] { - const offsets: number[] = []; - for (let i = 0; i < content.length; i++) { - if (content[i] === '\n') offsets.push(i); - } - return offsets; -} - -/** - * Binary search for 0-based line number at a given character offset. - * - * Returns the number of newlines that occur before `offset` in the content, - * which is the 0-based line number. When `offset` is beyond the last newline, - * returns `lineOffsets.length` (i.e., the last line index). - */ -function lineNumberAtOffset(lineOffsets: number[], offset: number): number { - let lo = 0; - let hi = lineOffsets.length; - while (lo < hi) { - const mid = (lo + hi) >>> 1; - if (lineOffsets[mid] < offset) lo = mid + 1; - else hi = mid; - } - return lo; -} diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts index 86bb5e587..244f178dd 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts @@ -43,7 +43,6 @@ import { type ExportedTypeMap, } from '../call-processor.js'; import { createSemanticModel, type MutableSemanticModel } from '../model/index.js'; -import { createASTCache } from '../ast-cache.js'; import { type PipelineProgress, getLanguageFromFilename } from 'gitnexus-shared'; import { readFileContents } from '../filesystem-walker.js'; import { isLanguageAvailable } from '../../tree-sitter/parser-loader.js'; @@ -466,14 +465,6 @@ export async function runChunkedParseAndResolve( let filesParsedSoFar = 0; - // Chunk-local tree-sitter cache, cleared between chunks — call / heritage / - // import processors read it during parse to avoid re-parsing within the same - // chunk. (The former cross-phase `scopeTreeCache` was only ever populated by - // the sequential parser, which has been removed; workers can't return native - // Trees across the MessageChannel, so scope-resolution re-parses as needed.) - const maxChunkFiles = chunks.reduce((max, c) => Math.max(max, c.length), 0); - const astCache = createASTCache(maxChunkFiles); - const exportedTypeMap: ExportedTypeMap = new Map(); const bindingAccumulator = new BindingAccumulator(); const allFetchCalls: ExtractedFetchCall[] = []; @@ -649,7 +640,6 @@ export async function runChunkedParseAndResolve( } filesParsedSoFar += chunkFiles.length; - astCache.clear(); if (verboseThroughputLog && chunkStartMs !== null) { const elapsedMs = Date.now() - chunkStartMs; @@ -943,7 +933,6 @@ export async function runChunkedParseAndResolve( // Finalize the accumulator and propagate any fixpoint-inferred exports before // `crossFile` disposes it downstream. Wrapped in try/catch so a cleanup // failure never masks a real parse error; disposal stays with `crossFile`. - astCache.clear(); try { bindingAccumulator.finalize(); const enriched = enrichExportedTypeMap(bindingAccumulator, graph, exportedTypeMap); diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index c5763a66d..9a3837494 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -2434,7 +2434,21 @@ parentPort!.on('message', (msg: WorkerIncomingMessage) => { return; } } catch (err) { - const message = err instanceof Error ? err.message : String(err); - parentPort!.postMessage({ type: 'error', error: message }); + // Carry the worker-side stack across the MessageChannel, not just the + // message. Without this, an unexpected worker throw (e.g. the minified + // `this.# is not a function` family) reaches the operator as a bare + // one-liner with no file:line — exactly what made #2068 undebuggable. The + // pool embeds `errorStack` into its death/circuit-breaker reason so the + // surfaced "Phase 'parse' failed" message points at the real frame (the + // stack's first line already carries the error's type + message). We send + // primitive fields (not the raw Error) so a non-cloneable `cause` payload + // can never turn the report itself into a `messageerror`. `errorStack` is + // optional on the wire, so an older pool ignores it. + const e = err instanceof Error ? err : new Error(String(err)); + parentPort!.postMessage({ + type: 'error', + error: e.message, + errorStack: e.stack, + }); } }); diff --git a/gitnexus/src/core/ingestion/workers/worker-pool.ts b/gitnexus/src/core/ingestion/workers/worker-pool.ts index 93fb6863d..5082453a0 100644 --- a/gitnexus/src/core/ingestion/workers/worker-pool.ts +++ b/gitnexus/src/core/ingestion/workers/worker-pool.ts @@ -310,7 +310,15 @@ type WorkerOutgoingMessage = | { type: 'progress'; filesProcessed: number } | { type: 'warning'; message: string } | { type: 'sub-batch-done' } - | { type: 'error'; error: string } + /** + * Worker-side caught error. `error` is the message; `errorStack` carries the + * worker thread's stack so the pool can embed a real file:line into its + * death / circuit-breaker reason instead of surfacing a bare one-liner (the + * #2068 diagnosability gap). `errorStack` is optional so an older worker + * build that only sends `error` still validates and degrades to message-only + * — and a newer pool reading it just gets no stack. + */ + | { type: 'error'; error: string; errorStack?: string } | { type: 'result'; data: unknown } /** * Authoritative in-flight signal: worker is about to process this file. @@ -656,6 +664,25 @@ function withStderr(worker: Worker, message: string): string { return tail ? `${message}. Worker stderr:\n${tail}` : message; } +/** + * Build a worker-death reason string that carries the worker-side stack when one + * is available (#2068). The stack is appended AFTER the `Worker N error: ` + * prefix so every prefix/substring consumer downstream — recoverAndResume → + * handleWorkerDeath → the circuit-breaker `WorkerPoolDispatchError` message, and + * the tests that regex-match those — keeps working unchanged, while the operator + * now gets the real frame instead of a bare one-liner. The stack's first line is + * normally the message itself; keeping both is harmless and the indented block + * scans cleanly in a log. The stack is capped at WORKER_STDERR_TAIL_LIMIT, + * mirroring the sibling stderr-tail bound, so a pathological error type (or a + * raised `Error.stackTraceLimit`) can't bloat the death reason. `stack` is + * `undefined` for an older worker build (or a thrown non-Error), in which case + * the reason is exactly the prior message-only form. + */ +function workerErrorReason(workerIndex: number, message: string, stack?: string): string { + const base = `Worker ${workerIndex} error: ${message}`; + return stack ? `${base}\n worker stack:\n${stack.slice(0, WORKER_STDERR_TAIL_LIMIT)}` : base; +} + /** * Wait for a freshly-spawned replacement worker to emit the * `{type:'ready'}` handshake from `parse-worker.ts` before treating its @@ -1832,7 +1859,7 @@ export const createWorkerPool = ( settled = true; cleanup(); void recoverAndResume( - `Worker ${workerIndex} error: ${msg.error}`, + workerErrorReason(workerIndex, msg.error, msg.errorStack), resolveExcludePaths(), ); } else if (msg.type === 'result') { @@ -1877,8 +1904,13 @@ export const createWorkerPool = ( if (!settled) { settled = true; cleanup(); + // The Node 'error' event fires on an UNCAUGHT worker throw (one that + // escaped the worker's own try/catch, or an async rejection). Unlike + // the `{type:'error'}` message, the event delivers a real Error whose + // `.stack` is the worker-side frame — carry it so the surfaced reason + // points at the actual failure site, not just `err.message` (#2068). void recoverAndResume( - `Worker ${workerIndex} error: ${err.message}`, + workerErrorReason(workerIndex, err.message, err.stack), resolveExcludePaths(), ); } diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index a41a9d201..51f5ce860 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -7,6 +7,7 @@ import path from 'path'; import os from 'os'; import crypto from 'crypto'; import lbug from '@ladybugdb/core'; +import { closeQueryResults } from './query-result-utils.js'; import { KnowledgeGraph } from '../graph/types.js'; import { NODE_TABLES, @@ -213,8 +214,18 @@ const DB_LOCK_RETRY_DELAY_MS = 500; * analyze` and either already happened or will happen on the next run. */ export const isReadOnlyDbError = (err: unknown): boolean => { - const msg = err instanceof Error ? err.message : String(err); - return /read-only database/i.test(msg); + // Walk the `cause` chain (bounded) so a wrapped read-only error — e.g. the + // pool adapter's `new Error('…read-only.', { cause: nativeReadOnlyErr })` — + // is still detected by callers that only see the wrapper (#2068 follow-up). + // The same strict regex is re-applied at each level, so a non-read-only + // chain stays false; the depth bound guards a cyclic `cause`. + let cur: unknown = err; + for (let depth = 0; depth < 5 && cur != null; depth++) { + const msg = cur instanceof Error ? cur.message : String(cur); + if (/read-only database/i.test(msg)) return true; + cur = cur instanceof Error ? (cur as { cause?: unknown }).cause : undefined; + } + return false; }; const isMissingFileError = (err: unknown): boolean => { @@ -392,12 +403,10 @@ const runWithSessionLock = async (operation: () => Promise): Promise => const normalizeCopyPath = (filePath: string): string => toNativeSafePath(filePath).replace(/\\/g, '/'); +// Single-result convenience wrapper over the shared best-effort closer +// (drainQueryResult / readQueryRows close one cursor at a time). const closeQueryResult = async (result: lbug.QueryResult): Promise => { - try { - await result.close(); - } catch { - // Best-effort cleanup only. - } + await closeQueryResults(result); }; const drainQueryResult = async ( @@ -1758,8 +1767,9 @@ export const queryImporters = async (targetFilePath: string): Promise WHERE r.type = 'IMPORTS' AND b.filePath = '${escaped}' RETURN DISTINCT a.filePath AS importer `; + let queryResult: lbug.QueryResult | lbug.QueryResult[] | undefined; try { - const queryResult = await conn.query(cypher); + queryResult = await conn.query(cypher); const result = Array.isArray(queryResult) ? queryResult[0] : queryResult; const rows = await result.getAll(); const out: string[] = []; @@ -1770,6 +1780,8 @@ export const queryImporters = async (targetFilePath: string): Promise return out; } catch { return []; + } finally { + if (queryResult) await closeQueryResults(queryResult); } }; @@ -1788,8 +1800,9 @@ export const deleteAllCommunitiesAndProcesses = async (): Promise<{ } let nodesDeleted = 0; for (const label of ['Community', 'Process']) { + let countResult: lbug.QueryResult | lbug.QueryResult[] | undefined; try { - const countResult = await conn.query(`MATCH (n:${label}) RETURN count(n) AS cnt`); + countResult = await conn.query(`MATCH (n:${label}) RETURN count(n) AS cnt`); const result = Array.isArray(countResult) ? countResult[0] : countResult; const rows = await result.getAll(); const count = Number(rows[0]?.cnt ?? rows[0]?.[0] ?? 0); @@ -1799,6 +1812,8 @@ export const deleteAllCommunitiesAndProcesses = async (): Promise<{ } } catch { // Table may not exist yet on a freshly-initialized DB — fine. + } finally { + if (countResult) await closeQueryResults(countResult); } } return { nodesDeleted }; diff --git a/gitnexus/src/core/lbug/pool-adapter.ts b/gitnexus/src/core/lbug/pool-adapter.ts index 1b868d8c1..030688d14 100644 --- a/gitnexus/src/core/lbug/pool-adapter.ts +++ b/gitnexus/src/core/lbug/pool-adapter.ts @@ -18,6 +18,7 @@ import fs from 'fs/promises'; import lbug from '@ladybugdb/core'; import { isReadOnlyDbError, loadFTSExtension } from './lbug-adapter.js'; +import { closeQueryResults } from './query-result-utils.js'; import { createLbugDatabase, isWalCorruptionError, @@ -41,8 +42,14 @@ interface PoolEntry { available: lbug.Connection[]; /** Number of connections currently checked out */ checkedOut: number; - /** Queued waiters for when all connections are busy */ - waiters: Array<(conn: lbug.Connection) => void>; + /** Queued waiters for when all connections are busy. Each carries `resolve` + * (hand off a freed connection) and `reject` (fail fast when the pool is + * closed before a connection frees, instead of hanging until the waiter + * timeout — #2068 follow-up). */ + waiters: Array<{ + resolve: (conn: lbug.Connection) => void; + reject: (err: Error) => void; + }>; lastUsed: number; dbPath: string; /** Set to true when the pool entry is closed — checkin will close orphaned connections */ @@ -176,6 +183,20 @@ function closeOne(repoId: string): void { entry.closed = true; + // Reject any callers still queued for a connection: the pool is going away + // (re-init / teardown / LRU eviction), so they must fail fast with an + // actionable error instead of hanging until WAITER_TIMEOUT_MS and then + // surfacing a misleading "pool exhausted" (#2068 follow-up). Draining the + // queue also guarantees checkin() below finds no waiter expecting a + // connection, so a connection returned after close is simply closed. + if (entry.waiters.length > 0) { + const closedErr = new Error( + `LadybugDB connection pool closed for repo "${repoId}" (re-init/teardown); retry the query.`, + ); + for (const waiter of entry.waiters) waiter.reject(closedErr); + entry.waiters.length = 0; + } + // Close available connections — fire-and-forget with .catch() to prevent // unhandled rejections. Native close() returns Promise but can crash // the N-API destructor on macOS/Windows; deferring to process exit lets @@ -680,14 +701,20 @@ function checkout(entry: PoolEntry): Promise { // At capacity — queue the caller with a timeout. return new Promise((resolve, reject) => { - const waiter = (conn: lbug.Connection) => { - clearTimeout(timer); - resolve(conn); + const waiter = { + resolve: (conn: lbug.Connection) => { + clearTimeout(timer); + resolve(conn); + }, + reject: (err: Error) => { + clearTimeout(timer); + reject(err); + }, }; const timer = setTimeout(() => { const idx = entry.waiters.indexOf(waiter); if (idx !== -1) entry.waiters.splice(idx, 1); - reject( + waiter.reject( new Error( `Connection pool exhausted: timed out after ${WAITER_TIMEOUT_MS}ms waiting for a free connection`, ), @@ -713,7 +740,7 @@ function checkin(entry: PoolEntry, conn: lbug.Connection): void { if (entry.waiters.length > 0) { // Hand directly to the next waiter — no intermediate available state const waiter = entry.waiters.shift()!; - waiter(conn); + waiter.resolve(conn); } else { entry.checkedOut--; entry.available.push(conn); @@ -756,22 +783,33 @@ export const executeParameterized = async ( const conn = await checkout(entry); silenceStdout(); activeQueryCount++; + let queryResult: lbug.QueryResult | lbug.QueryResult[] | undefined; try { const stmt = await withTimeout(conn.prepare(cypher), QUERY_TIMEOUT_MS, 'Prepare'); if (!stmt.isSuccess()) { const errMsg = await stmt.getErrorMessage(); throw new Error(`Prepare failed: ${errMsg}`); } - const queryResult = await withTimeout(conn.execute(stmt, params), QUERY_TIMEOUT_MS, 'Execute'); + queryResult = await withTimeout(conn.execute(stmt, params), QUERY_TIMEOUT_MS, 'Execute'); const result = Array.isArray(queryResult) ? queryResult[0] : queryResult; const rows = await result.getAll(); return rows; } catch (err) { if (isReadOnlyDbError(err)) { - throw new Error('Write operations are not allowed. The pool adapter is read-only.'); + // Preserve the native error as `cause` so the original frame/message is + // not lost behind the friendly read-only message (#2068 follow-up). + throw new Error('Write operations are not allowed. The pool adapter is read-only.', { + cause: err, + }); } throw err; } finally { + // Close the native QueryResult cursor(s) before returning the connection — + // getAll() drains rows but does not release the native cursor, so without + // this the cursor leaks for the connection's lifetime (#2068 follow-up). + // Best-effort via the shared helper; never masks the query result or a real + // error. + if (queryResult) await closeQueryResults(queryResult); activeQueryCount--; restoreStdout(); checkin(entry, conn); diff --git a/gitnexus/src/core/lbug/query-result-utils.ts b/gitnexus/src/core/lbug/query-result-utils.ts new file mode 100644 index 000000000..938e28a2b --- /dev/null +++ b/gitnexus/src/core/lbug/query-result-utils.ts @@ -0,0 +1,31 @@ +import lbug from '@ladybugdb/core'; + +/** + * Best-effort close of one or more native `QueryResult` cursors. + * + * `result.getAll()` materializes rows into a JS array but does not release the + * native cursor — leaving it open holds native resources for the connection's + * lifetime. Both the pooled adapter (`pool-adapter.ts`) and the direct adapter + * (`lbug-adapter.ts`) must release cursors after reading; this is the single + * shared implementation so neither re-rolls the array-normalize + swallow loop + * (a #2068 follow-up de-dup). Lives in its own leaf module so the two adapters + * depend on it rather than on each other. + * + * `conn.execute()` can return either a single `QueryResult` or an array + * (multi-statement); this normalizes both. Each close is independent and + * best-effort: a failing or absent `close()` on one cursor never throws and + * never prevents the others from closing — a cleanup failure must not mask the + * query result or a real error at the call site. + */ +export async function closeQueryResults( + queryResult: lbug.QueryResult | lbug.QueryResult[], +): Promise { + const results = Array.isArray(queryResult) ? queryResult : [queryResult]; + for (const r of results) { + try { + await r?.close(); + } catch { + // Best-effort cleanup only. + } + } +} diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts index 46ecf4030..77411b107 100644 --- a/gitnexus/src/mcp/local/local-backend.ts +++ b/gitnexus/src/mcp/local/local-backend.ts @@ -177,8 +177,19 @@ function logQueryError(context: string, err: unknown): void { logger.error({ context, err: msg }, 'GitNexus query failed'); } -const isReadOnlyDbError = (err: unknown): boolean => - /read-only database/i.test(err instanceof Error ? err.message : String(err)); +const isReadOnlyDbError = (err: unknown): boolean => { + // Walk the `cause` chain (bounded) so a wrapped read-only error (e.g. the + // pool adapter's `{ cause }` wrapper) is still detected here — this is the + // copy the MCP cypher handler uses to surface its curated read-only message + // (#2068 follow-up). Mirrors lbug-adapter's isReadOnlyDbError. + let cur: unknown = err; + for (let depth = 0; depth < 5 && cur != null; depth++) { + const msg = cur instanceof Error ? cur.message : String(cur); + if (/read-only database/i.test(msg)) return true; + cur = cur instanceof Error ? (cur as { cause?: unknown }).cause : undefined; + } + return false; +}; /** * Per-query latency telemetry for production aggregation (#553). diff --git a/gitnexus/src/types/pipeline.ts b/gitnexus/src/types/pipeline.ts index becf81757..5ad08cae6 100644 --- a/gitnexus/src/types/pipeline.ts +++ b/gitnexus/src/types/pipeline.ts @@ -19,9 +19,12 @@ export interface PipelineResult { */ resolutionOutcomes: readonly ResolutionOutcome[]; /** - * True if the parse phase spawned a worker pool for this run. False means - * the sequential fallback handled every chunk. Primarily a test affordance - * so regression suites can prove which path executed. + * True if a worker pool was actually constructed for this run. The worker + * pool is the sole parse path (sequential parsing was removed). False means + * no pool was needed: either there were no parseable files, or every chunk + * was a parse-cache hit and the cached worker output was replayed without + * spawning workers (a warm all-cache-hit run, #2038). Primarily a test + * affordance so regression suites can prove the pool engaged. */ usedWorkerPool: boolean; } diff --git a/gitnexus/test/integration/lbug-pool.test.ts b/gitnexus/test/integration/lbug-pool.test.ts index 259c15c13..561772a75 100644 --- a/gitnexus/test/integration/lbug-pool.test.ts +++ b/gitnexus/test/integration/lbug-pool.test.ts @@ -94,6 +94,86 @@ withTestLbugDB( }); }); + // ─── closeLbug rejects pending waiters (#2068 follow-up) ───────────── + // + // Before the fix, closeOne() never rejected queued waiters: a caller + // waiting for a free connection when the pool was closed (e.g. a staleness + // reinit under concurrent query load) hung for WAITER_TIMEOUT_MS (15s) and + // then surfaced a misleading "pool exhausted" error. Now they reject + // immediately with an actionable "pool closed" message. The pool caps at + // MAX_CONNS_PER_REPO (8); firing a synchronous burst larger than that queues + // the surplus as waiters, and closing synchronously (before any query + // settles) must reject every queued waiter at once. The default 5s test + // timeout also guards promptness — a regression would block ~15s and time + // out rather than reject. + describe('closeLbug waiter handling (#2068)', () => { + it('rejects queued waiters promptly with a pool-closed error on close', async () => { + await initLbug('test-repo', handle.dbPath); + + // Fire a burst larger than the 8-connection cap WITHOUT awaiting: the + // first 8 check out connections synchronously, the surplus queue as + // waiters — all before the synchronous closeLbug below runs. + const BURST = 24; + const MAX_CONNS = 8; + const inflight = Array.from({ length: BURST }, () => + executeQuery('test-repo', 'MATCH (n:Function) RETURN n.name AS name'), + ); + // Close in the same synchronous tick — no microtask has served a waiter. + const closing = closeLbug('test-repo'); + + const settled = await Promise.allSettled(inflight); + await closing; + + const reasons = settled + .filter((r): r is PromiseRejectedResult => r.status === 'rejected') + .map((r) => String(r.reason?.message ?? r.reason)); + + // The surplus (BURST - MAX_CONNS) waiters must reject with "pool closed". + const poolClosed = reasons.filter((m) => /pool closed/i.test(m)); + expect(poolClosed.length).toBeGreaterThanOrEqual(BURST - MAX_CONNS); + // And none should have hit the 15s "exhausted" waiter-timeout path. + expect(reasons.some((m) => /waiting for a free connection/i.test(m))).toBe(false); + + expect(isLbugReady('test-repo')).toBe(false); + }); + + it('settles in-flight queries and fully tears down when closed mid-flight', async () => { + // closeOne-vs-checkin interleave (F4b): with 8 connections in-flight and + // surplus callers queued, a synchronous close must (a) let every promise + // settle — no hang — and (b) fully delete the pool entry so checked-in + // connections are closed as orphans rather than handed to a rejected + // waiter. We assert the observable contract; the "orphan not handed to a + // rejected waiter" invariant is single-threaded-by-construction (closeOne + // drains waiters with no await before any checkin can run). + await initLbug('test-repo', handle.dbPath); + + const inflight = Array.from({ length: 16 }, () => + executeQuery('test-repo', 'MATCH (n:Function) RETURN n.name AS name'), + ); + const closing = closeLbug('test-repo'); + + // allSettled only resolves once EVERY query settled — proving none hangs + // (a 15s waiter-timeout regression would blow the default test timeout). + const settled = await Promise.allSettled(inflight); + await closing; + expect(settled).toHaveLength(16); + expect( + settled.some( + (r) => + r.status === 'rejected' && + /waiting for a free connection/i.test(String(r.reason?.message ?? r.reason)), + ), + ).toBe(false); + + // Pool entry fully gone — a subsequent query fails fast with the + // not-initialized error, not a hang or a stale connection. + expect(isLbugReady('test-repo')).toBe(false); + await expect(executeQuery('test-repo', 'MATCH (n) RETURN n LIMIT 1')).rejects.toThrow( + /not initialized/i, + ); + }); + }); + // ─── Parameterized queries ─────────────────────────────────────────── describe('executeParameterized', () => { diff --git a/gitnexus/test/unit/ast-cache.test.ts b/gitnexus/test/unit/ast-cache.test.ts deleted file mode 100644 index 66cc6e3c3..000000000 --- a/gitnexus/test/unit/ast-cache.test.ts +++ /dev/null @@ -1,95 +0,0 @@ -import { describe, it, expect, beforeEach } from 'vitest'; -import { createASTCache, type ASTCache } from '../../src/core/ingestion/ast-cache.js'; - -// Create a minimal mock tree object (mimics Parser.Tree interface) -function mockTree(id: string): any { - return { rootNode: { type: 'program', text: id }, delete: vi.fn() }; -} - -describe('ASTCache', () => { - let cache: ASTCache; - - beforeEach(() => { - cache = createASTCache(3); - }); - - describe('get / set', () => { - it('returns undefined for cache miss', () => { - expect(cache.get('nonexistent.ts')).toBeUndefined(); - }); - - it('returns cached tree on hit', () => { - const tree = mockTree('test'); - cache.set('src/index.ts', tree); - expect(cache.get('src/index.ts')).toBe(tree); - }); - - it('overwrites existing entry for same key', () => { - const tree1 = mockTree('v1'); - const tree2 = mockTree('v2'); - cache.set('src/index.ts', tree1); - cache.set('src/index.ts', tree2); - expect(cache.get('src/index.ts')).toBe(tree2); - }); - }); - - describe('LRU eviction', () => { - it('evicts least recently used when capacity exceeded', () => { - cache.set('a.ts', mockTree('a')); - cache.set('b.ts', mockTree('b')); - cache.set('c.ts', mockTree('c')); - // Cache is full (maxSize=3). Adding one more evicts 'a' - cache.set('d.ts', mockTree('d')); - expect(cache.get('a.ts')).toBeUndefined(); - expect(cache.get('b.ts')).toBeDefined(); - expect(cache.get('d.ts')).toBeDefined(); - }); - - it('accessing an entry makes it recently used', () => { - cache.set('a.ts', mockTree('a')); - cache.set('b.ts', mockTree('b')); - cache.set('c.ts', mockTree('c')); - // Touch 'a' to make it recently used - cache.get('a.ts'); - // Now 'b' is LRU - cache.set('d.ts', mockTree('d')); - expect(cache.get('a.ts')).toBeDefined(); - expect(cache.get('b.ts')).toBeUndefined(); - }); - }); - - describe('clear', () => { - it('removes all entries', () => { - cache.set('a.ts', mockTree('a')); - cache.set('b.ts', mockTree('b')); - cache.clear(); - expect(cache.get('a.ts')).toBeUndefined(); - expect(cache.get('b.ts')).toBeUndefined(); - expect(cache.stats().size).toBe(0); - }); - }); - - describe('stats', () => { - it('reports size and maxSize', () => { - expect(cache.stats()).toEqual({ size: 0, maxSize: 3 }); - cache.set('a.ts', mockTree('a')); - expect(cache.stats()).toEqual({ size: 1, maxSize: 3 }); - cache.set('b.ts', mockTree('b')); - expect(cache.stats()).toEqual({ size: 2, maxSize: 3 }); - }); - - it('uses default maxSize of 50', () => { - const defaultCache = createASTCache(); - expect(defaultCache.stats().maxSize).toBe(50); - }); - - it('clamps maxSize of 0 to 1 to prevent LRU cache error', () => { - const zeroCache = createASTCache(0); - expect(zeroCache.stats().maxSize).toBe(1); - // Should still function correctly - const tree = mockTree('test'); - zeroCache.set('a.ts', tree); - expect(zeroCache.get('a.ts')).toBe(tree); - }); - }); -}); diff --git a/gitnexus/test/unit/lbug-query-result-utils.test.ts b/gitnexus/test/unit/lbug-query-result-utils.test.ts new file mode 100644 index 000000000..9a7ac3ad8 --- /dev/null +++ b/gitnexus/test/unit/lbug-query-result-utils.test.ts @@ -0,0 +1,45 @@ +/** + * Unit tests: closeQueryResults — the shared best-effort cursor-close helper + * (#2068 follow-up). Guards the contract the two LadybugDB adapters rely on: + * a single result OR an array of results are all closed, and a failing/absent + * `close()` never throws and never skips the rest. + */ +import { describe, it, expect, vi } from 'vitest'; +import { closeQueryResults } from '../../src/core/lbug/query-result-utils.js'; + +// Minimal QueryResult stand-in — only `close()` matters here. +function fakeResult(close: () => unknown = () => undefined) { + return { close: vi.fn(close) } as unknown as import('@ladybugdb/core').QueryResult; +} + +describe('closeQueryResults', () => { + it('closes a single QueryResult exactly once', async () => { + const r = fakeResult(); + await closeQueryResults(r); + expect(r.close as ReturnType).toHaveBeenCalledTimes(1); + }); + + it('closes EVERY element of an array (not just the first)', async () => { + const rs = [fakeResult(), fakeResult(), fakeResult()]; + await closeQueryResults(rs); + for (const r of rs) { + expect(r.close as ReturnType).toHaveBeenCalledTimes(1); + } + }); + + it('keeps closing the rest when one close() rejects (best-effort, no throw)', async () => { + const ok1 = fakeResult(); + const bad = fakeResult(() => { + throw new Error('native close failed'); + }); + const ok2 = fakeResult(() => Promise.reject(new Error('async close failed'))); + const ok3 = fakeResult(); + await expect(closeQueryResults([ok1, bad, ok2, ok3])).resolves.toBeUndefined(); + expect(ok1.close as ReturnType).toHaveBeenCalledTimes(1); + expect(ok3.close as ReturnType).toHaveBeenCalledTimes(1); + }); + + it('is a no-op on an empty array', async () => { + await expect(closeQueryResults([])).resolves.toBeUndefined(); + }); +}); diff --git a/gitnexus/test/unit/lbug-readonly-error.test.ts b/gitnexus/test/unit/lbug-readonly-error.test.ts index 4e412c323..514ddc07d 100644 --- a/gitnexus/test/unit/lbug-readonly-error.test.ts +++ b/gitnexus/test/unit/lbug-readonly-error.test.ts @@ -41,6 +41,34 @@ describe('isReadOnlyDbError', () => { expect(isReadOnlyDbError(undefined)).toBe(false); }); + it('detects a read-only error wrapped as a `cause` (#2068 — pool-adapter wrapper)', () => { + // The pool adapter rethrows native read-only failures as a friendly + // message with the original error preserved on `cause`. Detection must see + // through the wrapper so the MCP/HTTP handlers surface their curated message. + const wrapped = new Error('Write operations are not allowed. The pool adapter is read-only.', { + cause: new Error('Cannot execute write operations in a read-only database!'), + }); + expect(isReadOnlyDbError(wrapped)).toBe(true); + }); + + it('does NOT match a wrapper whose cause chain is unrelated', () => { + const wrapped = new Error('Query failed', { cause: new Error('Connection refused') }); + expect(isReadOnlyDbError(wrapped)).toBe(false); + }); + + it('terminates on a cyclic cause chain without hanging', () => { + const a = new Error('boom a') as Error & { cause?: unknown }; + const b = new Error('boom b') as Error & { cause?: unknown }; + a.cause = b; + b.cause = a; + expect(isReadOnlyDbError(a)).toBe(false); + }); + + it('stops cleanly on a non-Error cause', () => { + const wrapped = new Error('outer', { cause: 'just a string' }); + expect(isReadOnlyDbError(wrapped)).toBe(false); + }); + it('does NOT match unrelated errors that the ensure path must still surface', () => { // Lock contention — handled separately by isDbBusyError; must not be // silenced by the read-only filter. diff --git a/gitnexus/test/unit/worker-pool-error-stack.test.ts b/gitnexus/test/unit/worker-pool-error-stack.test.ts new file mode 100644 index 000000000..fbd828142 --- /dev/null +++ b/gitnexus/test/unit/worker-pool-error-stack.test.ts @@ -0,0 +1,176 @@ +import { describe, expect, it, beforeEach, afterEach } from 'vitest'; +import { EventEmitter } from 'node:events'; +import path from 'node:path'; +import { pathToFileURL } from 'node:url'; +import fs from 'node:fs'; +import os from 'node:os'; + +import { + createWorkerPool, + WorkerPoolDispatchError, +} from '../../src/core/ingestion/workers/worker-pool.js'; + +/** + * #2068 regression: a worker-side crash must carry its stack across the + * MessageChannel so the surfaced "Phase 'parse' failed" error points at a real + * frame instead of a bare one-liner (the issue's `this.#q is not a function` + * reached the operator with no file:line because the worker only sent + * `err.message`). These tests assert the worker stack rides through both worker + * failure channels — the `{type:'error'}` message (a caught worker throw) and + * the Node `'error'` event (an uncaught worker throw) — into the + * `WorkerPoolDispatchError` the parse phase rejects with, and that an older + * worker build that omits the stack still degrades cleanly. + */ + +type NodeWorker = import('node:worker_threads').Worker; + +type FakeAction = + | { kind: 'error-message'; error: string; errorStack?: string } + | { kind: 'error-event'; message: string; stack: string }; + +const nextActions: FakeAction[] = []; + +/** + * Minimal worker double: emits the readiness handshake on construction, then + * runs one scripted action per dispatched sub-batch. Unlike the resilience + * suite's double, this one can emit the `{type:'error', errorStack}` MESSAGE + * (the worker's own caught-error path) in addition to the Node `'error'` event. + */ +class FakeWorker extends EventEmitter { + constructor() { + super(); + queueMicrotask(() => { + this.emit('online'); + this.emit('message', { type: 'ready' }); + }); + } + + postMessage(rawMsg: unknown): void { + const m = rawMsg as { type?: string }; + // Only a real dispatch drives an action; ignore the pool's `flush` reply. + if (!m || m.type !== 'sub-batch') return; + const action = nextActions.shift(); + if (!action) return; + queueMicrotask(() => { + if (action.kind === 'error-message') { + this.emit('message', { + type: 'error', + error: action.error, + errorStack: action.errorStack, + }); + } else { + const e = new Error(action.message); + e.stack = action.stack; + this.emit('error', e); + } + }); + } + + async terminate(): Promise { + this.emit('exit', 0); + return 0; + } +} + +let tempDir: string; +let workerUrl: URL; + +beforeEach(() => { + nextActions.length = 0; + tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-worker-error-stack-')); + const workerPath = path.join(tempDir, 'fake-worker.js'); + fs.writeFileSync(workerPath, '// fake'); + workerUrl = pathToFileURL(workerPath) as URL; +}); + +afterEach(() => { + try { + fs.rmSync(tempDir, { recursive: true, force: true }); + } catch { + // best-effort cleanup + } +}); + +// Trip the breaker on the very first death so the reason string is surfaced +// verbatim in the rejected WorkerPoolDispatchError. +const TRIP_ON_FIRST_DEATH = { + consecutiveFailureThreshold: 1, + maxRespawnsPerSlot: 0, +} as const; + +async function dispatchAndCatch(pool: ReturnType): Promise { + try { + await pool.dispatch<{ path: string; content: string }, unknown>([ + { path: 'src/a.ts', content: '' }, + ]); + return undefined; + } catch (e) { + return e; + } +} + +describe('worker-pool error stack propagation (#2068)', () => { + it('embeds the worker stack from a {type:error} message into the surfaced error', async () => { + const workerStack = + 'TypeError: this.#q is not a function\n' + + ' at frobnicate (/dist/core/ingestion/workers/parse-worker.js:1234:56)'; + const pool = createWorkerPool(workerUrl, 1, { + workerFactory: () => new FakeWorker() as unknown as NodeWorker, + ...TRIP_ON_FIRST_DEATH, + }); + nextActions.push({ + kind: 'error-message', + error: 'this.#q is not a function', + errorStack: workerStack, + }); + + const caught = await dispatchAndCatch(pool); + + expect(caught).toBeInstanceOf(WorkerPoolDispatchError); + const msg = (caught as Error).message; + expect(msg).toContain('this.#q is not a function'); + expect(msg).toContain('worker stack:'); + expect(msg).toContain('frobnicate (/dist/core/ingestion/workers/parse-worker.js:1234:56)'); + await pool.terminate(); + }); + + // The Node 'error' event fires on an UNCAUGHT JS throw / async rejection (which + // carries a real JS stack). A true NATIVE abort (tree-sitter SIGSEGV / OOM kill) + // instead fires the 'exit' event and is intentionally stackless — no JS frame + // exists — so it is NOT exercised here. + it('embeds the worker stack from a Node error event (uncaught throw) into the surfaced error', async () => { + const workerStack = + 'Error: uncaught worker throw\n' + + ' at processFileGroup (/dist/core/ingestion/workers/parse-worker.js:777:9)'; + const pool = createWorkerPool(workerUrl, 1, { + workerFactory: () => new FakeWorker() as unknown as NodeWorker, + ...TRIP_ON_FIRST_DEATH, + }); + nextActions.push({ kind: 'error-event', message: 'uncaught worker throw', stack: workerStack }); + + const caught = await dispatchAndCatch(pool); + + expect(caught).toBeInstanceOf(WorkerPoolDispatchError); + const msg = (caught as Error).message; + expect(msg).toContain('worker stack:'); + expect(msg).toContain('processFileGroup (/dist/core/ingestion/workers/parse-worker.js:777:9)'); + await pool.terminate(); + }); + + it('degrades to message-only when an older worker build omits errorStack', async () => { + const pool = createWorkerPool(workerUrl, 1, { + workerFactory: () => new FakeWorker() as unknown as NodeWorker, + ...TRIP_ON_FIRST_DEATH, + }); + // No errorStack — the wire field is optional for back/forward compat. + nextActions.push({ kind: 'error-message', error: 'legacy worker failure' }); + + const caught = await dispatchAndCatch(pool); + + expect(caught).toBeInstanceOf(WorkerPoolDispatchError); + const msg = (caught as Error).message; + expect(msg).toContain('legacy worker failure'); + expect(msg).not.toContain('worker stack:'); + await pool.terminate(); + }); +}); From df5ce1f49b6c6e1609c0bff1a5d2d2ce3c1d18ad Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 8 Jun 2026 08:09:43 +0100 Subject: [PATCH 10/17] fix(ingestion): close remaining open language parsing-layer coverage gaps (#1919) (#2072) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(c): skip computed #include MACRO instead of emitting a garbage import source (F5) * fix(cpp): emit a Variable per name for structured-binding declarations (F9) * fix(dart): extract static const/final class fields (F26) * fix(dart): capture old-style function typedefs (F28) * fix(dart): read real top-level variable shape instead of a dead type field (F29) * fix(kotlin): capture callable references (F47) * fix(kotlin): anchor infix-call capture to the operator only (F49) * fix(kotlin): extract secondary constructors as members (F48) * fix(kotlin): capture destructuring declarations (F51) * fix(kotlin): index companion-object properties as fields (F52) * test(kotlin): assert callable-reference coverage runs on the worker path (F47) * fix(swift): extract protocol property requirements (F75) * fix(swift): recognize enum_class_body as a method body node (F79) * test(ingestion): rebaseline swift captures-golden + scope-capture fingerprints (#1919) * fix(kotlin): attribute secondary-constructor body calls to the Constructor node (#1919 review CF1) A Kotlin secondary constructor's body executes statements like a method body, but the registry-primary scope-resolution path had no Function scope or Constructor def for it. A call inside the body resolved its caller anchor up to the enclosing Class scope, mis-attributing the CALLS edge to the class rather than the Constructor. Add `(secondary_constructor) @scope.function` to the Kotlin scope query so the body becomes its own scope, and synthesize a `@declaration.constructor` (named `constructor`, qualified `.constructor`, with parameter metadata) so the scope owns a Constructor def that bridges to the structure-phase Constructor node. Also add an arity-disambiguating lookup key for overloadable callables: two same-name secondary constructors of different arity (e.g. a zero-arg vs a 2-arg) share the qualified key whose first-write-wins assignment is source-order- dependent — so a zero-arg overload could resolve to a sibling. The structure node id encodes `#`; mirror that in the bridge keyspace and match by the def's parameterCount. Same-arity overloads collapse onto one arity key exactly as before, so no regression there. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(kotlin): do not own function-local property bindings under the enclosing class (#1919 review CF3) Kotlin emits destructuring / loop bindings (`val (a,b) = pair`, `for ((k,v) in m)`) as `@definition.property` to dodge the block-scope local-symbol pruner. When such a binding sits inside a method body of a class, the structure-phase owner walk found the enclosing class and emitted a spurious HAS_PROPERTY edge (e.g. `C -> k`), treating a function-local as a class member. Guard the Property owner resolution: if a function-like ancestor is reached before any class container, the property is function-local and gets no owner edge (it falls back to a File DEFINES edge). Language-agnostic — genuine class fields sit directly in the class body with no intervening function, so they keep their HAS_PROPERTY owner edge. Co-Authored-By: Claude Opus 4.8 (1M context) * test(kotlin): guard non-companion property isStatic=false (#1919 review CF4) Add a field-extraction case for a plain non-companion class `class C { val x: Int = 1 }` asserting the property `x` has isStatic=false, guarding the `isInsideKotlinCompanion` walk against false-positives. Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(kotlin): dedup type_identifier lookup in extractOwnerName (#1919 review CF5) The `node.namedChildren.find(c => c.type === 'type_identifier')?.text` lookup was duplicated across the companion and non-companion branches of the Kotlin field-extractor's extractOwnerName. Hoist it into a single local, preserving the existing behavior (anonymous companion falls back to "Companion"; other nodes prefer the `name` field, else the type_identifier text, else undefined). Co-Authored-By: Claude Opus 4.8 (1M context) * fix(dart): capture generic old-style function typedefs (#1919 review CF2) * test(dart): guard multi-name field count and top-level-var labels (#1919 review CF4) * docs(swift): correct isStatic comment re multi-modifier hasKeyword (#1919 review CF5) * test(ingestion): rebaseline dart+kotlin scope-capture fingerprints after review remediation (#1919) * fix(ingestion): correct CF3 owner-strip boundary set for accessor/init bodies and Dart signatures (#1919 review) The CF3 property-ownership guard used FUNCTION_NODE_TYPES, which (a) includes Dart bare signatures (function_signature/method_signature) — over-stripping every Dart class getter/setter's HAS_PROPERTY owner — and (b) omits Kotlin anonymous_initializer/getter/setter and Swift computed accessors — under- stripping destructuring/locals inside init{} and accessor bodies, emitting spurious Class->local HAS_PROPERTY edges. Introduces a guard-specific LOCAL_SCOPE_BODY_NODE_TYPES set (signatures excluded, accessor/init bodies included). Adds Dart accessor-ownership + Kotlin init/accessor destructuring regression fixtures. Both confirmed on the worker pipeline; no cross-language regression (1597 cross-language tests green). --------- Co-authored-by: Claude Opus 4.8 (1M context) --- gitnexus/bench/scope-capture/baselines.json | 35 ++- .../field-extractors/configs/dart.ts | 105 ++++--- .../ingestion/field-extractors/configs/jvm.ts | 47 ++- .../field-extractors/configs/swift.ts | 37 ++- .../languages/c/import-decomposer.ts | 18 +- .../languages/cpp/import-decomposer.ts | 15 +- .../core/ingestion/languages/dart/query.ts | 40 +++ .../ingestion/languages/kotlin/captures.ts | 129 ++++++++ .../core/ingestion/languages/kotlin/query.ts | 30 ++ .../method-extractors/configs/jvm.ts | 33 +- .../method-extractors/configs/swift.ts | 15 +- .../scope-resolution/graph-bridge/ids.ts | 10 + .../graph-bridge/node-lookup.ts | 14 + .../src/core/ingestion/tree-sitter-queries.ts | 131 +++++++- .../src/core/ingestion/utils/ast-helpers.ts | 70 +++++ .../variable-extractors/configs/c-cpp.ts | 50 +++ .../variable-extractors/configs/dart.ts | 123 +++++--- .../variable-extractors/configs/jvm.ts | 45 ++- .../core/ingestion/workers/parse-worker.ts | 57 +++- .../lang-resolution/c-coverage/main.c | 9 + .../cpp-structured-binding/App.cpp | 5 + .../dart-accessor-owner/main.dart | 12 + .../dart-coverage/typedefs.dart | 5 + .../dart-static-fields/config.dart | 6 + .../dart-toplevel-vars/globals.dart | 7 + .../kotlin-companion-fields/Companions.kt | 15 + .../kotlin-coverage/callable_refs.kt | 15 + .../kotlin-destructuring/Destructuring.kt | 8 + .../kotlin-local-property-owner/Locals.kt | 31 ++ .../kotlin-secondary-ctor/Constructors.kt | 14 + .../swift-enum-members/Direction.swift | 22 ++ .../swift-protocol-property/Repository.swift | 9 + .../expected-captures.json | 8 + .../integration/resolvers/c-coverage.test.ts | 87 ++++++ .../test/integration/resolvers/cpp.test.ts | 15 + .../resolvers/dart-coverage.test.ts | 118 +++++++ .../test/integration/resolvers/dart.test.ts | 135 ++++++++ .../resolvers/kotlin-coverage.test.ts | 176 +++++++++++ .../test/integration/resolvers/kotlin.test.ts | 228 ++++++++++++++ .../test/integration/resolvers/swift.test.ts | 107 +++++++ gitnexus/test/unit/field-extraction.test.ts | 294 ++++++++++++++++++ gitnexus/test/unit/method-extraction.test.ts | 131 ++++++++ .../test/unit/variable-extraction.test.ts | 202 ++++++++++++ 43 files changed, 2514 insertions(+), 149 deletions(-) create mode 100644 gitnexus/test/fixtures/lang-resolution/c-coverage/main.c create mode 100644 gitnexus/test/fixtures/lang-resolution/dart-accessor-owner/main.dart create mode 100644 gitnexus/test/fixtures/lang-resolution/dart-coverage/typedefs.dart create mode 100644 gitnexus/test/fixtures/lang-resolution/dart-static-fields/config.dart create mode 100644 gitnexus/test/fixtures/lang-resolution/dart-toplevel-vars/globals.dart create mode 100644 gitnexus/test/fixtures/lang-resolution/kotlin-companion-fields/Companions.kt create mode 100644 gitnexus/test/fixtures/lang-resolution/kotlin-coverage/callable_refs.kt create mode 100644 gitnexus/test/fixtures/lang-resolution/kotlin-destructuring/Destructuring.kt create mode 100644 gitnexus/test/fixtures/lang-resolution/kotlin-local-property-owner/Locals.kt create mode 100644 gitnexus/test/fixtures/lang-resolution/kotlin-secondary-ctor/Constructors.kt create mode 100644 gitnexus/test/fixtures/lang-resolution/swift-enum-members/Direction.swift create mode 100644 gitnexus/test/fixtures/lang-resolution/swift-protocol-property/Repository.swift create mode 100644 gitnexus/test/integration/resolvers/c-coverage.test.ts create mode 100644 gitnexus/test/integration/resolvers/dart-coverage.test.ts create mode 100644 gitnexus/test/integration/resolvers/kotlin-coverage.test.ts diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index 3c3e4e0c2..79c504677 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -11,17 +11,18 @@ "_note": "Updated for F17-F23 fixes (P2: TIMES guard, ADD GIVING, SQL AS alias). See PR #1959." }, "c": { - "fingerprint": "39f3a8346bb58159e9d79e2db6e1dd34ec6d70028e507ffd39548667dc658aa2", + "fingerprint": "12a196b2d6249c8d86a931b12ecebc2a0cdf8d6f47683acdd0d8e9d8bc7657f5", "scaling_budget": 1.5, - "_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance \u2014 flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96.", - "_note": "#1983: + c-static-linkage-worker fixture (caller.c/lib.c/lib.h/local.c \u2014 worker-path static-linkage side-channel test). Pure fixture-corpus drift: no c/captures.ts or query change branch-vs-main, existing fixtures' captures byte-identical (c-captures.test.ts 45/45), scaling stays linear (~0.97). The baseline was missed when the fixture landed; regenerated here. fingerprint 0de009b->39f3a83." + "_added": "#1956: c added to the scope-capture bench (was UNBENCHED). C has no inheritance — flat scale source. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in c/captures.ts (threaded c.node, byte-identical over c-* fixtures); scaling 3.475 -> 0.96.", + "_note": "#1983: + c-static-linkage-worker fixture (caller.c/lib.c/lib.h/local.c — worker-path static-linkage side-channel test). Pure fixture-corpus drift: no c/captures.ts or query change branch-vs-main, existing fixtures' captures byte-identical (c-captures.test.ts 45/45), scaling stays linear (~0.97). The baseline was missed when the fixture landed; regenerated here. fingerprint 0de009b->39f3a83.", + "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression)." }, "cpp": { - "fingerprint": "6d6207ae1df3943c5fae28983e0c294e55225456e7cf39af1d46fda21b6787c4", + "fingerprint": "fd3d3768cdebbb4767d7cf18b8d2df19d61de969c816d7f4d6b599f947811356", "scaling_budget": 1.5, "_added": "#1956: cpp added to the scope-capture bench (was UNBENCHED). Heritage-bearing scale source (: public Base, public Mixin) drives emitCppInheritanceCaptures at scale. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in cpp/captures.ts (~12 sites, threaded c.node, byte-identical over 263 cpp-* fixtures); scaling 2.30 -> 1.12.", - "_rebaselined": "#1965 / #1923 F4: uninitialized non-leading multi-declarators now emit @declaration.variable captures; cpp-adl-inner-callable-outer-noncallable data::Pair a, b adds the legitimate fixture drift. Linear (~1.06).", - "_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift \u2014 no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures \u2014 pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture \u2014 pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae." + "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression).", + "_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift — no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures — pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture — pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae." }, "csharp": { "_rebaselined": "#1956 synth-widening: + csharp-qualified-base fixture; the synth now walks record_declaration + struct_declaration base_lists and handles alias_qualified_name (matching the #1940 legacy leg), so record/struct heritage now emits. csharp-record-base gains a record inherits capture. (record->record SAME-namespace EXTENDS is a separate registry resolution gap, tracked as follow-up.) Linear (~1.00). (Earlier #1956: heritage-bearing scale source.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged. | #1924 F16: record primary-constructor base bindings now exclude constructor arguments; capture fingerprint changes, scaling remains linear. | #2036 review follow-up: csharp-record-base now exercises primary-constructor base dispatch end to end; +2 capture groups, scaling remains linear.", @@ -32,8 +33,8 @@ "rust": { "fingerprint": "ac610bbe97666bf285923479dd7b43a2fe4c5354aae8df1bcbafdc04fb220f82", "scaling_budget": 1.5, - "_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) \u2014 legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", - "_note": "PR #1934: F66/F68 let-binding pattern narrowing; F71 union (Struct-labeled, now materialized via legacy @definition.struct + resolvable); F72 macro FULLY WIRED \u2014 @declaration.macro/@reference.macro + MacroRegistry \u2192 USES edges to Macro nodes (never a same-named fn). + rust-macro / rust-union fixtures and merged with origin/main #1975 rust-scoped-impl; fingerprint re-baselined (scaling ~0.99, fixture_count 126). #1992: + rust-nested-tail-collision-generic and rust-generic-impl-same-method-name (F3) fixtures \u2014 pure fixture-corpus drift, no scope-extractor change; fixture_count 127->129, fingerprint 56ffc1c0->b00aea0f." + "_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) — legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", + "_note": "PR #1934: F66/F68 let-binding pattern narrowing; F71 union (Struct-labeled, now materialized via legacy @definition.struct + resolvable); F72 macro FULLY WIRED — @declaration.macro/@reference.macro + MacroRegistry → USES edges to Macro nodes (never a same-named fn). + rust-macro / rust-union fixtures and merged with origin/main #1975 rust-scoped-impl; fingerprint re-baselined (scaling ~0.99, fixture_count 126). #1992: + rust-nested-tail-collision-generic and rust-generic-impl-same-method-name (F3) fixtures — pure fixture-corpus drift, no scope-extractor change; fixture_count 127->129, fingerprint 56ffc1c0->b00aea0f." }, "php": { "fingerprint": "bc2c27c5ba26d5aea61142a2a99fb772222f5b969205260eb7a71b4c0bd73cdb", @@ -45,18 +46,18 @@ "fingerprint": "b5ea93bb3d0469c3821a8c70f5d5991c6f326e41097c119ad691154301dcc753", "scaling_budget": 1.5, "_rebaselined": "#1956 synth-widening: + ruby-qualified-base fixture; synth now reduces a scope_resolution superclass (class C < Mod::Super) to its trailing constant (matching the #1940 legacy leg), at parity. Linear (~1.03). (Earlier #1956: heritage-bearing scale source.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", - "_note": "F62: + scope_resolution class/module declaration captures \u2014 fixture count 78\u219281, fingerprint drift expected. #1975: + ruby-tail-collision fixture (Foo::Bar vs Baz::Bar stay distinct nodes) \u2014 pure fixture-corpus drift, scope-extractor captures unchanged; 81\u219282. #1991: + ruby-nested-mixin-tail-collision fixture (85\u219286). Recomputed on the #942 merge (fixture-comment rewording shifts capture byte-positions, capture LOGIC unchanged): bf6b13a -> b5ea93bb." + "_note": "F62: + scope_resolution class/module declaration captures — fixture count 78→81, fingerprint drift expected. #1975: + ruby-tail-collision fixture (Foo::Bar vs Baz::Bar stay distinct nodes) — pure fixture-corpus drift, scope-extractor captures unchanged; 81→82. #1991: + ruby-nested-mixin-tail-collision fixture (85→86). Recomputed on the #942 merge (fixture-comment rewording shifts capture byte-positions, capture LOGIC unchanged): bf6b13a -> b5ea93bb." }, "swift": { - "fingerprint": "53325c6345161c5a495f997297af5a24fb718fd3e6647040160f8ab2a2c8e4c0", + "fingerprint": "180ac68e780bdf6f9089d53f51cbb9a66aed3e7774631cc3fcbaae5020213998", "scaling_budget": 1.5, - "_rebaselined": "#1956: swift-qualified-base fixture + heritage-bearing scale source (class: Base, Serviceable \u2014 extends + protocol conformance); linear (~1.03)." + "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression)." }, "dart": { - "fingerprint": "a9e882b537765e8fd0ddfcd33b38b253dd86fc5ddffa6e4bf5a85ed8ee615eaa", + "fingerprint": "94bf2c26e1ba96f4211634aa572c0a989b503e717e75dfc5df04f66c417de80f", "scaling_budget": 1.5, "_added": "#939: dart added to the scope-capture bench with the registry-primary migration. Heritage-bearing scale source (Entity extends Base implements Marker) gates the @reference.inherits synth + the postfix-chain reference walk at scale. emitDartScopeCaptures threads tree-sitter captured nodes (no findNodeAtRange root-walk), so it is linear (~1.0).", - "_rebaselined": "#1970 review + tri-review follow-ups: constructor-call retag, cascade calls, built-in suppression, enum scope, #1926 F24/F25, named-ctor dedup (crash fix), container-name binding suppression; heritage file-affinity resolution. Fixtures: member-call-contexts, constructor-body, named-constructor-body, heritage-name-collision, construct-cascade." + "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0." }, "java": { "fingerprint": "9b29cafe32873b4902bda311bd089ffc04efe08f13557b966d29544be514080a", @@ -67,8 +68,8 @@ "typescript": { "fingerprint": "3f44a4a6892698df2d145c8ff2812c3b318807648983c88aca28fbd694f172f9", "scaling_budget": 1.5, - "_rebaselined": "#1962: F44 (class scope@), F85 (enum member declarations), F87 (optional_parameter type annotations) add new captures \u2014 fingerprint drift expected.", - "_note": "#1968: F44, F85, F87 \u2014 fingerprint drift expected." + "_rebaselined": "#1962: F44 (class scope@), F85 (enum member declarations), F87 (optional_parameter type annotations) add new captures — fingerprint drift expected.", + "_note": "#1968: F44, F85, F87 — fingerprint drift expected." }, "javascript": { "fingerprint": "d72f03c6c502235d2d4b74d66baa5c7d361f040d7a1b72e84acad61210d05ae8", @@ -77,9 +78,9 @@ "_rebaselined": "#1956 synth-widening: + javascript-qualified-base fixture; synthesizeJsInheritanceReferences now handles a member_expression base (class S extends ns.Base -> Base), matching the #1940 legacy leg + the TS terminalTsTypeNameNode property_identifier case, at parity. Linear (~1.05). | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged." }, "kotlin": { - "fingerprint": "a16400622892183581b8f5f8fa01f07842d19b8cf49ed021df52bd17009d749f", + "fingerprint": "90aa832978d9744e50058e77a04748390a7e34e36b309f6c1d178eb07280b7ea", "scaling_budget": 1.5, "_added": "#1951: bench coverage added (was ungated); scale source heritage-bearing (: Base()); js/kotlin O(n^2) findNodeAtRange-per-match fixed to threaded captured node, now linear.", - "_rebaselined": "#1956 synth-widening: + kotlin-qualified-base fixture; synthesizeKotlinInheritanceReferences now handles the explicit_delegation form (class F : Iface by d -> Iface), matching the #1940 legacy leg, at parity. Linear (~0.87). | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged. | #1930 F45: default parameters now emit optional-arity metadata; capture fingerprint changes, scaling remains linear." + "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0." } } diff --git a/gitnexus/src/core/ingestion/field-extractors/configs/dart.ts b/gitnexus/src/core/ingestion/field-extractors/configs/dart.ts index 52f4c0ca7..00c89b607 100644 --- a/gitnexus/src/core/ingestion/field-extractors/configs/dart.ts +++ b/gitnexus/src/core/ingestion/field-extractors/configs/dart.ts @@ -2,15 +2,59 @@ import { SupportedLanguages } from 'gitnexus-shared'; import type { FieldExtractionConfig } from '../generic.js'; +import type { FieldVisibility } from '../../field-types.js'; +import type { SyntaxNode } from '../../utils/ast-helpers.js'; import { hasKeyword } from './helpers.js'; import { extractSimpleTypeName } from '../../type-extractors/shared.js'; /** * Dart field extraction config. * - * Dart class fields appear as declaration nodes inside class_body. + * Dart class fields appear as `declaration` nodes inside `class_body`. + * Two shapes carry the field name(s): + * - instance / plain fields → `initialized_identifier_list` + * (`int z = 0;`, `int a = 1, b = 2;`) + * - `static const` / `static final` / `const` fields → `static_final_declaration_list` + * (`static const a = 1;`, `static final String b = 'x', c = 'y';`) + * Both shapes may declare SEVERAL fields in one declaration, so name extraction + * is multi-name (`extractNames`). The structure query (`DART_QUERIES`) emits one + * `@definition.property` per name for both shapes; this config enriches each. + * * Visibility is convention-based: underscore prefix = private. */ + +/** All field names declared by a `declaration` node, across both Dart shapes. */ +function extractDartFieldNames(node: SyntaxNode): string[] { + const names: string[] = []; + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (!child) continue; + + // instance / plain fields: initialized_identifier_list > initialized_identifier > identifier + if (child.type === 'initialized_identifier_list') { + for (let j = 0; j < child.namedChildCount; j++) { + const init = child.namedChild(j); + if (init?.type === 'initialized_identifier') { + const ident = init.firstNamedChild; + if (ident?.type === 'identifier') names.push(ident.text); + } + } + } + + // static const / final fields: static_final_declaration_list > static_final_declaration > identifier + if (child.type === 'static_final_declaration_list') { + for (let j = 0; j < child.namedChildCount; j++) { + const decl = child.namedChild(j); + if (decl?.type === 'static_final_declaration') { + const ident = decl.firstNamedChild; + if (ident?.type === 'identifier') names.push(ident.text); + } + } + } + } + return names; +} + export const dartConfig: FieldExtractionConfig = { language: SupportedLanguages.Dart, typeDeclarationNodes: ['class_definition'], @@ -18,31 +62,20 @@ export const dartConfig: FieldExtractionConfig = { bodyNodeTypes: ['class_body'], defaultVisibility: 'public', + // One AST `declaration` node may declare several fields (`int a, b;`, + // `static final String b = 'x', c = 'y';`), so use the multi-name path. extractName(node) { - // declaration > initialized_identifier_list > initialized_identifier > identifier - for (let i = 0; i < node.namedChildCount; i++) { - const child = node.namedChild(i); - if (child?.type === 'initialized_identifier_list') { - for (let j = 0; j < child.namedChildCount; j++) { - const init = child.namedChild(j); - if (init?.type === 'initialized_identifier') { - const ident = init.firstNamedChild; - if (ident?.type === 'identifier') return ident.text; - } - } - } - if (child?.type === 'initialized_identifier') { - const ident = child.firstNamedChild; - if (ident?.type === 'identifier') return ident.text; - } - } - // fallback: look for direct identifier - const name = node.childForFieldName('name'); - return name?.text; + return extractDartFieldNames(node)[0]; + }, + + extractNames(node) { + return extractDartFieldNames(node); }, extractType(node) { - // declaration > type_identifier (first named child usually) + // declaration > type_identifier (the type annotation, present for both the + // instance-field shape and `static final String b = …`). `static const a = 1;` + // has no annotation → undefined (untyped). for (let i = 0; i < node.namedChildCount; i++) { const child = node.namedChild(i); if (child && (child.type === 'type_identifier' || child.type === 'function_type')) { @@ -52,22 +85,16 @@ export const dartConfig: FieldExtractionConfig = { return undefined; }, - extractVisibility(node) { - // Dart uses _ prefix for private - // Walk to find the identifier name - for (let i = 0; i < node.namedChildCount; i++) { - const child = node.namedChild(i); - if (child?.type === 'initialized_identifier_list') { - for (let j = 0; j < child.namedChildCount; j++) { - const init = child.namedChild(j); - if (init?.type === 'initialized_identifier') { - const ident = init.firstNamedChild; - if (ident?.text?.startsWith('_')) return 'private'; - } - } - } - } - return 'public'; + // Per-name: Dart convention is underscore-prefixed = private. A single + // declaration can mix visibilities (`static const _p = 1, q = 2;`), so the + // decision is keyed on the individual field name. + extractVisibilityForName(_node, name): FieldVisibility { + return name.startsWith('_') ? 'private' : 'public'; + }, + + extractVisibility(node): FieldVisibility { + const first = extractDartFieldNames(node)[0]; + return first?.startsWith('_') ? 'private' : 'public'; }, isStatic(node) { @@ -75,6 +102,8 @@ export const dartConfig: FieldExtractionConfig = { }, isReadonly(node) { + // `final` / `const` (both `final_builtin`/`const_builtin` nodes whose text + // is `final`/`const`) are read-only. return hasKeyword(node, 'final') || hasKeyword(node, 'const'); }, }; diff --git a/gitnexus/src/core/ingestion/field-extractors/configs/jvm.ts b/gitnexus/src/core/ingestion/field-extractors/configs/jvm.ts index 9d6cdbf07..37015a998 100644 --- a/gitnexus/src/core/ingestion/field-extractors/configs/jvm.ts +++ b/gitnexus/src/core/ingestion/field-extractors/configs/jvm.ts @@ -5,6 +5,7 @@ import type { FieldExtractionConfig } from '../generic.js'; import { findVisibility, hasKeyword, hasModifier, typeFromField } from './helpers.js'; import { extractSimpleTypeName } from '../../type-extractors/shared.js'; import type { FieldVisibility } from '../../field-types.js'; +import type { SyntaxNode } from '../../utils/ast-helpers.js'; // --------------------------------------------------------------------------- // Java @@ -73,13 +74,49 @@ export const javaConfig: FieldExtractionConfig = { const KOTLIN_VIS = new Set(['public', 'private', 'protected', 'internal']); +/** A property_declaration is a companion-object member when its nearest + * class-body ancestor is the body of a companion_object (F52, issue #1919). + * Companion members are addressed statically through the enclosing class + * (`C.TAG`), so they are marked static. */ +function isInsideKotlinCompanion(node: SyntaxNode): boolean { + for (let cur = node.parent; cur !== null; cur = cur.parent) { + if (cur.type === 'class_body') return cur.parent?.type === 'companion_object'; + if (cur.type === 'companion_object') return true; + } + return false; +} + export const kotlinConfig: FieldExtractionConfig = { language: SupportedLanguages.Kotlin, - typeDeclarationNodes: ['class_declaration', 'object_declaration'], + // F52: include companion_object so a companion property's innermost + // class-container owner (findEnclosingClassNode returns the companion_object) + // is recognized as a type declaration and its nested class_body is walked. + // The structure query already creates the Property node and owns it on the + // ENCLOSING class for anonymous companions / on the named companion Class — + // this entry only drives field-metadata enrichment, so it does NOT change + // ownership or emit a second node (no double-count). + typeDeclarationNodes: ['class_declaration', 'object_declaration', 'companion_object'], fieldNodeTypes: ['property_declaration'], bodyNodeTypes: ['class_body'], defaultVisibility: 'public', + // F52: an anonymous `companion object { ... }` has no name child, so the + // generic factory's `childForFieldName('name')` owner lookup is empty and + // `extract()` would bail before walking the body. Supply a stable owner + // name (the named companion's identifier, else "Companion") so the body IS + // walked; the resulting FieldInfo map is keyed by field NAME only, so the + // owner name does not affect which Property node gets enriched. + extractOwnerName(node) { + const typeIdentifierText = node.namedChildren.find((c) => c.type === 'type_identifier')?.text; + if (node.type === 'companion_object') { + // Anonymous companions have no type_identifier — fall back to "Companion". + return typeIdentifierText ?? 'Companion'; + } + const name = node.childForFieldName('name'); + if (name) return name.text; + return typeIdentifierText; + }, + extractName(node) { // property_declaration > variable_declaration > simple_identifier for (let i = 0; i < node.namedChildCount; i++) { @@ -124,9 +161,11 @@ export const kotlinConfig: FieldExtractionConfig = { return findVisibility(node, KOTLIN_VIS, 'public', 'modifiers'); }, - isStatic(_node) { - // Kotlin doesn't have static; companion object members are handled separately - return false; + isStatic(node) { + // Kotlin has no `static`, but companion-object members are accessed + // statically through the enclosing class (`C.TAG`) — mark them static + // so the field metadata reflects that (F52). + return isInsideKotlinCompanion(node); }, isReadonly(node) { diff --git a/gitnexus/src/core/ingestion/field-extractors/configs/swift.ts b/gitnexus/src/core/ingestion/field-extractors/configs/swift.ts index 75c70ab95..6e27ff702 100644 --- a/gitnexus/src/core/ingestion/field-extractors/configs/swift.ts +++ b/gitnexus/src/core/ingestion/field-extractors/configs/swift.ts @@ -2,7 +2,7 @@ import { SupportedLanguages } from 'gitnexus-shared'; import type { FieldExtractionConfig } from '../generic.js'; -import { hasKeyword, findVisibility } from './helpers.js'; +import { hasKeyword, hasModifier, findVisibility } from './helpers.js'; import { extractSimpleTypeName } from '../../type-extractors/shared.js'; import type { FieldVisibility } from '../../field-types.js'; @@ -17,18 +17,33 @@ const SWIFT_VIS = new Set([ /** * Swift field extraction config. * - * Handles property_declaration inside class_body / protocol_body. + * Handles property_declaration inside class_body / protocol_body and + * protocol_property_declaration inside protocol_body (F75 — protocol property + * requirements like "var title: String { get }"). + * * tree-sitter-swift uses property_declaration for stored/computed properties. + * A protocol property requirement parses to its own node type, + * protocol_property_declaration, whose name lives in a "name:" pattern field + * (pattern > value_binding_pattern + simple_identifier(bound_identifier)), its + * type in a sibling type_annotation, and its "{ get }" / "{ get set }" in a + * protocol_property_requirements child. Note: Swift reuses the "name:" field + * across many positions (func name, every parameter label, parameter/return + * type), so the name is synthesized from the simple_identifier inside the + * pattern rather than read blindly off "name:". */ export const swiftConfig: FieldExtractionConfig = { language: SupportedLanguages.Swift, typeDeclarationNodes: ['class_declaration', 'protocol_declaration'], - fieldNodeTypes: ['property_declaration'], + fieldNodeTypes: ['property_declaration', 'protocol_property_declaration'], bodyNodeTypes: ['class_body', 'protocol_body'], defaultVisibility: 'internal', extractName(node) { - // property_declaration > pattern > simple_identifier + // property_declaration > pattern > simple_identifier, and + // protocol_property_declaration > name: (pattern ... simple_identifier). + // For protocol_property_declaration the pattern wraps a leading + // value_binding_pattern ("var") plus the simple_identifier — the loop + // below skips the binding keyword and returns the identifier. for (let i = 0; i < node.namedChildCount; i++) { const child = node.namedChild(i); if (child?.type === 'pattern') { @@ -62,7 +77,19 @@ export const swiftConfig: FieldExtractionConfig = { }, isStatic(node) { - return hasKeyword(node, 'static') || hasKeyword(node, 'class'); + // `static`/`class` (type-level) modifiers live inside a `modifiers` + // wrapper for both property_declaration and protocol_property_declaration + // (e.g. `static var shared: P { get }`), so check the wrapper too. + // `hasKeyword` compares each direct child by `.text` equality: it matches a + // single-modifier wrapper (`modifiers.text === 'static'`) but fails for a + // multi-modifier wrapper (`private static` → `modifiers.text === 'private static'`), + // which `hasModifier` handles by descending into the wrapper's children. + return ( + hasKeyword(node, 'static') || + hasKeyword(node, 'class') || + hasModifier(node, 'modifiers', 'static') || + hasModifier(node, 'modifiers', 'class') + ); }, isReadonly(node) { diff --git a/gitnexus/src/core/ingestion/languages/c/import-decomposer.ts b/gitnexus/src/core/ingestion/languages/c/import-decomposer.ts index 5cef27430..77b54f120 100644 --- a/gitnexus/src/core/ingestion/languages/c/import-decomposer.ts +++ b/gitnexus/src/core/ingestion/languages/c/import-decomposer.ts @@ -5,10 +5,20 @@ import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/as * Decompose a `preproc_include` node into a CaptureMatch with structured * import captures. C #include maps to a wildcard import (all symbols * from the header are visible). + * + * Only literal include paths are emitted as import sources: + * #include → system_lib_string + * #include "local.h" → string_literal + * A computed include like `#include HEADER_MACRO` carries an `identifier` + * path node (the macro name, not a header path). Emitting it as an import + * source produces a garbage literal edge, so we skip it entirely — matching + * the convention in interpretCImport, which drops imports with no resolvable + * source (issue #1919 F5). */ export function splitCInclude(node: SyntaxNode): CaptureMatch | null { // node.type === 'preproc_include' // path field: (string_literal (string_content)) | (system_lib_string) + // | (identifier) ← computed macro include, NOT a header path const pathNode = node.childForFieldName?.('path') ?? null; if (pathNode === null) { // Fallback: scan children @@ -24,7 +34,13 @@ export function splitCInclude(node: SyntaxNode): CaptureMatch | null { return buildIncludeCapture(node, pathNode); } -function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch { +function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch | null { + // Skip computed includes (`#include MACRO`) — the path is an `identifier`, + // not a literal header path. Emitting it would create a garbage import. + if (pathNode.type !== 'string_literal' && pathNode.type !== 'system_lib_string') { + return null; + } + let raw: string; if (pathNode.type === 'string_literal') { // string_literal has children: `"`, string_content, `"` diff --git a/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts b/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts index eb6b252ce..ae6843923 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts @@ -5,6 +5,13 @@ import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/as * Decompose a `preproc_include` node into a CaptureMatch with structured * import captures. C++ #include maps to a wildcard import (all symbols * from the header are visible). Identical to C's splitCInclude. + * + * Only literal include paths are emitted as import sources: + * #include → system_lib_string + * #include "User.h" → string_literal + * A computed include like `#include HEADER_MACRO` carries an `identifier` + * path node (the macro name, not a header path); we skip it so it never + * becomes a garbage literal import source (issue #1919 F5). */ export function splitCppInclude(node: SyntaxNode): CaptureMatch | null { const pathNode = node.childForFieldName?.('path') ?? null; @@ -21,7 +28,13 @@ export function splitCppInclude(node: SyntaxNode): CaptureMatch | null { return buildIncludeCapture(node, pathNode); } -function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch { +function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch | null { + // Skip computed includes (`#include MACRO`) — the path is an `identifier`, + // not a literal header path. Emitting it would create a garbage import. + if (pathNode.type !== 'string_literal' && pathNode.type !== 'system_lib_string') { + return null; + } + let raw: string; if (pathNode.type === 'string_literal') { const content = pathNode.namedChildren.find((c) => c.type === 'string_content'); diff --git a/gitnexus/src/core/ingestion/languages/dart/query.ts b/gitnexus/src/core/ingestion/languages/dart/query.ts index c00cfbeb5..7d66b1c4a 100644 --- a/gitnexus/src/core/ingestion/languages/dart/query.ts +++ b/gitnexus/src/core/ingestion/languages/dart/query.ts @@ -39,6 +39,46 @@ const DART_SCOPE_QUERY = ` (extension_declaration name: (identifier) @declaration.name) @declaration.class (enum_declaration name: (identifier) @declaration.name) @declaration.enum +; ── Declarations — type aliases (old-style + new-style function typedefs) ──── +; Both forms parse as type_alias; the name position differs, and a generic +; parameter list intervenes for the generic variants. Per #1919 review CF2, +; a generic type_parameters node sits between the name and the next anchor, so +; the non-generic adjacency patterns silently drop the generic forms. Four +; standalone patterns (NOT one alternation — the tree-sitter 0.21 hazard drops +; sibling branches) keep the name capture unambiguous and single-match per form: +; non-generic old-style typedef int Cmp(int a, int b); +; children: return-type, NAME, formal_parameter_list +; generic old-style typedef int Cmp(T a, T b); (CF2) +; children: return-type, NAME, type_parameters, formal_parameter_list +; non-generic new-style typedef Pred = bool Function(int); +; children: NAME, "=", function_type +; generic new-style typedef Mapper = T Function(T); +; children: NAME, type_parameters, "=", function_type +; The alias name is the type_identifier immediately before the param list (old) +; or before "=" (new); for the generic forms it is the one immediately before +; the intervening type_parameters. Mirrors Kotlin's @declaration.type_alias +; rule; the generic scope-extractor maps "type_alias" → TypeAlias. +(type_alias + (type_identifier) @declaration.name + . + (formal_parameter_list)) @declaration.type_alias +(type_alias + (type_identifier) @declaration.name + . + (type_parameters) + . + (formal_parameter_list)) @declaration.type_alias +(type_alias + (type_identifier) @declaration.name + . + "=") @declaration.type_alias +(type_alias + (type_identifier) @declaration.name + . + (type_parameters) + . + "=") @declaration.type_alias + ; ── Declarations — top-level functions (parent is program, not method) ─────── (program (function_signature diff --git a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts index 2b14266a6..16d2a3db8 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts @@ -39,6 +39,7 @@ export function emitKotlinScopeCaptures( out.push(...synthesizeKotlinSmartCastBindings(tree.rootNode)); out.push(...synthesizeKotlinLambdaBindings(tree.rootNode, returnTypes)); out.push(...synthesizeKotlinInheritanceReferences(tree.rootNode)); + out.push(...synthesizeKotlinSecondaryConstructorDeclarations(tree.rootNode)); for (const match of getKotlinScopeQuery().matches(tree.rootNode)) { const grouped: Record = {}; @@ -87,6 +88,40 @@ export function emitKotlinScopeCaptures( } } + // Callable references (`::method`, `Type::new`, `obj::m`) — F47 (#1919). + // The query captures the referenced member as `@reference.name`, an + // optional receiver type as `@reference.receiver`, and the whole node as + // `@reference.callable`. Rewrite into a call reference so it participates + // in call-graph resolution: a bare `::member` resolves as a free call; + // a `Receiver::member` resolves as a member call against the receiver + // type. The function/constructor is referenced (not invoked), so no + // arity/argument metadata is attached. + if (grouped['@reference.callable'] !== undefined) { + const nameCap = grouped['@reference.name']; + const callableNode = groupedNodes['@reference.callable']; + if (nameCap !== undefined && callableNode !== undefined) { + const receiverCap = grouped['@reference.receiver']; + // The anchor Capture must carry the call-form tag as its `name` — + // the scope-extractor reads `Capture.name` (not the map key) to + // classify the reference kind, so re-wrap via nodeToCapture rather + // than reusing the `@reference.callable`-named Capture (whose head + // `callable` resolves to no ReferenceKind and silently drops it). + if (receiverCap !== undefined) { + out.push({ + '@reference.call.member': nodeToCapture('@reference.call.member', callableNode), + '@reference.name': nameCap, + '@reference.receiver': receiverCap, + }); + } else { + out.push({ + '@reference.call.free': nodeToCapture('@reference.call.free', callableNode), + '@reference.name': nameCap, + }); + } + } + continue; + } + if ( grouped['@reference.call.free'] !== undefined && grouped['@reference.receiver'] !== undefined @@ -253,6 +288,100 @@ function synthesizeKotlinInheritanceReferences(rootNode: SyntaxNode): CaptureMat return out; } +/** + * The enclosing type name for a node nested in a class/object/companion body. + * Walks up to the first `class_declaration` / `object_declaration` / + * `companion_object` ancestor and returns its `type_identifier` name node. + * Used to qualify a secondary-constructor declaration as `.constructor`. + */ +function kotlinEnclosingTypeNameNode(node: SyntaxNode): SyntaxNode | null { + for (let cur: SyntaxNode | null = node.parent; cur !== null; cur = cur.parent) { + if ( + cur.type === 'class_declaration' || + cur.type === 'object_declaration' || + cur.type === 'companion_object' + ) { + const nameNode = cur.namedChildren.find((c) => c.type === 'type_identifier'); + return nameNode ?? null; + } + } + return null; +} + +/** + * Synthesize a `@declaration.constructor` capture for each Kotlin + * `secondary_constructor` (issue #1919 review CF1). The structure phase already + * materializes a `Constructor` graph node (`Constructor:file:Class.constructor#`), + * but the registry-primary scope-resolution path had no Constructor *def* in the + * scope tree — so a call inside the constructor body resolved its caller anchor + * up to the enclosing Class def, mis-attributing the CALLS edge to the class. + * + * Paired with `(secondary_constructor) @scope.function` in query.ts: that rule + * makes the constructor body its own Function scope; this declaration places a + * Constructor def in that scope so `pickCallerCallableDef` anchors calls on the + * Constructor. The def is keyed to match the structure-phase node id: + * - `@declaration.qualified_name` = `.constructor` so the bridge's + * qualified key (`:file::Constructor::Class.constructor`) hits the node. + * - `@declaration.parameter-types` so two same-name secondary constructors are + * disambiguated by the bridge's parameter-types key (`~Int,Int`), matching + * the `#`-suffixed structure node for the overload with the same + * parameter shape. (The zero-arg overload carries no parameter types and + * resolves via the qualified/simple key to the `#0` node.) + * + * The anchor spans the whole `secondary_constructor` node — same range as the + * `@scope.function` it pairs with — so the def is owned by that Function scope + * and the constructor name auto-hoists to the enclosing class scope (exactly the + * binding shape a normal method declaration produces). + */ +function synthesizeKotlinSecondaryConstructorDeclarations(rootNode: SyntaxNode): CaptureMatch[] { + const out: CaptureMatch[] = []; + for (const ctorNode of descendantsOfType(rootNode, 'secondary_constructor')) { + const keyword = ctorNode.namedChildren.find((c) => c.type === 'constructor'); + // The `constructor` keyword is an anonymous token; fall back to the node + // itself for the name capture position when the named-child lookup misses. + const nameAnchor = keyword ?? ctorNode; + const classNameNode = kotlinEnclosingTypeNameNode(ctorNode); + const qualifiedName = + classNameNode !== null ? `${classNameNode.text}.constructor` : 'constructor'; + + const match: Record = { + '@declaration.constructor': nodeToCapture('@declaration.constructor', ctorNode), + '@declaration.name': syntheticCapture('@declaration.name', nameAnchor, 'constructor'), + '@declaration.qualified_name': syntheticCapture( + '@declaration.qualified_name', + ctorNode, + qualifiedName, + ), + }; + + const arity = computeKotlinArityMetadata(ctorNode); + if (arity.parameterCount !== undefined) { + match['@declaration.parameter-count'] = syntheticCapture( + '@declaration.parameter-count', + ctorNode, + String(arity.parameterCount), + ); + } + if (arity.requiredParameterCount !== undefined) { + match['@declaration.required-parameter-count'] = syntheticCapture( + '@declaration.required-parameter-count', + ctorNode, + String(arity.requiredParameterCount), + ); + } + if (arity.parameterTypes !== undefined) { + match['@declaration.parameter-types'] = syntheticCapture( + '@declaration.parameter-types', + ctorNode, + JSON.stringify(arity.parameterTypes), + ); + } + + out.push(match); + } + return out; +} + /** * The bare simple-name `type_identifier` of a `user_type`. Strips generic * type arguments (`Base` → `Base`) and qualifier tails (`pkg.Base` → `Base`) diff --git a/gitnexus/src/core/ingestion/languages/kotlin/query.ts b/gitnexus/src/core/ingestion/languages/kotlin/query.ts index 9209655d8..31613a38e 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/query.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/query.ts @@ -9,6 +9,16 @@ const KOTLIN_SCOPE_QUERY = ` (companion_object) @scope.class (function_declaration) @scope.function +;; Secondary-constructor body scope (issue #1919 review CF1). A +;; secondary constructor's "constructor(...) { ... }" body executes statements +;; just like a method body, so it must be its OWN Function scope — otherwise a +;; call inside the body resolves its caller anchor up to the enclosing Class +;; scope (the class's Class def), mis-attributing the CALLS edge to the class +;; rather than the Constructor. The matching @declaration.constructor is +;; synthesized in captures.ts (synthesizeKotlinSecondaryConstructorDeclarations) +;; so this scope owns a Constructor def keyed to the structure-phase node id. +(secondary_constructor) @scope.function + ;; Companion-object marker (issue #1756 / U4). Side-channel capture that ;; lets populateCompanionMembersOnEnclosingClass distinguish a companion ;; Class scope from a regular Class scope without inspecting ownedDefs. @@ -117,6 +127,26 @@ const KOTLIN_SCOPE_QUERY = ` (function_value_parameters) [(user_type) (nullable_type) (function_type)] @type-binding.type) @type-binding.return +;; References — callable references ("::method", "Type::new", "obj::m") — F47. +;; A "callable_reference" references a function/constructor as a value (no +;; call_suffix), so the registry-primary call path never saw it. Real-parse +;; (issue #1919) shows the canonical shape inside a function body is: +;; "::topLevelFn" -> (callable_reference :: (simple_identifier)) member only +;; "String::length" -> (callable_reference (type_identifier) :: (simple_identifier)) +;; "obj::method" -> (callable_reference (type_identifier) :: (simple_identifier)) +;; "Type::new" -> (callable_reference (type_identifier) :: (simple_identifier)) +;; The receiver (real type OR object) is always a "type_identifier"; the +;; referenced member is the LAST "simple_identifier". One rule with an +;; optional receiver and an end-anchored member covers all four forms with +;; exactly one match per callable_reference (no sibling-branch double-match). +;; (NOTE: a qualified "A.B::m" parses as a nested navigation_expression, not a +;; callable_reference, and is already captured by the read.member rule below.) +;; emitKotlinScopeCaptures rewrites this into a free/member call reference. +(callable_reference + (type_identifier)? @reference.receiver + (simple_identifier) @reference.name + .) @reference.callable + ;; References — direct calls / constructor syntax (call_expression (simple_identifier) @reference.name) @reference.call.free diff --git a/gitnexus/src/core/ingestion/method-extractors/configs/jvm.ts b/gitnexus/src/core/ingestion/method-extractors/configs/jvm.ts index 3233dc938..47e7773cb 100644 --- a/gitnexus/src/core/ingestion/method-extractors/configs/jvm.ts +++ b/gitnexus/src/core/ingestion/method-extractors/configs/jvm.ts @@ -186,9 +186,30 @@ function kotlinParameterHasDefaultValue(param: SyntaxNode): boolean { return false; } +/** + * Member name for a Kotlin method node. A `secondary_constructor` + * (`constructor(...) { }`) has no name child — its only identity token is + * the anonymous `constructor` keyword — so it is named "constructor" (F48, + * issue #1919), matching the @name the KOTLIN_QUERIES structure rule captures + * off that keyword so method-extractor enrichment keys (`name:line`) align. + * Multiple secondary constructors collide on this name but are disambiguated + * downstream by the `#` ID suffix the worker appends to Constructors. + */ +function extractKotlinMethodName(node: SyntaxNode): string | undefined { + if (node.type === 'secondary_constructor') return 'constructor'; + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child?.type === 'simple_identifier') return child.text; + } + return undefined; +} + function extractKotlinParameters(node: SyntaxNode): ParameterInfo[] { const params: ParameterInfo[] = []; - // Kotlin: function_declaration > function_value_parameters > parameter + // Kotlin: function_declaration / secondary_constructor > + // function_value_parameters > parameter. Both node types nest the + // parameter list the same way, so the same walk extracts a secondary + // constructor's parameters (F48). for (let i = 0; i < node.namedChildCount; i++) { const child = node.namedChild(i); if (child && child.type === 'function_value_parameters') { @@ -271,16 +292,10 @@ function extractKotlinReturnType(node: SyntaxNode): string | undefined { export const kotlinMethodConfig: MethodExtractionConfig = { language: SupportedLanguages.Kotlin, typeDeclarationNodes: ['class_declaration', 'object_declaration', 'companion_object'], - methodNodeTypes: ['function_declaration'], + methodNodeTypes: ['function_declaration', 'secondary_constructor'], bodyNodeTypes: ['class_body'], staticOwnerTypes: new Set(['companion_object', 'object_declaration']), - extractName(node) { - for (let i = 0; i < node.namedChildCount; i++) { - const child = node.namedChild(i); - if (child?.type === 'simple_identifier') return child.text; - } - return undefined; - }, + extractName: extractKotlinMethodName, extractReturnType: extractKotlinReturnType, diff --git a/gitnexus/src/core/ingestion/method-extractors/configs/swift.ts b/gitnexus/src/core/ingestion/method-extractors/configs/swift.ts index 6c0fdfbe3..1ee630b6a 100644 --- a/gitnexus/src/core/ingestion/method-extractors/configs/swift.ts +++ b/gitnexus/src/core/ingestion/method-extractors/configs/swift.ts @@ -264,16 +264,23 @@ function extractSwiftAnnotations(node: SyntaxNode): string[] { export const swiftMethodConfig: MethodExtractionConfig = { language: SupportedLanguages.Swift, - // Keep this conservative until Swift type-shape coverage is expanded. - // TODO: Verify struct_declaration, enum_declaration, extension_declaration, actor_declaration - // node types once tree-sitter-swift loads on Node 22, and add them here if they are distinct. + // tree-sitter-swift collapses class / struct / enum / extension / actor into a + // single `class_declaration` node (distinguished by the `declaration_kind` + // field) — verified by real parse. There is NO separate `enum_declaration` + // node type, so it must NOT be listed here (it would fail the grammar-drift + // gate). The enum's owner node is therefore already covered by + // `class_declaration`; F79 only needed the enum BODY node added below. // protocol_declaration is a separate, confirmed node type. typeDeclarationNodes: ['class_declaration', 'protocol_declaration'], // function_declaration for class/struct methods, protocol_function_declaration for protocol methods methodNodeTypes: ['function_declaration', 'protocol_function_declaration'], - bodyNodeTypes: ['class_body', 'protocol_body'], + // class_body for class/struct/extension/actor, protocol_body for protocols, + // enum_class_body for enums (F79). Without enum_class_body the factory only + // reached enum methods via the generic findBodies fallback, which logs a + // dev-mode "body field type not in bodyNodeTypes" warning. + bodyNodeTypes: ['class_body', 'protocol_body', 'enum_class_body'], extractName: extractSwiftName, extractReturnType: extractSwiftReturnType, diff --git a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/ids.ts b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/ids.ts index 5da650393..7c4b57d35 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/ids.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/ids.ts @@ -115,6 +115,7 @@ export function resolveDefGraphId( type?: NodeLabel; parameterTypes?: readonly string[]; parameterTypeClasses?: readonly ParameterTypeClass[]; + parameterCount?: number; templateArguments?: readonly string[]; templateConstraints?: unknown; /** #1982 bridge-held namespace path; see `SymbolDefinition.namespacePrefix`. */ @@ -171,6 +172,15 @@ export function resolveDefGraphId( const pHit = nodeLookup.get(pKey); if (pHit !== undefined) return pHit; } + // Arity-disambiguating key (see node-lookup.ts): route a same-name overload + // to the structure node with the matching parameter count. Critical for a + // zero-arg overload (no parameterTypes) that would otherwise collapse onto a + // sibling overload via the source-order-dependent qualified key. + if (isOverloadableCallable(def.type) && def.parameterCount !== undefined) { + const aKey = qualifiedKey(filePath, def.type, `${qn}#${def.parameterCount}`); + const aHit = nodeLookup.get(aKey); + if (aHit !== undefined) return aHit; + } if ( (def.type === 'Class' || def.type === 'Struct' || diff --git a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/node-lookup.ts b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/node-lookup.ts index e39ddc4b7..9ab1649a1 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/node-lookup.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/node-lookup.ts @@ -109,6 +109,20 @@ export function buildGraphNodeLookup(graph: KnowledgeGraph): GraphNodeLookup { // Each overload is unique — set unconditionally. if (!lookup.has(pKey)) lookup.set(pKey, node.id); } + // Arity-disambiguating key: include the parameter count so two same-name + // overloads of DIFFERENT arity route to distinct graph nodes even when the + // shorter overload carries no parameter types (e.g. a Kotlin zero-arg + // secondary constructor vs a 2-arg one — both share the qualified key, whose + // first-write-wins assignment is source-order-dependent). The structure-phase + // node id encodes `#`; this mirrors it in the lookup keyspace so + // resolveDefGraphId can match by the def's own parameterCount. Same-arity + // overloads collapse onto one arity key (first-write-wins) — identical to the + // pre-existing qualified-key behavior, so no regression there. + const pCount = (props as { parameterCount?: number }).parameterCount; + if (pCount !== undefined && isOverloadableCallable(node.label)) { + const aKey = qualifiedKey(props.filePath, node.label, `${keyQualified}#${pCount}`); + if (!lookup.has(aKey)) lookup.set(aKey, node.id); + } const pClasses = (props as { parameterTypeClasses?: readonly ParameterTypeClass[] }) .parameterTypeClasses; const shapeTag = parameterShapeIdTag(pTypes, pClasses); diff --git a/gitnexus/src/core/ingestion/tree-sitter-queries.ts b/gitnexus/src/core/ingestion/tree-sitter-queries.ts index d6e25bd57..540ad1c74 100644 --- a/gitnexus/src/core/ingestion/tree-sitter-queries.ts +++ b/gitnexus/src/core/ingestion/tree-sitter-queries.ts @@ -1005,6 +1005,19 @@ export const CPP_QUERIES = ` declarator: (init_declarator declarator: (identifier) @name)) @definition.variable +; Structured bindings: auto [a, b] = makePair(); (one @name per bound identifier) +(declaration + declarator: (init_declarator + declarator: (structured_binding_declarator + (identifier) @name))) @definition.variable + +; Structured bindings, reference form: auto& [x, y] = tup; +(declaration + declarator: (init_declarator + declarator: (reference_declarator + (structured_binding_declarator + (identifier) @name)))) @definition.variable + ; Write access: obj.field = value (assignment_expression left: (field_expression @@ -1326,11 +1339,47 @@ export const KOTLIN_QUERIES = ` (function_declaration (simple_identifier) @name) @definition.function +; ── Secondary constructors (F49 sibling F48, issue #1919) ──────────────── +; "constructor(...) { }" inside a class body is a secondary_constructor with +; no name child — its only identity token is the anonymous "constructor" +; keyword, captured here as @name so the node is named "constructor" +; (matching kotlinMethodConfig.extractName). Multiple secondary constructors +; share that name but get distinct ids via the worker's # suffix. +(secondary_constructor + "constructor" @name) @definition.constructor + ; ── Properties ─────────────────────────────────────────────────────────── (property_declaration (variable_declaration (simple_identifier) @name)) @definition.property +; ── Destructuring declarations (F51, issue #1919) ──────────────────────── +; "val (a, b) = pair" binds several names through a multi_variable_declaration +; (NOT a variable_declaration), which the property rule above misses. Emit one +; @definition.property per bound name — the SAME label every other Kotlin val/var +; gets (KOTLIN_QUERIES has no @definition.variable rule, so a single "val x" +; is already a Property; matching that keeps destructured names consistent and +; out of the block-scope local-symbol pruner that drops Variable/Const/Static). +; The Kotlin "_" discard placeholder is filtered out here via (#not-eq? @name "_") +; — these locals have no enclosing class, so the field-extractor enrichment path +; never runs and cannot do the filtering itself. Each rule is a standalone +; pattern (NOT a top-level [...] alternation), so the predicate is safe under +; tree-sitter 0.21.1 (no sibling-branch drop). Loop destructuring +; "for ((k, v) in m)" nests the SAME multi_variable_declaration directly under the +; for_statement (no property_declaration wrapper); the scope-path loop binding only +; handles the single variable_declaration form, so this rule does not double-emit. +((property_declaration + (multi_variable_declaration + (variable_declaration + (simple_identifier) @name))) @definition.property + (#not-eq? @name "_")) + +((for_statement + (multi_variable_declaration + (variable_declaration + (simple_identifier) @name))) @definition.property + (#not-eq? @name "_")) + ; Primary constructor val/var parameters (data class, value class, regular class) ; binding_pattern_kind contains "val" or "var" — without it, the param is not a property (class_parameter @@ -1365,8 +1414,23 @@ export const KOTLIN_QUERIES = ` (type_identifier) @call.name)) @call ; ── Infix function calls (e.g., a to b, x until y) ────────────────────── +; tree-sitter-kotlin models infix_expression as three UNNAMED-FIELD children: +; (operand) (operator) (operand) — all three are simple_identifier for +; "a to b". The old rule "(infix_expression (simple_identifier) @call.name)" +; matched EVERY simple_identifier child, so it captured the operands a/b as +; spurious @call.name calls (F49, issue #1919). There is no operator: field to +; anchor on, so anchor positionally: the operator is the middle child, flanked +; by an operand on each side. End-anchored on both sides so only the lone +; middle simple_identifier (the infix function) is captured; chained +; "a to b to c" still matches each nested infix_expression's own operator. (infix_expression - (simple_identifier) @call.name) @call + . + (_) + . + (simple_identifier) @call.name + . + (_) + .) @call ; Write access: obj.field = value (assignment @@ -1413,6 +1477,13 @@ export const SWIFT_QUERIES = ` ; Properties (stored and computed) (property_declaration (pattern (simple_identifier) @name)) @definition.property +; Protocol property requirements (F75): "var title: String { get }" parses to a +; protocol_property_declaration (NOT property_declaration). Its name is a +; "name:" pattern field wrapping a value_binding_pattern + the bound +; simple_identifier; match the inner identifier so the requirement is emitted +; as a property symbol of the protocol. +(protocol_property_declaration (pattern (simple_identifier) @name)) @definition.property + ; Enum cases (enum_entry (simple_identifier) @name) @definition.property @@ -1459,12 +1530,36 @@ export const DART_QUERIES = ` (enum_declaration name: (identifier) @name) @definition.enum -; ── Type aliases ───────────────────────────────────────────────────────────── -; Anchor "=" after the name to avoid capturing the RHS type +; ── Type aliases — new-style (typedef Pred = bool Function(int);) ──────────── +; Anchor "=" after the name to avoid capturing the RHS type. The name is the +; first type_identifier (the alias), the RHS function_type follows the "=". (type_alias (type_identifier) @name "=") @definition.type +; ── Type aliases — old-style (typedef int Cmp(int a, int b);) ──────────────── +; The old-style function typedef has NO "=" — it parses as a type_alias whose +; children are: return type_identifier, NAME type_identifier, formal_parameter_list. +; Anchor @name as the type_identifier immediately before the parameter list so we +; capture the alias name (Cmp), not the leading return type (int). +(type_alias + (type_identifier) @name + . + (formal_parameter_list)) @definition.type + +; ── Type aliases — generic old-style (typedef int Cmp(T a, T b);) ───────── +; #1919 review CF2: a generic inserts a type_parameters node between the +; NAME and the parameter list, so the non-generic adjacency above misses it. +; Standalone pattern (NOT an alternation arm) anchoring @name immediately before +; type_parameters, which is immediately before the parameter list. The new-style +; "=" rule above is unanchored and already covers generic new-style (Mapper). +(type_alias + (type_identifier) @name + . + (type_parameters) + . + (formal_parameter_list)) @definition.type + ; ── Top-level functions (parent is program, not method_signature) ──────────── (program (function_signature @@ -1503,6 +1598,19 @@ export const DART_QUERIES = ` (initialized_identifier (identifier) @name))) @definition.property +; ── static const / static final / const class fields ──────────────────────── +; A "static const a = 1;" / "static final String b = ..., c = ...;" field parses +; with a static_final_declaration_list (NOT an initialized_identifier_list), so +; the field rules above miss them. One @name per static_final_declaration, so a +; multi-name declaration yields a Property per name. Anchored on declaration (not +; class_body) so top-level final/const variables — whose +; static_final_declaration_list is a direct child of program, not wrapped in a +; declaration — never match here. +(declaration + (static_final_declaration_list + (static_final_declaration + (identifier) @name))) @definition.property + ; ── Getters ────────────────────────────────────────────────────────────────── (method_signature (getter_signature @@ -1513,11 +1621,22 @@ export const DART_QUERIES = ` (setter_signature name: (identifier) @name)) @definition.property -; ── Top-level variable declarations (const maxSize = 100, final x = 5, var y = 0) ── -(declaration +; ── Top-level variable declarations ────────────────────────────────────────── +; Top-level Dart variables are NOT wrapped in a declaration node (that wrapper +; only occurs for class-body members). They sit as loose siblings under program: +; var name = 'x'; int x = 5; → initialized_identifier_list +; final int count = 3; const a = 1, b = 2; → static_final_declaration_list +; Anchor both rules under (program) so class-body fields (which reuse the same +; inner node types) are never matched here. One @name per declared name so +; multi-name forms (const a = 1, b = 2;) yield a Variable per name. +(program (initialized_identifier_list (initialized_identifier - (identifier) @name))) @definition.variable + (identifier) @name)) @definition.variable) +(program + (static_final_declaration_list + (static_final_declaration + (identifier) @name)) @definition.variable) ; ── Imports ────────────────────────────────────────────────────────────────── (import_or_export diff --git a/gitnexus/src/core/ingestion/utils/ast-helpers.ts b/gitnexus/src/core/ingestion/utils/ast-helpers.ts index 125c65aef..c391d7a0c 100644 --- a/gitnexus/src/core/ingestion/utils/ast-helpers.ts +++ b/gitnexus/src/core/ingestion/utils/ast-helpers.ts @@ -180,6 +180,7 @@ export const FUNCTION_NODE_TYPES = new Set([ 'anonymous_function', // Kotlin 'lambda_literal', + 'secondary_constructor', // F48: methodNodeTypes superset invariant // Swift 'init_declaration', 'deinit_declaration', @@ -583,6 +584,19 @@ export const findEnclosingClassInfo = ( ) { label = 'Interface'; } + // class_declaration with a `declaration_kind` field collapses several + // type kinds onto one node (tree-sitter-swift: class / struct / enum / + // extension / actor). The structure query labels struct → Struct and + // enum → Enum; refine the owner label to match so a member edge + // (HAS_METHOD / HAS_PROPERTY) anchors on the real Enum/Struct node id + // rather than a non-existent `Class:` id (F79). Gated on the field + // being present, so it is a no-op for grammars whose class_declaration + // has no `declaration_kind` field (e.g. Kotlin). + if (current.type === 'class_declaration' && label === 'Class') { + const declKind = current.childForFieldName?.('declaration_kind')?.text; + if (declKind === 'struct') label = 'Struct'; + else if (declKind === 'enum') label = 'Enum'; + } const templateArguments = extractTemplateArguments(nameNode.text); const classIdName = templateArguments !== undefined @@ -819,6 +833,62 @@ export const inferFunctionLabel = (nodeType: string): NodeLabel => /** Argument list node types shared between countCallArguments and call-resolution helpers. */ export const CALL_ARGUMENT_LIST_TYPES = new Set(['arguments', 'argument_list', 'value_arguments']); +/** + * Function/method parameter-list node types across grammars. Used to tell a + * PARAMETER-property (a constructor parameter that is also a class field, e.g. + * TypeScript `constructor(public name: string)`) apart from a function-BODY + * local: a property reached through one of these — rather than through the + * function's executable body — is a genuine class member, so the + * function-local-property guard must NOT strip its owner edge. + */ +export const PARAMETER_LIST_NODE_TYPES = new Set([ + 'formal_parameters', // TypeScript / JavaScript + 'parameters', // Python / C# + 'parameter_list', // Java / Go / C / Swift + 'function_value_parameters', // Kotlin + 'class_parameters', // Scala-like / future grammars +]); + +/** + * Executable local-scope boundaries for the property-ownership guard + * (`isFunctionLocalProperty` in parse-worker.ts). A `Property` capture whose + * nearest enclosing scope — walking up before any class container — is one of + * these executable bodies is a function-local binding, NOT a class member, so it + * must not receive a class `HAS_PROPERTY` owner edge. + * + * Derived from FUNCTION_NODE_TYPES, with two deliberate adjustments found by the + * #1919 review of the original guard: + * - EXCLUDES Dart's bare signature wrappers (`function_signature` / + * `method_signature`). A Dart getter/setter NAME lives under `method_signature`, + * yet it is a class-member declaration, not a local inside an executable body; + * treating the signature as a scope boundary OVER-stripped every Dart class + * accessor's owner edge. (Signatures are Dart-only; no language emits a + * legitimately-function-local Property under one.) + * - INCLUDES accessor + initializer bodies (Kotlin `anonymous_initializer` / + * `getter` / `setter`, Swift `computed_property` / `computed_getter` / + * `computed_setter` / `computed_modify`). Destructuring/locals inside these ARE + * function-local, yet they are absent from FUNCTION_NODE_TYPES; omitting them + * UNDER-stripped and emitted spurious class `HAS_PROPERTY` edges for + * `init {}` / accessor-body destructuring bindings. + * + * Kept separate from FUNCTION_NODE_TYPES because that set has many other consumers + * (e.g. enclosing-callable resolution) where signatures must remain function nodes + * and accessor bodies must not. + */ +export const LOCAL_SCOPE_BODY_NODE_TYPES: ReadonlySet = new Set( + [...FUNCTION_NODE_TYPES] + .filter((t) => t !== 'function_signature' && t !== 'method_signature') + .concat([ + 'anonymous_initializer', // Kotlin: init { } + 'getter', // Kotlin: val x get() { } + 'setter', // Kotlin: var x set(v) { } + 'computed_property', // Swift: var x: T { get set } + 'computed_getter', // Swift: get { } + 'computed_setter', // Swift: set { } + 'computed_modify', // Swift: _modify { } + ]), +); + // ============================================================================ // Generic AST traversal helpers (shared by parse-worker + php-helpers) // ============================================================================ diff --git a/gitnexus/src/core/ingestion/variable-extractors/configs/c-cpp.ts b/gitnexus/src/core/ingestion/variable-extractors/configs/c-cpp.ts index 29bf952ed..6df07b015 100644 --- a/gitnexus/src/core/ingestion/variable-extractors/configs/c-cpp.ts +++ b/gitnexus/src/core/ingestion/variable-extractors/configs/c-cpp.ts @@ -37,6 +37,55 @@ function extractCVarName(node: SyntaxNode): string | undefined { return undefined; } +/** + * Locate the `structured_binding_declarator` inside a `declaration` node, if any. + * + * C++ structured bindings (`auto [a, b] = pair;`) parse as: + * declaration → init_declarator → structured_binding_declarator → identifier+ + * The reference form (`auto& [x, y] = tup;`) wraps it one level deeper: + * declaration → init_declarator → reference_declarator → structured_binding_declarator + */ +function findStructuredBindingDeclarator(node: SyntaxNode): SyntaxNode | undefined { + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child?.type !== 'init_declarator') continue; + const declarator = child.childForFieldName('declarator'); + if (declarator?.type === 'structured_binding_declarator') return declarator; + // `auto& [x, y]` → reference_declarator wraps the structured_binding_declarator. + if (declarator?.type === 'reference_declarator') { + const inner = declarator.namedChildren.find( + (c: SyntaxNode) => c.type === 'structured_binding_declarator', + ); + if (inner) return inner; + } + } + return undefined; +} + +/** + * Extract every bound name from a C/C++ declaration. + * + * For a structured binding `auto [a, b] = ...;` this returns each bound + * identifier (`['a', 'b']`); the binding's `structured_binding_declarator` + * lists one `identifier` per name. For an ordinary single-name declaration + * (`int n = 0;`) it falls back to the single name extractor so behaviour is + * unchanged (issue #1919 F9). + */ +function extractCVarNames(node: SyntaxNode): string[] { + const binding = findStructuredBindingDeclarator(node); + if (binding !== undefined) { + const names: string[] = []; + for (let i = 0; i < binding.namedChildCount; i++) { + const child = binding.namedChild(i); + if (child?.type === 'identifier') names.push(child.text); + } + return names; + } + + const single = extractCVarName(node); + return single !== undefined ? [single] : []; +} + function extractCVarType(node: SyntaxNode): string | undefined { const typeNode = node.childForFieldName('type'); if (typeNode) return extractSimpleTypeName(typeNode) ?? typeNode.text?.trim(); @@ -60,6 +109,7 @@ const shared: Omit = { variableNodeTypes: ['declaration'], extractName: extractCVarName, + extractNames: extractCVarNames, extractType: extractCVarType, extractVisibility(node): VariableVisibility { diff --git a/gitnexus/src/core/ingestion/variable-extractors/configs/dart.ts b/gitnexus/src/core/ingestion/variable-extractors/configs/dart.ts index bd626212b..ee55db97b 100644 --- a/gitnexus/src/core/ingestion/variable-extractors/configs/dart.ts +++ b/gitnexus/src/core/ingestion/variable-extractors/configs/dart.ts @@ -8,56 +8,79 @@ import type { SyntaxNode } from '../../utils/ast-helpers.js'; /** * Dart variable extraction config. * - * Dart has top-level variable and constant declarations: - * - `const maxSize = 100;` - * - `final String name = "dart";` - * - `var counter = 0;` - * - `int x = 5;` + * Top-level Dart variables are NOT wrapped in a `declaration` node (that wrapper + * only occurs for class-body members). The structure query (`DART_QUERIES`) + * captures them as `@definition.variable` on the loose container node, which is + * one of two real shapes: * - * tree-sitter-dart uses: - * - declaration (with initialized_identifier_list) for file-scope variables + * - `var name = 'x';` / `int x = 5;` + * → initialized_identifier_list > initialized_identifier > identifier + * - `final int count = 3;` / `const a = 1, b = 2;` + * → static_final_declaration_list > static_final_declaration > identifier + * + * The variable extractor is invoked on that captured container node to enrich + * the Variable symbol with name(s)/type/const/mutable metadata. + * + * NOTE: the const/final modifier (`const_builtin` / `final_builtin`) and the + * type annotation (`type_identifier`) are siblings of the captured container — + * they live on the parent (program), NOT inside it — so const-ness and the type + * are read from the captured node's parent. (This is why the previous + * `type_identifier`-as-direct-child read found nothing.) */ -function extractDartVarName(node: SyntaxNode): string | undefined { - // declaration → initialized_variable_definition → identifier - for (let i = 0; i < node.namedChildCount; i++) { - const child = node.namedChild(i); - if (child?.type === 'initialized_variable_definition') { - const name = child.childForFieldName('name'); - if (name) return name.text; - // Fallback: first identifier - for (let j = 0; j < child.namedChildCount; j++) { - const gc = child.namedChild(j); - if (gc?.type === 'identifier') return gc.text; - } - } - // declaration → initialized_identifier_list → initialized_identifier → identifier - if (child?.type === 'initialized_identifier_list') { - for (let j = 0; j < child.namedChildCount; j++) { - const gc = child.namedChild(j); - if (gc?.type === 'initialized_identifier') { - const ident = gc.namedChildren.find((c: SyntaxNode) => c.type === 'identifier'); - if (ident) return ident.text; - } - } +/** The `initialized_identifier` / `static_final_declaration` name children. */ +function nameNodes(container: SyntaxNode): SyntaxNode[] { + const out: SyntaxNode[] = []; + for (let i = 0; i < container.namedChildCount; i++) { + const entry = container.namedChild(i); + if (!entry) continue; + if (entry.type === 'initialized_identifier' || entry.type === 'static_final_declaration') { + const ident = entry.firstNamedChild; + if (ident?.type === 'identifier') out.push(ident); } } - return undefined; + return out; } +function extractDartVarNames(node: SyntaxNode): string[] { + return nameNodes(node).map((n) => n.text); +} + +/** + * Scan the container's immediately-preceding siblings (the modifier / type + * nodes of THIS declaration), stopping at the previous statement's `;` so a + * neighbouring declaration's modifiers/type never bleed in. Top-level Dart + * declarations sit as loose siblings under `program` separated by `;`: + * final int count = 3; var name = 'x'; const a = 1, b = 2; + * so the leading `type_identifier` / `const_builtin` / `final_builtin` of a + * declaration are the siblings between the prior `;` and the captured container. + */ +function scanLeadingSiblings(node: SyntaxNode): SyntaxNode[] { + const out: SyntaxNode[] = []; + let sib = node.previousSibling; + while (sib !== null && sib.type !== ';') { + out.push(sib); + sib = sib.previousSibling; + } + return out; +} + +/** + * The declared type annotation, read from the captured container's leading + * sibling `type_identifier`. Returns undefined for inferred (`var`) + * declarations, which have an `inferred_type` sibling instead. + */ function extractDartVarType(node: SyntaxNode): string | undefined { - // Look for type_identifier directly on the node - for (let i = 0; i < node.namedChildCount; i++) { - const child = node.namedChild(i); - if (child?.type === 'type_identifier') return child.text; + for (const sib of scanLeadingSiblings(node)) { + if (sib.type === 'type_identifier') return sib.text; } return undefined; } -function hasDartKeyword(node: SyntaxNode, keyword: string): boolean { - for (let i = 0; i < node.childCount; i++) { - const child = node.child(i); - if (child?.text === keyword) return true; +/** Whether a `const_builtin` / `final_builtin` leads this declaration. */ +function hasReadonlyModifier(node: SyntaxNode): boolean { + for (const sib of scanLeadingSiblings(node)) { + if (sib.type === 'const_builtin' || sib.type === 'final_builtin') return true; } return false; } @@ -66,28 +89,32 @@ export const dartVariableConfig: VariableExtractionConfig = { language: SupportedLanguages.Dart, constNodeTypes: [], staticNodeTypes: [], - variableNodeTypes: ['declaration'], + // The two real top-level container shapes captured as @definition.variable. + variableNodeTypes: ['initialized_identifier_list', 'static_final_declaration_list'], - extractName: extractDartVarName, + extractName: (node) => extractDartVarNames(node)[0], + extractNames: extractDartVarNames, extractType: extractDartVarType, - extractVisibility(node): VariableVisibility { - const name = extractDartVarName(node); - if (!name) return 'public'; - // Dart convention: underscore prefix = library-private + extractVisibilityForName(_node, name): VariableVisibility { + // Dart convention: underscore prefix = library-private. return name.startsWith('_') ? 'private' : 'public'; }, - isConst(node) { - return hasDartKeyword(node, 'const') || hasDartKeyword(node, 'final'); + extractVisibility(node): VariableVisibility { + const first = extractDartVarNames(node)[0]; + if (!first) return 'public'; + return first.startsWith('_') ? 'private' : 'public'; }, + isConst: hasReadonlyModifier, + isStatic(_node) { - // Top-level Dart variables are not static + // Top-level Dart variables are not static. return false; }, isMutable(node) { - return !hasDartKeyword(node, 'const') && !hasDartKeyword(node, 'final'); + return !hasReadonlyModifier(node); }, }; diff --git a/gitnexus/src/core/ingestion/variable-extractors/configs/jvm.ts b/gitnexus/src/core/ingestion/variable-extractors/configs/jvm.ts index 6ebaa3024..4c5b96757 100644 --- a/gitnexus/src/core/ingestion/variable-extractors/configs/jvm.ts +++ b/gitnexus/src/core/ingestion/variable-extractors/configs/jvm.ts @@ -54,6 +54,19 @@ export const javaVariableConfig: VariableExtractionConfig = { }, }; +/** Single-binding name of a Kotlin property_declaration: + * property_declaration → variable_declaration → simple_identifier. */ +function kotlinSingleVarName(node: SyntaxNode): string | undefined { + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child?.type === 'variable_declaration') { + const ident = child.namedChildren.find((c: SyntaxNode) => c.type === 'simple_identifier'); + return ident?.text; + } + } + return undefined; +} + /** * Kotlin variable extraction config. * @@ -66,16 +79,32 @@ export const kotlinVariableConfig: VariableExtractionConfig = { staticNodeTypes: [], variableNodeTypes: ['property_declaration'], - extractName(node) { - // property_declaration → variable_declaration → simple_identifier - for (let i = 0; i < node.namedChildCount; i++) { - const child = node.namedChild(i); - if (child?.type === 'variable_declaration') { - const ident = child.namedChildren.find((c: SyntaxNode) => c.type === 'simple_identifier'); - return ident?.text; + extractName: kotlinSingleVarName, + + // F51 (issue #1919): destructuring declarations bind several names at one + // node. `val (a, b) = pair` parses as + // property_declaration → multi_variable_declaration + // → variable_declaration → simple_identifier (one per name) + // (real-parse-verified). The single-name `extractName` above misses these + // entirely. When a multi_variable_declaration is present we return EACH + // bound name; the Kotlin `_` placeholder is a discard and is skipped. A + // plain single declaration falls through to the existing variable_declaration + // shape so `val x = 1` still yields exactly one name (no double-count). + extractNames(node) { + const multi = node.namedChildren.find( + (c: SyntaxNode) => c.type === 'multi_variable_declaration', + ); + if (multi !== undefined) { + const names: string[] = []; + for (const decl of multi.namedChildren) { + if (decl.type !== 'variable_declaration') continue; + const ident = decl.namedChildren.find((c: SyntaxNode) => c.type === 'simple_identifier'); + if (ident !== undefined && ident.text !== '_') names.push(ident.text); } + return names; } - return undefined; + const single = kotlinSingleVarName(node); + return single !== undefined ? [single] : []; }, extractType(node) { diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index 9a3837494..b0f787855 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -58,6 +58,7 @@ import { getLanguageFromFilename } from 'gitnexus-shared'; import { buildConcreteTypedefDefinitionRanges, FUNCTION_NODE_TYPES, + findAncestorBeforeBoundary, getDefinitionNodeFromCaptures, findEnclosingClassInfo, findObjectLiteralBindingInfo, @@ -69,6 +70,8 @@ import { isQualifiableScopeLabel, qualifyRustImplTargetByModScope, CLASS_CONTAINER_TYPES, + PARAMETER_LIST_NODE_TYPES, + LOCAL_SCOPE_BODY_NODE_TYPES, type SyntaxNode, } from '../utils/ast-helpers.js'; import { extractCallArgTypes, type MixedChainStep } from '../utils/call-analysis.js'; @@ -1779,14 +1782,52 @@ const processFileGroup = ( provider.classExtractor!.qualifyScopeName?.(node, simpleName) ?? null : undefined; - const enclosingClassInfo = needsOwner - ? cachedFindEnclosingClassInfo( - nameNode || definitionNode, - file.path, - provider.resolveEnclosingOwner, - getQualifiedOwnerName, - ) - : null; + // A Property declared inside a function/lambda BODY is a function-LOCAL + // binding (e.g. Kotlin `val (a,b) = pair` or a `for ((k,v) in m)` loop + // destructuring emitted as `@definition.property` to dodge the local-symbol + // pruner), NOT a class member. Such locals must not get a HAS_PROPERTY owner + // edge from the enclosing class. Detect them by walking from the def node: + // if a function-like ancestor is reached BEFORE any class container, the + // property is enclosed by a function. Language-agnostic — genuine class + // fields sit directly in the class body with no intervening function, so + // they are unaffected (#1919 review CF3). + // + // EXCEPTION: a constructor PARAMETER property (TypeScript + // `constructor(public name: string)`) is also enclosed by a function, but + // it IS a class member — it is reached through the parameter list, not the + // executable body. So only strip the owner when the property is NOT inside + // a parameter list of that function (i.e. it's a body local). + const propOwnerNode = nameNode || definitionNode; + // A Property is function-local (and must NOT get a class HAS_PROPERTY owner) + // when its nearest enclosing executable body — reached before any class + // container — is a function/accessor/initializer body, AND it is not a + // constructor parameter-property (rescued by the param-list carve-out). + // Uses LOCAL_SCOPE_BODY_NODE_TYPES (not FUNCTION_NODE_TYPES): the latter + // mis-includes Dart bare signatures (over-stripping accessors) and omits + // Kotlin/Swift init+accessor bodies (under-stripping their locals) — see + // the #1919 review of this guard. + const isFunctionLocalProperty = + nodeLabel === 'Property' && + propOwnerNode !== undefined && + findAncestorBeforeBoundary( + propOwnerNode, + LOCAL_SCOPE_BODY_NODE_TYPES, + CLASS_CONTAINER_TYPES, + ) !== null && + findAncestorBeforeBoundary( + propOwnerNode, + PARAMETER_LIST_NODE_TYPES, + LOCAL_SCOPE_BODY_NODE_TYPES, + ) === null; + const enclosingClassInfo = + needsOwner && !isFunctionLocalProperty + ? cachedFindEnclosingClassInfo( + nameNode || definitionNode, + file.path, + provider.resolveEnclosingOwner, + getQualifiedOwnerName, + ) + : null; const enclosingClassId = enclosingClassInfo?.qualifiedClassId ?? enclosingClassInfo?.classId ?? null; const objectLiteralOwnerInfo = diff --git a/gitnexus/test/fixtures/lang-resolution/c-coverage/main.c b/gitnexus/test/fixtures/lang-resolution/c-coverage/main.c new file mode 100644 index 000000000..42cfe583d --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/c-coverage/main.c @@ -0,0 +1,9 @@ +#include +#include "local.h" + +#define HDR "computed.h" +#include HDR + +int main(void) { + return 0; +} diff --git a/gitnexus/test/fixtures/lang-resolution/cpp-structured-binding/App.cpp b/gitnexus/test/fixtures/lang-resolution/cpp-structured-binding/App.cpp index bd2e390f3..f8b6a1f87 100644 --- a/gitnexus/test/fixtures/lang-resolution/cpp-structured-binding/App.cpp +++ b/gitnexus/test/fixtures/lang-resolution/cpp-structured-binding/App.cpp @@ -15,3 +15,8 @@ void processRepoMap(std::map repoMap) { repo.save(); } } + +// F9 — plain (non-for-loop) structured binding declaration. Each bound name +// must emit its own Variable node. +std::pair makePair(); +auto [firstId, secondId] = makePair(); diff --git a/gitnexus/test/fixtures/lang-resolution/dart-accessor-owner/main.dart b/gitnexus/test/fixtures/lang-resolution/dart-accessor-owner/main.dart new file mode 100644 index 000000000..a27985a88 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/dart-accessor-owner/main.dart @@ -0,0 +1,12 @@ +// CF3 review (#1919): a Dart class getter/setter is a class-member declaration +// (its name lives under method_signature), NOT a function-local — it must keep +// its HAS_PROPERTY owner edge from the class. +class Box { + int normalField = 1; + + int get answer => 42; + + set answer(int v) { + normalField = v; + } +} diff --git a/gitnexus/test/fixtures/lang-resolution/dart-coverage/typedefs.dart b/gitnexus/test/fixtures/lang-resolution/dart-coverage/typedefs.dart new file mode 100644 index 000000000..ac6413799 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/dart-coverage/typedefs.dart @@ -0,0 +1,5 @@ +typedef int Cmp(int a, int b); +typedef int Cmp2(T a, T b); +typedef Pred = bool Function(int); +typedef Mapper = T Function(T); +typedef int _Internal(int); diff --git a/gitnexus/test/fixtures/lang-resolution/dart-static-fields/config.dart b/gitnexus/test/fixtures/lang-resolution/dart-static-fields/config.dart new file mode 100644 index 000000000..603c7d71c --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/dart-static-fields/config.dart @@ -0,0 +1,6 @@ +class Config { + static const int maxRetries = 3; + static final String host = 'localhost', scheme = 'https'; + int port = 8080; + static const _secret = 'hidden'; +} diff --git a/gitnexus/test/fixtures/lang-resolution/dart-toplevel-vars/globals.dart b/gitnexus/test/fixtures/lang-resolution/dart-toplevel-vars/globals.dart new file mode 100644 index 000000000..a77a4de3a --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/dart-toplevel-vars/globals.dart @@ -0,0 +1,7 @@ +final int count = 3; +var name = 'x'; +const a = 1, b = 2; + +class Holder { + int z = 0; +} diff --git a/gitnexus/test/fixtures/lang-resolution/kotlin-companion-fields/Companions.kt b/gitnexus/test/fixtures/lang-resolution/kotlin-companion-fields/Companions.kt new file mode 100644 index 000000000..6f823a891 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/kotlin-companion-fields/Companions.kt @@ -0,0 +1,15 @@ +package coverage + +class C { + companion object { + const val TAG = "c" + val instances = 0 + fun create() {} + } +} + +class NamedComp { + companion object Factory { + val cfgX = 1 + } +} diff --git a/gitnexus/test/fixtures/lang-resolution/kotlin-coverage/callable_refs.kt b/gitnexus/test/fixtures/lang-resolution/kotlin-coverage/callable_refs.kt new file mode 100644 index 000000000..9dee807e8 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/kotlin-coverage/callable_refs.kt @@ -0,0 +1,15 @@ +package coverage + +fun topLevelFn(): Int = 1 + +class Obj { + fun method(): String = "m" +} + +fun useCallableRefs() { + val a = ::topLevelFn + val b = String::length + val obj = Obj() + val c = obj::method + val d = Type::new +} diff --git a/gitnexus/test/fixtures/lang-resolution/kotlin-destructuring/Destructuring.kt b/gitnexus/test/fixtures/lang-resolution/kotlin-destructuring/Destructuring.kt new file mode 100644 index 000000000..53c421d26 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/kotlin-destructuring/Destructuring.kt @@ -0,0 +1,8 @@ +package coverage + +fun useDestructuring(pair: Pair, map: Map) { + val (a, b) = pair + val (_, second) = pair + for ((k, v) in map) { } + val x = 1 +} diff --git a/gitnexus/test/fixtures/lang-resolution/kotlin-local-property-owner/Locals.kt b/gitnexus/test/fixtures/lang-resolution/kotlin-local-property-owner/Locals.kt new file mode 100644 index 000000000..2a9cf86fc --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/kotlin-local-property-owner/Locals.kt @@ -0,0 +1,31 @@ +package coverage + +class C(val field: Int) { + val classProp: Int = field + + // CF3 review (#1919): destructuring inside an init {} block is a + // function-local binding (anonymous_initializer is an executable body), + // NOT a class member — it must not be owned by C. + init { + val (ix, iy) = field to field + println(ix) + println(iy) + } + + // ...and the same for locals inside a property accessor (getter) body. + // `derived` is a genuine class property (owned); `gx`/`gy` are not. + val derived: Int + get() { + val (gx, gy) = field to field + return gx + gy + } + + fun process(map: Map) { + for ((k, v) in map) { + println(k) + println(v) + } + val pair = Pair(1, 2) + val (a, b) = pair + } +} diff --git a/gitnexus/test/fixtures/lang-resolution/kotlin-secondary-ctor/Constructors.kt b/gitnexus/test/fixtures/lang-resolution/kotlin-secondary-ctor/Constructors.kt new file mode 100644 index 000000000..c34d90a54 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/kotlin-secondary-ctor/Constructors.kt @@ -0,0 +1,14 @@ +package coverage + +class Point(val x: Int) { + constructor(a: Int, b: String) : this(a) { helper() } + constructor() : this(0) { helper(); other() } + fun describe(): String = "p" +} + +class OnlyPrimary(val v: Int) { + fun method(): Int = v +} + +fun helper() {} +fun other() {} diff --git a/gitnexus/test/fixtures/lang-resolution/swift-enum-members/Direction.swift b/gitnexus/test/fixtures/lang-resolution/swift-enum-members/Direction.swift new file mode 100644 index 000000000..c00e5a6db --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/swift-enum-members/Direction.swift @@ -0,0 +1,22 @@ +enum Direction { + case north + case south + + func describe() -> String { + return "direction" + } + + var label: String { + return "dir" + } + + static func make() -> Direction { + return .north + } +} + +class Compass { + func heading() -> String { + return "n" + } +} diff --git a/gitnexus/test/fixtures/lang-resolution/swift-protocol-property/Repository.swift b/gitnexus/test/fixtures/lang-resolution/swift-protocol-property/Repository.swift new file mode 100644 index 000000000..700efefd4 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/swift-protocol-property/Repository.swift @@ -0,0 +1,9 @@ +protocol Repository { + var title: String { get } + var count: Int { get set } + static var shared: Repository { get } +} + +class FileRepository { + var name: String = "" +} diff --git a/gitnexus/test/fixtures/swift-captures-golden/expected-captures.json b/gitnexus/test/fixtures/swift-captures-golden/expected-captures.json index aabf63ddd..1482e4674 100644 --- a/gitnexus/test/fixtures/swift-captures-golden/expected-captures.json +++ b/gitnexus/test/fixtures/swift-captures-golden/expected-captures.json @@ -59,6 +59,10 @@ "captureGroups": 12, "digest": "a9c40a2dcb51c6e6a9adbaa869183eed90a81302729d4d7a0c300084ea2978e1" }, + "swift-enum-members/Direction.swift": { + "captureGroups": 18, + "digest": "6378c8860113e9d0331f4da4d05e64e6933a995493131d2f2165f64f82978b0b" + }, "swift-export-visibility/App.swift": { "captureGroups": 10, "digest": "6be49190e8a120ed1c3c3812afaa717f60fc12dae82d4b1b7158a358c53ea75a" @@ -199,6 +203,10 @@ "captureGroups": 10, "digest": "bd01b5adcd523ceee95772cc74ded0dbc1d70449fc9e9d17e975299dcbac783f" }, + "swift-protocol-property/Repository.swift": { + "captureGroups": 7, + "digest": "28e5a3c93f0dc2c0f940a6def1491d4a339e86fdc71858ecd61f4aea372afd6a" + }, "swift-qualified-base/Sources/Derived.swift": { "captureGroups": 4, "digest": "39e6ba35775ce624d1fb982204d2e2f0fcbbaf3046e317242602d9f245eb5a9d" diff --git a/gitnexus/test/integration/resolvers/c-coverage.test.ts b/gitnexus/test/integration/resolvers/c-coverage.test.ts new file mode 100644 index 000000000..7967435d0 --- /dev/null +++ b/gitnexus/test/integration/resolvers/c-coverage.test.ts @@ -0,0 +1,87 @@ +/** + * Regression tests for C/C++ scope-resolution coverage gaps (issue #1919). + * + * F5 — a computed `#include MACRO` must NOT become a literal import source. + * The macro name is an `identifier` path node (not a header path), so emitting + * it would create a garbage import edge. Literal `` (system_lib_string) + * and `"local.h"` (string_literal) includes must keep emitting correct sources. + */ +import { describe, it, expect } from 'vitest'; +import fs from 'fs'; +import path from 'path'; +import { fileURLToPath } from 'url'; +import { emitCScopeCaptures } from '../../../src/core/ingestion/languages/c/index.js'; +import { emitCppScopeCaptures } from '../../../src/core/ingestion/languages/cpp/index.js'; +import type { CaptureMatch } from 'gitnexus-shared'; + +const here = path.dirname(fileURLToPath(import.meta.url)); +const FIXTURE = path.resolve( + here, + '..', + '..', + 'fixtures', + 'lang-resolution', + 'c-coverage', + 'main.c', +); + +function importSources(matches: readonly CaptureMatch[]): string[] { + return matches + .filter((m) => m['@import.source'] !== undefined) + .map((m) => m['@import.source'].text); +} + +// --------------------------------------------------------------------------- +// F5 — computed #include MACRO is not emitted as a literal import source (C) +// --------------------------------------------------------------------------- + +describe('F5 — computed #include MACRO (C)', () => { + const src = fs.readFileSync(FIXTURE, 'utf8'); + const matches = emitCScopeCaptures(src, 'main.c') as CaptureMatch[]; + const sources = importSources(matches); + + it('emits import sources for literal and "local.h"', () => { + expect(sources).toContain('stdio.h'); + expect(sources).toContain('local.h'); + }); + + it('does NOT emit a garbage import source for #include HDR', () => { + // The macro name and any expansion text must never surface as a source. + expect(sources).not.toContain('HDR'); + expect(sources).not.toContain('computed.h'); + // Exactly the two literal includes — no spurious third source. + expect(sources).toHaveLength(2); + }); + + it('marks the system header as a system include', () => { + const systemSources = matches + .filter((m) => m['@import.system'] !== undefined) + .map((m) => m['@import.source']?.text); + expect(systemSources).toContain('stdio.h'); + expect(systemSources).not.toContain('local.h'); + }); +}); + +// --------------------------------------------------------------------------- +// F5 — computed #include MACRO is not emitted as a literal import source (C++) +// --------------------------------------------------------------------------- + +describe('F5 — computed #include MACRO (C++)', () => { + // Inline C++ source mixing literal + computed includes — the cpp decomposer + // path (splitCppInclude) is independently exercised here. + const src = + '#include \n#include "User.h"\n#define HDR "computed.h"\n#include HDR\n\nint main() { return 0; }\n'; + const matches = emitCppScopeCaptures(src, 'main.cpp') as CaptureMatch[]; + const sources = importSources(matches); + + it('emits import sources for literal and "User.h"', () => { + expect(sources).toContain('map'); + expect(sources).toContain('User.h'); + }); + + it('does NOT emit a garbage import source for #include HDR', () => { + expect(sources).not.toContain('HDR'); + expect(sources).not.toContain('computed.h'); + expect(sources).toHaveLength(2); + }); +}); diff --git a/gitnexus/test/integration/resolvers/cpp.test.ts b/gitnexus/test/integration/resolvers/cpp.test.ts index 961454c11..bfae951e3 100644 --- a/gitnexus/test/integration/resolvers/cpp.test.ts +++ b/gitnexus/test/integration/resolvers/cpp.test.ts @@ -893,6 +893,21 @@ describe('C++ structured binding in range-for', () => { ); expect(wrongSave).toBeUndefined(); }); + + // F9 — a plain structured-binding declaration emits one Variable per bound name. + it('emits a Variable for each name in `auto [firstId, secondId] = makePair();`', () => { + const vars = getNodesByLabelFull(result, 'Variable').map((v) => v.name); + expect(vars).toContain('firstId'); + expect(vars).toContain('secondId'); + }); + + it('classifies top-level structured-binding names as module scope', () => { + const bound = getNodesByLabelFull(result, 'Variable').filter( + (v) => v.name === 'firstId' || v.name === 'secondId', + ); + expect(bound).toHaveLength(2); + for (const v of bound) expect(v.properties.scope).toBe('module'); + }); }); // --------------------------------------------------------------------------- diff --git a/gitnexus/test/integration/resolvers/dart-coverage.test.ts b/gitnexus/test/integration/resolvers/dart-coverage.test.ts new file mode 100644 index 000000000..d7c0113ec --- /dev/null +++ b/gitnexus/test/integration/resolvers/dart-coverage.test.ts @@ -0,0 +1,118 @@ +/** + * Regression tests for Dart scope-resolution / structure coverage gaps + * (issue #1919). Mirrors python-parsing-coverage.test.ts: the F28 scope-capture + * assertions exercise emitDartScopeCaptures directly, and a pipeline check + * verifies the TypeAlias symbol exists end-to-end. + * + * F28 — old-style function typedef (`typedef int Cmp(int a, int b);`) was never + * captured: DART_SCOPE_QUERY had no type_alias rule, and DART_QUERIES only + * captured the new-style (`=`-anchored) form. Both forms must now surface as a + * type-alias declaration / TypeAlias symbol. + * + * #1919 review CF2 — the GENERIC forms (`typedef int Cmp2(T a, T b);` and + * `typedef Mapper = T Function(T);`) were still dropped: a generic + * type_parameters node sits between the alias name and the next anchor, so the + * non-generic adjacency patterns never matched. Standalone generic patterns now + * capture them too. + */ +import { describe, it, expect, beforeAll } from 'vitest'; +import path from 'path'; +import { emitDartScopeCaptures } from '../../../src/core/ingestion/languages/dart/captures.js'; +import { FIXTURES, getNodesByLabel, runPipelineFromRepo, type PipelineResult } from './helpers.js'; +import { + isLanguageAvailable, + loadParser, + loadLanguage, +} from '../../../src/core/tree-sitter/parser-loader.js'; +import { SupportedLanguages } from '../../../src/config/supported-languages.js'; +import type { CaptureMatch } from 'gitnexus-shared'; + +let dartAvailable = isLanguageAvailable(SupportedLanguages.Dart); +if (dartAvailable) { + try { + await loadParser(); + await loadLanguage(SupportedLanguages.Dart); + } catch { + dartAvailable = false; + } +} + +const TYPEDEFS = `typedef int Cmp(int a, int b); +typedef int Cmp2(T a, T b); +typedef Pred = bool Function(int); +typedef Mapper = T Function(T); +typedef int _Internal(int);`; + +/** All @declaration.type_alias matches, as (name) tuples. */ +function typeAliasNames(src: string): string[] { + const matches = emitDartScopeCaptures(src, 'test.dart') as CaptureMatch[]; + return matches + .filter((m) => m['@declaration.type_alias'] !== undefined) + .map((m) => m['@declaration.name']?.text) + .filter((n): n is string => Boolean(n)); +} + +// --------------------------------------------------------------------------- +// F28 — typedef capture (scope layer) +// --------------------------------------------------------------------------- + +describe.skipIf(!dartAvailable)('F28 — Dart typedef capture (scope layer)', () => { + it('captures the old-style function typedef as a type-alias declaration', () => { + const names = typeAliasNames(TYPEDEFS); + expect(names).toContain('Cmp'); + }); + + it('still captures the new-style typedef (regression)', () => { + const names = typeAliasNames(TYPEDEFS); + expect(names).toContain('Pred'); + }); + + it('captures a private old-style typedef', () => { + const names = typeAliasNames(TYPEDEFS); + expect(names).toContain('_Internal'); + }); + + it('captures the generic old-style typedef (CF2)', () => { + const names = typeAliasNames(TYPEDEFS); + expect(names).toContain('Cmp2'); + }); + + it('captures the generic new-style typedef (CF2)', () => { + const names = typeAliasNames(TYPEDEFS); + expect(names).toContain('Mapper'); + }); + + it('emits exactly one declaration per typedef (no double-match)', () => { + const names = typeAliasNames(TYPEDEFS); + expect(names.sort()).toEqual(['Cmp', 'Cmp2', 'Pred', 'Mapper', '_Internal'].sort()); + }); +}); + +// --------------------------------------------------------------------------- +// F28 — typedef symbols exist end-to-end (structure phase) +// --------------------------------------------------------------------------- + +describe.skipIf(!dartAvailable)('F28 — Dart typedef symbols (end-to-end)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'dart-coverage'), () => {}); + }, 60000); + + it('creates TypeAlias nodes for old-style, new-style, generic, and private typedefs', () => { + const aliases = getNodesByLabel(result, 'TypeAlias'); + expect(aliases).toContain('Cmp'); // old-style (covers F28) + expect(aliases).toContain('Cmp2'); // generic old-style (covers CF2) + expect(aliases).toContain('Pred'); // new-style (regression) + expect(aliases).toContain('Mapper'); // generic new-style (covers CF2) + expect(aliases).toContain('_Internal'); // private old-style + }); + + it('emits exactly one TypeAlias per typedef (no duplicates)', () => { + const aliases = getNodesByLabel(result, 'TypeAlias'); + const fromFixture = aliases.filter((n) => + ['Cmp', 'Cmp2', 'Pred', 'Mapper', '_Internal'].includes(n), + ); + expect(fromFixture.sort()).toEqual(['Cmp', 'Cmp2', 'Mapper', 'Pred', '_Internal'].sort()); + }); +}); diff --git a/gitnexus/test/integration/resolvers/dart.test.ts b/gitnexus/test/integration/resolvers/dart.test.ts index 41a2db88a..e362edc33 100644 --- a/gitnexus/test/integration/resolvers/dart.test.ts +++ b/gitnexus/test/integration/resolvers/dart.test.ts @@ -707,6 +707,118 @@ describe.skipIf(!dartAvailable)('Dart named-constructor body (no file drop)', () }); }); +// --------------------------------------------------------------------------- +// F26 (issue #1919): static const / static final class fields. +// `static const`/`static final` fields parse with a static_final_declaration_list +// (not initialized_identifier_list), so the legacy field rules missed them and +// no Property node was created end-to-end. They must surface as Property nodes +// marked static + readonly, one per name in a multi-name declaration. +// --------------------------------------------------------------------------- + +describe.skipIf(!dartAvailable)('Dart static const/final fields (F26)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'dart-static-fields'), () => {}); + }, 60000); + + it('captures static const and static final fields as Properties', () => { + const properties = getNodesByLabel(result, 'Property'); + expect(properties).toContain('maxRetries'); // static const + expect(properties).toContain('host'); // static final, name 1 of 2 + expect(properties).toContain('scheme'); // static final, name 2 of 2 + expect(properties).toContain('port'); // instance field (regression) + expect(properties).toContain('_secret'); // private static const + }); + + it('emits HAS_PROPERTY edges for static fields', () => { + const propEdges = getRelationships(result, 'HAS_PROPERTY'); + expect(edgeSet(propEdges)).toEqual( + expect.arrayContaining([ + 'Config → maxRetries', + 'Config → host', + 'Config → scheme', + 'Config → port', + 'Config → _secret', + ]), + ); + }); + + it('marks static const/final fields as static + readonly', () => { + const props = getNodesByLabelFull(result, 'Property'); + for (const name of ['maxRetries', 'host', 'scheme', '_secret']) { + const p = props.find((n) => n.name === name); + expect(p, name).toBeDefined(); + expect(p!.properties.isStatic, name).toBe(true); + expect(p!.properties.isReadonly, name).toBe(true); + } + }); + + it('keeps the instance field non-static, non-readonly (regression)', () => { + const props = getNodesByLabelFull(result, 'Property'); + const port = props.find((n) => n.name === 'port'); + expect(port).toBeDefined(); + expect(port!.properties.isStatic).toBe(false); + expect(port!.properties.isReadonly).toBe(false); + }); +}); + +// --------------------------------------------------------------------------- +// F29 (issue #1919): top-level Dart variables. Top-level vars are loose +// siblings under `program` (no `declaration` wrapper), so the structure query +// never captured them and no Variable node existed end-to-end. They must now +// surface as Variable nodes with the real type/const metadata read from the +// captured container's leading siblings (not a phantom `type` field). +// --------------------------------------------------------------------------- + +describe.skipIf(!dartAvailable)('Dart top-level variables (F29)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'dart-toplevel-vars'), () => {}); + }, 60000); + + it('captures top-level variables (typed final, inferred var, multi-name const)', () => { + const vars = getNodesByLabel(result, 'Variable'); + expect(vars).toContain('count'); // final int count = 3; (covers F29) + expect(vars).toContain('name'); // var name = 'x'; + expect(vars).toContain('a'); // const a = 1, b = 2; + expect(vars).toContain('b'); + }); + + it('reads the real type for a typed final and leaves an inferred var untyped', () => { + const vars = getNodesByLabelFull(result, 'Variable'); + const count = vars.find((n) => n.name === 'count'); + expect(count).toBeDefined(); + expect(count!.properties.declaredType).toBe('int'); + expect(count!.properties.isConst).toBe(true); + + const name = vars.find((n) => n.name === 'name'); + expect(name).toBeDefined(); + // inferred `var` → no declaredType from a phantom field; mutable. + expect(name!.properties.declaredType).toBeUndefined(); + expect(name!.properties.isMutable).toBe(true); + }); + + it('keeps the class instance field as a Property, not a Variable (regression)', () => { + const vars = getNodesByLabel(result, 'Variable'); + expect(vars).not.toContain('z'); + const props = getNodesByLabel(result, 'Property'); + expect(props).toContain('z'); + }); + + it('does NOT emit top-level vars as Property nodes (#1919 review CF4)', () => { + // Guards the `(program …)` vs `(declaration …)` anchor split: top-level + // siblings under `program` must surface as Variable, never Property. If the + // top-level anchor regressed to the class-field `(declaration …)` rule, these + // names would mis-classify as class Properties. + const props = getNodesByLabel(result, 'Property'); + for (const name of ['count', 'a', 'b', 'name']) { + expect(props).not.toContain(name); + } + }); +}); + // --------------------------------------------------------------------------- // Heritage cross-file simple-name collision (PR #1970 tri-review P2). // console_logger.dart and file_logger.dart each declare `class Logger`; each @@ -736,3 +848,26 @@ describe.skipIf(!dartAvailable)('Dart heritage cross-file name collision', () => expect(fileEdge!.targetFilePath).toContain('file_logger.dart'); }); }); + +// --------------------------------------------------------------------------- +// CF3 (#1919 review): Dart class getters/setters keep their class owner edge. +// --------------------------------------------------------------------------- +// A Dart accessor's name lives under `method_signature`; the CF3 owner-strip +// guard must NOT treat that signature as an executable body and strip the +// HAS_PROPERTY owner (the over-strip regression this guards against). +describe('CF3 — Dart class accessors keep their class owner', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'dart-accessor-owner'), () => {}); + }, 60000); + + it('owns the getter/setter property `answer` and the stored field under Box', () => { + const owned = getRelationships(result, 'HAS_PROPERTY') + .filter((e) => e.source === 'Box') + .map((e) => e.target) + .sort(); + expect(owned).toContain('answer'); + expect(owned).toContain('normalField'); + }); +}); diff --git a/gitnexus/test/integration/resolvers/kotlin-coverage.test.ts b/gitnexus/test/integration/resolvers/kotlin-coverage.test.ts new file mode 100644 index 000000000..74ac3df49 --- /dev/null +++ b/gitnexus/test/integration/resolvers/kotlin-coverage.test.ts @@ -0,0 +1,176 @@ +/** + * Regression tests for Kotlin parsing-layer coverage gaps (issue #1919). + * + * Mirrors dart-coverage.test.ts / python-parsing-coverage.test.ts: the scope-layer + * assertions exercise emitKotlinScopeCaptures directly; F49 also exercises the + * legacy KOTLIN_QUERIES structure-query bank (the live spurious-edge source). + * + * F47 — callable references (`::method`, `Type::new`, `obj::method`) were never + * captured: KOTLIN_SCOPE_QUERY had no callable_reference rule, so they never + * participated in call-graph resolution. + * + * F49 — the legacy KOTLIN_QUERIES infix rule captured ALL three simple_identifier + * children of an infix_expression (`a to b` → `a`, `to`, `b`) as @call.name, + * emitting spurious call references for the operands. The fix anchors the + * capture to the operator (the middle child) only. + */ +import { describe, it, expect, beforeAll } from 'vitest'; +import path from 'path'; +import Parser from 'tree-sitter'; +import Kotlin from 'tree-sitter-kotlin'; +import { emitKotlinScopeCaptures } from '../../../src/core/ingestion/languages/kotlin/captures.js'; +import { KOTLIN_QUERIES } from '../../../src/core/ingestion/tree-sitter-queries.js'; +import { FIXTURES, getRelationships, runPipelineFromRepo, type PipelineResult } from './helpers.js'; +import type { CaptureMatch } from 'gitnexus-shared'; + +// --------------------------------------------------------------------------- +// F47 — callable references (scope layer) +// --------------------------------------------------------------------------- + +const CALLABLE_REFS = `fun useCallableRefs() { + val a = ::topLevelFn + val b = String::length + val c = obj::method + val d = Type::new +}`; + +/** All call references emitted for the source, as { name, receiver, form }. */ +function callReferences( + src: string, +): Array<{ name: string; receiver?: string; form: 'free' | 'member' }> { + const matches = emitKotlinScopeCaptures(src, 'test.kt') as CaptureMatch[]; + const out: Array<{ name: string; receiver?: string; form: 'free' | 'member' }> = []; + for (const m of matches) { + if (m['@reference.call.free'] !== undefined && m['@reference.name'] !== undefined) { + out.push({ name: m['@reference.name'].text, form: 'free' }); + } else if (m['@reference.call.member'] !== undefined && m['@reference.name'] !== undefined) { + out.push({ + name: m['@reference.name'].text, + receiver: m['@reference.receiver']?.text, + form: 'member', + }); + } + } + return out; +} + +describe('F47 — Kotlin callable references (scope layer)', () => { + it('captures a bare `::topLevelFn` reference as a free call', () => { + const refs = callReferences(CALLABLE_REFS); + const free = refs.find((r) => r.name === 'topLevelFn'); + expect(free).toBeDefined(); + expect(free!.form).toBe('free'); + }); + + it('captures `String::length` as a member call with receiver String', () => { + const refs = callReferences(CALLABLE_REFS); + const ref = refs.find((r) => r.name === 'length'); + expect(ref).toBeDefined(); + expect(ref!.form).toBe('member'); + expect(ref!.receiver).toBe('String'); + }); + + it('captures `obj::method` as a member call with receiver obj', () => { + const refs = callReferences(CALLABLE_REFS); + const ref = refs.find((r) => r.name === 'method'); + expect(ref).toBeDefined(); + expect(ref!.form).toBe('member'); + expect(ref!.receiver).toBe('obj'); + }); + + it('captures `Type::new` (constructor reference) as a member call with receiver Type', () => { + const refs = callReferences(CALLABLE_REFS); + const ref = refs.find((r) => r.name === 'new'); + expect(ref).toBeDefined(); + expect(ref!.form).toBe('member'); + expect(ref!.receiver).toBe('Type'); + }); + + it('emits exactly one call reference per callable_reference (no double-match)', () => { + const refs = callReferences(CALLABLE_REFS).filter((r) => + ['topLevelFn', 'length', 'method', 'new'].includes(r.name), + ); + expect(refs.map((r) => r.name).sort()).toEqual(['length', 'method', 'new', 'topLevelFn']); + }); + + it('does not capture the receiver type as its own free call', () => { + const refs = callReferences(CALLABLE_REFS); + // String / Type / obj are receivers, never standalone call targets. + expect(refs.some((r) => r.form === 'free' && r.name === 'String')).toBe(false); + expect(refs.some((r) => r.form === 'free' && r.name === 'Type')).toBe(false); + }); +}); + +// --------------------------------------------------------------------------- +// F47 — callable references resolve to CALLS edges end-to-end (worker path) +// --------------------------------------------------------------------------- + +describe('F47 — Kotlin callable references (end-to-end)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'kotlin-coverage'), () => {}); + }, 60000); + + it('resolves a bare `::topLevelFn` reference to a CALLS edge on the local function', () => { + const calls = getRelationships(result, 'CALLS'); + const ref = calls.find((c) => c.source === 'useCallableRefs' && c.target === 'topLevelFn'); + expect(ref).toBeDefined(); + }); + + it('resolves an `obj::method` member reference to a CALLS edge on Obj.method', () => { + const calls = getRelationships(result, 'CALLS'); + const ref = calls.find((c) => c.source === 'useCallableRefs' && c.target === 'method'); + expect(ref).toBeDefined(); + }); + + it('runs through the worker pool (parity: capture edits survive the worker boundary)', () => { + expect(result.usedWorkerPool).toBe(true); + }); +}); + +// --------------------------------------------------------------------------- +// F49 — infix-call query captures only the operator (legacy structure bank) +// --------------------------------------------------------------------------- +// +// Characterized end-to-end (issue #1919): the live spurious @call.name edges +// for `a to b` originate from the legacy KOTLIN_QUERIES bank (still wired as +// provider.treeSitterQueries / used by the worker structure phase). The +// registry KOTLIN_SCOPE_QUERY has no infix rule, so the fix is in +// tree-sitter-queries.ts only. These tests compile that live query and assert +// the call captures directly. + +/** @call.name capture texts produced by the live KOTLIN_QUERIES structure bank. */ +function structureCallNames(src: string): string[] { + const parser = new Parser(); + parser.setLanguage(Kotlin as Parameters[0]); + const query = new Parser.Query(Kotlin as Parameters[0], KOTLIN_QUERIES); + const tree = parser.parse(src); + const names: string[] = []; + for (const match of query.matches(tree.rootNode)) { + for (const c of match.captures) { + if (c.name === 'call.name') names.push(c.node.text); + } + } + return names; +} + +describe('F49 — Kotlin infix call captures only the operator', () => { + it('`val p = a to b` captures exactly one call (`to`), zero for the operands', () => { + const names = structureCallNames(`fun f() {\n val p = a to b\n}`); + expect(names).toEqual(['to']); + }); + + it('`a to b to c` captures only the `to` operators, never the operands', () => { + const names = structureCallNames(`fun f() {\n val q = a to b to c\n}`); + expect(names.sort()).toEqual(['to', 'to']); + expect(names.includes('a')).toBe(false); + expect(names.includes('b')).toBe(false); + expect(names.includes('c')).toBe(false); + }); + + it('a normal call `foo(a, b)` still produces exactly one call to `foo`', () => { + const names = structureCallNames(`fun f() {\n foo(a, b)\n}`); + expect(names).toEqual(['foo']); + }); +}); diff --git a/gitnexus/test/integration/resolvers/kotlin.test.ts b/gitnexus/test/integration/resolvers/kotlin.test.ts index 6eec8a342..9211ed605 100644 --- a/gitnexus/test/integration/resolvers/kotlin.test.ts +++ b/gitnexus/test/integration/resolvers/kotlin.test.ts @@ -2672,3 +2672,231 @@ describe('Kotlin isStaticOnly across other receiver cases (#1756 / U3)', () => { expect(createCalls.length).toBe(0); }); }); + +// --------------------------------------------------------------------------- +// F48 (issue #1919): secondary constructors are extracted as members +// --------------------------------------------------------------------------- + +describe('F48 — Kotlin secondary constructors', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'kotlin-secondary-ctor'), () => {}); + }, 60000); + + it('creates a Constructor node for each secondary constructor', () => { + // Point declares two secondary constructors; both surface as Constructors. + const ctors = getNodesByLabel(result, 'Constructor'); + expect(ctors).toEqual(['constructor', 'constructor']); + }); + + it('owns both secondary constructors under the enclosing class Point', () => { + const owned = getRelationships(result, 'HAS_METHOD').filter( + (e) => e.targetLabel === 'Constructor', + ); + expect(owned.length).toBe(2); + expect(owned.every((e) => e.source === 'Point')).toBe(true); + }); + + it('does not synthesize a constructor for a class with only a primary constructor (no double-count)', () => { + // OnlyPrimary has a primary ctor + one method, and must yield no Constructor node. + const ctorOwners = getRelationships(result, 'HAS_METHOD') + .filter((e) => e.targetLabel === 'Constructor') + .map((e) => e.source); + expect(ctorOwners).not.toContain('OnlyPrimary'); + // Its regular method is still extracted. + expect(getNodesByLabel(result, 'Method')).toContain('method'); + }); + + // ── CF1 (#1919 review): secondary-ctor body calls attribute to the Constructor ── + // The fixture's two secondary constructors call free functions in their bodies: + // constructor(a: Int, b: String) : this(a) { helper() } // arity 2 + // constructor() : this(0) { helper(); other() } // arity 0 + // Each body call must source from ITS OWN Constructor node (with the correct + // arity suffix), NOT from the File node and NOT from the enclosing Class. + it('attributes a secondary-constructor body call to the Constructor node, not File or Class', () => { + const helperCalls = getRelationships(result, 'CALLS').filter((e) => e.target === 'helper'); + // helper() is called from both secondary constructors. + expect(helperCalls.length).toBeGreaterThanOrEqual(2); + for (const call of helperCalls) { + expect(call.sourceLabel).toBe('Constructor'); + expect(call.sourceLabel).not.toBe('File'); + expect(call.sourceLabel).not.toBe('Class'); + } + }); + + it('disambiguates secondary-ctor body calls by arity (# Constructor node id)', () => { + const calls = getRelationships(result, 'CALLS'); + // `other()` is only called from the zero-arg `constructor()` body → must + // source from the arity-0 Constructor node id, never the arity-2 one. + const otherCall = calls.find((e) => e.target === 'other'); + expect(otherCall).toBeDefined(); + expect(otherCall!.sourceLabel).toBe('Constructor'); + expect(otherCall!.rel.sourceId).toBe('Constructor:Constructors.kt:Point.constructor#0'); + + // `helper()` is called from BOTH constructors; the set of caller ids must be + // exactly the two distinct arity-tagged Constructor nodes (no collapse onto one). + const helperSourceIds = new Set( + calls.filter((e) => e.target === 'helper').map((e) => e.rel.sourceId), + ); + expect(helperSourceIds).toEqual( + new Set([ + 'Constructor:Constructors.kt:Point.constructor#0', + 'Constructor:Constructors.kt:Point.constructor#2', + ]), + ); + }); + + it('still attributes a normal method body call to the Method (regression guard)', () => { + // `describe()` is an expression-body method with no call; add a sibling check + // that no secondary-ctor regression mis-routes method-owned calls. The Method + // node for `describe` exists and is owned by Point. + const describeOwned = getRelationships(result, 'HAS_METHOD').filter( + (e) => e.target === 'describe' && e.source === 'Point', + ); + expect(describeOwned.length).toBe(1); + }); +}); + +// --------------------------------------------------------------------------- +// F51 (issue #1919): destructuring declarations emit one binding per name +// --------------------------------------------------------------------------- + +describe('F51 — Kotlin destructuring declarations', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'kotlin-destructuring'), () => {}); + }, 60000); + + it('emits one binding per destructured name in `val (a, b) = pair`', () => { + const props = getNodesByLabel(result, 'Property'); + expect(props).toContain('a'); + expect(props).toContain('b'); + }); + + it('emits bindings for loop destructuring `for ((k, v) in map)`', () => { + const props = getNodesByLabel(result, 'Property'); + expect(props).toContain('k'); + expect(props).toContain('v'); + }); + + it('skips the `_` discard placeholder but keeps `second`', () => { + const props = getNodesByLabel(result, 'Property'); + expect(props).toContain('second'); + expect(props).not.toContain('_'); + }); + + it('emits exactly the expected binding set (no double-count, plain `val x` once)', () => { + const props = getNodesByLabel(result, 'Property'); + expect(props).toEqual(['a', 'b', 'k', 'second', 'v', 'x']); + }); +}); + +// --------------------------------------------------------------------------- +// CF3 (#1919 review): function-local property bindings are NOT class members +// --------------------------------------------------------------------------- +// Kotlin emits destructuring / loop bindings as `@definition.property` to dodge +// the block-scope local-symbol pruner. When such a binding sits inside a METHOD +// body of a class, it must NOT receive a HAS_PROPERTY owner edge from the class — +// it is a function-local, not a class field. Genuine class fields (primary-ctor +// `val` params and class-body `val`/`var`) must still be owned by the class. + +describe('CF3 — Kotlin function-local bindings are not class properties', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo( + path.join(FIXTURES, 'kotlin-local-property-owner'), + () => {}, + ); + }, 60000); + + it('does NOT own loop-destructuring bindings (k, v) under the enclosing class C', () => { + const owned = getRelationships(result, 'HAS_PROPERTY') + .filter((e) => e.source === 'C') + .map((e) => e.target); + expect(owned).not.toContain('k'); + expect(owned).not.toContain('v'); + }); + + it('does NOT own a `val (a, b) = pair` destructuring binding under class C', () => { + const owned = getRelationships(result, 'HAS_PROPERTY') + .filter((e) => e.source === 'C') + .map((e) => e.target); + expect(owned).not.toContain('a'); + expect(owned).not.toContain('b'); + // The intermediate `val pair` local is likewise not a class property. + expect(owned).not.toContain('pair'); + }); + + it('does NOT own destructuring inside an init {} block (ix, iy) under class C', () => { + const owned = getRelationships(result, 'HAS_PROPERTY') + .filter((e) => e.source === 'C') + .map((e) => e.target); + expect(owned).not.toContain('ix'); + expect(owned).not.toContain('iy'); + }); + + it('does NOT own destructuring inside a property accessor body (gx, gy) under class C', () => { + const owned = getRelationships(result, 'HAS_PROPERTY') + .filter((e) => e.source === 'C') + .map((e) => e.target); + expect(owned).not.toContain('gx'); + expect(owned).not.toContain('gy'); + }); + + it('still owns genuine class fields + the computed property under C, nothing else', () => { + const owned = getRelationships(result, 'HAS_PROPERTY') + .filter((e) => e.source === 'C') + .map((e) => e.target) + .sort(); + // Exact set: catches both over-strip (a real member dropped) and under-strip + // (a function-local wrongly owned). + expect(owned).toEqual(['classProp', 'derived', 'field']); + }); +}); + +// --------------------------------------------------------------------------- +// F52 (issue #1919): companion-object properties are indexed as fields +// --------------------------------------------------------------------------- + +describe('F52 — Kotlin companion-object properties', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'kotlin-companion-fields'), () => {}); + }, 60000); + + it('indexes anonymous-companion `const val TAG` as a static, readonly field', () => { + const tag = getNodesByLabelFull(result, 'Property').find((n) => n.name === 'TAG'); + expect(tag).toBeDefined(); + expect(tag!.properties.isStatic).toBe(true); + expect(tag!.properties.isReadonly).toBe(true); + }); + + it('indexes a NAMED-companion property `cfgX` as a field', () => { + const x = getNodesByLabelFull(result, 'Property').find((n) => n.name === 'cfgX'); + expect(x).toBeDefined(); + expect(x!.properties.isStatic).toBe(true); + }); + + it('emits each companion field exactly once (no double emission)', () => { + // Exact field set + a one-per-name count guards against the companion-scope + // machinery re-emitting the same property. + const props = getNodesByLabel(result, 'Property'); + expect(props).toEqual(['TAG', 'cfgX', 'instances']); + expect(props.filter((p) => p === 'TAG')).toHaveLength(1); + }); + + it('owns anonymous-companion fields on the ENCLOSING class C (companion function is not a field)', () => { + const owned = getRelationships(result, 'HAS_PROPERTY'); + const cFields = owned + .filter((e) => e.source === 'C') + .map((e) => e.target) + .sort(); + expect(cFields).toEqual(['TAG', 'instances']); + // The companion's `create` function is a Method, never a Property/field. + expect(getNodesByLabel(result, 'Property')).not.toContain('create'); + }); +}); diff --git a/gitnexus/test/integration/resolvers/swift.test.ts b/gitnexus/test/integration/resolvers/swift.test.ts index 52179d9e5..9b80d92a2 100644 --- a/gitnexus/test/integration/resolvers/swift.test.ts +++ b/gitnexus/test/integration/resolvers/swift.test.ts @@ -1250,3 +1250,110 @@ describe.skipIf(!swiftAvailable)('Swift nested-type extension (extension Foo.Bar expect(baseCall!.rel.targetId).toBe('Function:Types.swift:Bar.base#0'); }); }); + +// --------------------------------------------------------------------------- +// F75: protocol property requirements (`var title: String { get }`) are +// extracted as Property symbols owned by the protocol. Before the fix these +// protocol_property_declaration nodes were dropped (the structure query and +// field config only knew property_declaration). +// --------------------------------------------------------------------------- + +describe.skipIf(!swiftAvailable)('Swift protocol property requirements (F75)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'swift-protocol-property'), () => {}, { + skipGraphPhases: true, + }); + }, 60000); + + it('detects the Repository protocol and its property requirements', () => { + expect(getNodesByLabel(result, 'Interface')).toContain('Repository'); + const properties = getNodesByLabel(result, 'Property'); + expect(properties).toContain('title'); + expect(properties).toContain('count'); + expect(properties).toContain('shared'); + }); + + it('emits HAS_PROPERTY edges from the protocol to each requirement', () => { + const propEdges = getRelationships(result, 'HAS_PROPERTY'); + expect(edgeSet(propEdges)).toEqual( + expect.arrayContaining(['Repository → title', 'Repository → count', 'Repository → shared']), + ); + }); + + it('populates type + static metadata on protocol requirement Property nodes', () => { + const properties = getNodesByLabelFull(result, 'Property'); + + const title = properties.find( + (p) => p.name === 'title' && p.properties.filePath === 'Repository.swift', + ); + expect(title).toBeDefined(); + expect(title!.properties.declaredType).toBe('String'); + expect(title!.properties.isStatic).toBe(false); + + const count = properties.find( + (p) => p.name === 'count' && p.properties.filePath === 'Repository.swift', + ); + expect(count).toBeDefined(); + expect(count!.properties.declaredType).toBe('Int'); + + const shared = properties.find( + (p) => p.name === 'shared' && p.properties.filePath === 'Repository.swift', + ); + expect(shared).toBeDefined(); + expect(shared!.properties.isStatic).toBe(true); + }); + + it('still extracts the class stored property exactly once (regression)', () => { + const propEdges = getRelationships(result, 'HAS_PROPERTY'); + const nameEdges = propEdges.filter((e) => e.target === 'name' && e.source === 'FileRepository'); + expect(nameEdges).toHaveLength(1); + }); +}); + +// --------------------------------------------------------------------------- +// F79: methods/members declared inside a Swift enum (enum_class_body) are +// extracted via the proper body-node path. Before the fix they only resolved +// through the generic findBodies fallback, which logs a dev-mode warning. +// --------------------------------------------------------------------------- + +describe.skipIf(!swiftAvailable)('Swift enum members (F79)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'swift-enum-members'), () => {}, { + skipGraphPhases: true, + }); + }, 60000); + + it('extracts enum methods owned by the enum', () => { + const hasMethod = getRelationships(result, 'HAS_METHOD'); + const enumMethods = hasMethod + .filter((e) => e.source === 'Direction') + .map((e) => e.target) + .sort(); + expect(enumMethods).toContain('describe'); + expect(enumMethods).toContain('make'); + }); + + it('extracts each enum method exactly once (no double-count)', () => { + const hasMethod = getRelationships(result, 'HAS_METHOD'); + const describeEdges = hasMethod.filter( + (e) => e.target === 'describe' && e.source === 'Direction', + ); + expect(describeEdges).toHaveLength(1); + }); + + it('extracts an enum computed property as a Property of the enum', () => { + const propEdges = getRelationships(result, 'HAS_PROPERTY'); + const labelEdge = propEdges.find((e) => e.target === 'label' && e.source === 'Direction'); + expect(labelEdge).toBeDefined(); + }); + + it('still extracts class methods (no regression / double-count)', () => { + const hasMethod = getRelationships(result, 'HAS_METHOD'); + const headingEdges = hasMethod.filter((e) => e.target === 'heading' && e.source === 'Compass'); + expect(headingEdges).toHaveLength(1); + }); +}); diff --git a/gitnexus/test/unit/field-extraction.test.ts b/gitnexus/test/unit/field-extraction.test.ts index 3032eeded..db6a7b0ae 100644 --- a/gitnexus/test/unit/field-extraction.test.ts +++ b/gitnexus/test/unit/field-extraction.test.ts @@ -6,6 +6,9 @@ import { pythonConfig } from '../../src/core/ingestion/field-extractors/configs/ import { goConfig } from '../../src/core/ingestion/field-extractors/configs/go.js'; import { cppConfig } from '../../src/core/ingestion/field-extractors/configs/c-cpp.js'; import { rubyConfig } from '../../src/core/ingestion/field-extractors/configs/ruby.js'; +import { dartConfig } from '../../src/core/ingestion/field-extractors/configs/dart.js'; +import { kotlinConfig } from '../../src/core/ingestion/field-extractors/configs/jvm.js'; +import { swiftConfig } from '../../src/core/ingestion/field-extractors/configs/swift.js'; import type { FieldExtractorContext } from '../../src/core/ingestion/field-types.js'; import type { TypeEnvironment } from '../../src/core/ingestion/type-env.js'; import { createSemanticModel } from '../../src/core/ingestion/model/semantic-model.js'; @@ -16,6 +19,21 @@ import Go from 'tree-sitter-go'; import Cpp from 'tree-sitter-cpp'; import Ruby from 'tree-sitter-ruby'; import CSharp from 'tree-sitter-c-sharp'; +import Dart from 'tree-sitter-dart'; + +let Kotlin: unknown; +try { + Kotlin = require('tree-sitter-kotlin'); +} catch { + // Kotlin grammar may not be installed +} + +let Swift: unknown; +try { + Swift = require('tree-sitter-swift'); +} catch { + // Swift grammar is an optional dependency; may not be installed +} import { csharpConfig as csharpFieldConfig } from '../../src/core/ingestion/field-extractors/configs/csharp.js'; import { SupportedLanguages } from '../../src/config/supported-languages.js'; @@ -1073,3 +1091,279 @@ describe('GenericFieldExtractor — C# primary constructor fields', () => { expect(countField).toBeDefined(); }); }); + +// --------------------------------------------------------------------------- +// Dart config — F26: static const / static final class fields +// --------------------------------------------------------------------------- + +describe('GenericFieldExtractor — Dart', () => { + const parser = new Parser(); + const extractor = createFieldExtractor(dartConfig); + const mockContext = createMockContext(); + mockContext.language = SupportedLanguages.Dart; + mockContext.filePath = 'test.dart'; + + it('extracts a static const field as static + readonly (F26)', () => { + parser.setLanguage(Dart); + const tree = parser.parse(`class C { + static const a = 1; +}`); + const classNode = tree.rootNode.child(0); + expect(classNode).toBeDefined(); + expect(extractor.isTypeDeclaration(classNode!)).toBe(true); + + const result = extractor.extract(classNode!, mockContext); + expect(result).not.toBeNull(); + expect(result!.ownerFqn).toBe('C'); + const a = result!.fields.find((f) => f.name === 'a'); + expect(a).toBeDefined(); + expect(a!.isStatic).toBe(true); + expect(a!.isReadonly).toBe(true); + expect(a!.visibility).toBe('public'); + }); + + it('extracts multi-name static final fields, all static + readonly (F26)', () => { + parser.setLanguage(Dart); + const tree = parser.parse(`class C { + static final String b = 'x', c = 'y'; +}`); + const classNode = tree.rootNode.child(0); + const result = extractor.extract(classNode!, mockContext); + + expect(result).not.toBeNull(); + // Exact-count guard (#1919 review CF4): `find()` below passes even on a + // double-emit, so assert b/c surface exactly twice total — one field each, + // no duplicate from the static_final_declaration_list multi-name path. + expect(result!.fields.filter((f) => f.name === 'b' || f.name === 'c')).toHaveLength(2); + const b = result!.fields.find((f) => f.name === 'b'); + const c = result!.fields.find((f) => f.name === 'c'); + expect(b).toBeDefined(); + expect(c).toBeDefined(); + for (const f of [b!, c!]) { + expect(f.isStatic).toBe(true); + expect(f.isReadonly).toBe(true); + expect(f.type).toBe('String'); + } + }); + + it('still extracts instance fields (regression)', () => { + parser.setLanguage(Dart); + const tree = parser.parse(`class C { + int z = 0; +}`); + const classNode = tree.rootNode.child(0); + const result = extractor.extract(classNode!, mockContext); + + expect(result).not.toBeNull(); + const z = result!.fields.find((f) => f.name === 'z'); + expect(z).toBeDefined(); + expect(z!.isStatic).toBe(false); + expect(z!.isReadonly).toBe(false); + expect(z!.type).toBe('int'); + }); + + it('marks an underscore-prefixed static const as private (F26)', () => { + parser.setLanguage(Dart); + const tree = parser.parse(`class C { + static const _p = 1; +}`); + const classNode = tree.rootNode.child(0); + const result = extractor.extract(classNode!, mockContext); + + expect(result).not.toBeNull(); + const p = result!.fields.find((f) => f.name === '_p'); + expect(p).toBeDefined(); + expect(p!.visibility).toBe('private'); + expect(p!.isStatic).toBe(true); + expect(p!.isReadonly).toBe(true); + }); +}); + +// --------------------------------------------------------------------------- +// Kotlin config — F52: companion-object properties indexed as fields +// --------------------------------------------------------------------------- + +const describeKotlin = Kotlin ? describe : describe.skip; + +describeKotlin('GenericFieldExtractor — Kotlin (F52 companion)', () => { + const parser = new Parser(); + const extractor = createFieldExtractor(kotlinConfig); + const mockContext = createMockContext(); + mockContext.language = SupportedLanguages.Kotlin; + mockContext.filePath = 'test.kt'; + + /** The first companion_object node in `src`. */ + function companion(src: string): Parser.SyntaxNode { + parser.setLanguage(Kotlin as Parser.Language); + const tree = parser.parse(src); + let found: Parser.SyntaxNode | undefined; + const walk = (n: Parser.SyntaxNode) => { + if (n.type === 'companion_object') found ??= n; + for (let i = 0; i < n.namedChildCount; i++) { + const c = n.namedChild(i); + if (c) walk(c); + } + }; + walk(tree.rootNode); + if (!found) throw new Error('no companion_object found'); + return found; + } + + it('extracts a `const val` companion property as a static, readonly field', () => { + const node = companion(`class C { + companion object { + const val TAG = "c" + } +}`); + expect(extractor.isTypeDeclaration(node)).toBe(true); + const result = extractor.extract(node, mockContext); + expect(result).not.toBeNull(); + const tag = result!.fields.find((f) => f.name === 'TAG'); + expect(tag).toBeDefined(); + expect(tag!.isStatic).toBe(true); + expect(tag!.isReadonly).toBe(true); + }); + + it('extracts a property from a NAMED companion object', () => { + const node = companion(`class C { + companion object Factory { + val x = 1 + } +}`); + const result = extractor.extract(node, mockContext); + const x = result!.fields.find((f) => f.name === 'x'); + expect(x).toBeDefined(); + expect(x!.isStatic).toBe(true); + expect(x!.isReadonly).toBe(true); + }); + + it('indexes only the property, not the function, and emits it exactly once', () => { + const node = companion(`class C { + companion object { + val onlyField = 1 + fun create() {} + } +}`); + const result = extractor.extract(node, mockContext); + const fieldNames = result!.fields.map((f) => f.name); + expect(fieldNames).toEqual(['onlyField']); // function excluded, no duplication + }); + + // CF4 (#1919 review): guard the new `isInsideKotlinCompanion` walk against + // false-positives — a plain (non-companion) class property must be isStatic=false. + it('reports a plain non-companion class property as isStatic=false (CF4)', () => { + const classNode = firstNodeOfType( + `class C { + val x: Int = 1 +}`, + 'class_declaration', + ); + expect(extractor.isTypeDeclaration(classNode)).toBe(true); + const result = extractor.extract(classNode, mockContext); + expect(result).not.toBeNull(); + const x = result!.fields.find((f) => f.name === 'x'); + expect(x).toBeDefined(); + expect(x!.isStatic).toBe(false); + }); + + /** Parse `src` and return the first node of the given type (depth-first). */ + function firstNodeOfType(src: string, type: string): Parser.SyntaxNode { + parser.setLanguage(Kotlin as Parser.Language); + const tree = parser.parse(src); + let found: Parser.SyntaxNode | undefined; + const walk = (n: Parser.SyntaxNode) => { + if (found) return; + if (n.type === type) { + found = n; + return; + } + for (let i = 0; i < n.namedChildCount; i++) { + const c = n.namedChild(i); + if (c) walk(c); + } + }; + walk(tree.rootNode); + if (!found) throw new Error(`no ${type} found`); + return found; + } +}); + +// --------------------------------------------------------------------------- +// Swift config — F75: protocol property requirements extracted as fields +// --------------------------------------------------------------------------- + +const describeSwift = Swift ? describe : describe.skip; + +describeSwift('GenericFieldExtractor — Swift (F75 protocol property requirements)', () => { + const parser = new Parser(); + const extractor = createFieldExtractor(swiftConfig); + const mockContext = createMockContext(); + mockContext.language = SupportedLanguages.Swift; + mockContext.filePath = 'test.swift'; + + /** Parse `src` and return the first class/protocol declaration node. */ + function declNode(src: string): Parser.SyntaxNode { + parser.setLanguage(Swift as Parser.Language); + const tree = parser.parse(src); + const node = tree.rootNode.child(0); + if (!node) throw new Error('no declaration node'); + return node; + } + + it('extracts a `{ get }` protocol property requirement as a field (F75)', () => { + const node = declNode(`protocol P { + var title: String { get } +}`); + expect(extractor.isTypeDeclaration(node)).toBe(true); + const result = extractor.extract(node, mockContext); + expect(result).not.toBeNull(); + expect(result!.ownerFqn).toBe('P'); + const title = result!.fields.find((f) => f.name === 'title'); + expect(title).toBeDefined(); + expect(title!.type).toBe('String'); + expect(title!.isStatic).toBe(false); + }); + + it('extracts a `{ get set }` protocol property requirement (F75)', () => { + const node = declNode(`protocol P { + var count: Int { get set } +}`); + const result = extractor.extract(node, mockContext); + const count = result!.fields.find((f) => f.name === 'count'); + expect(count).toBeDefined(); + expect(count!.type).toBe('Int'); + }); + + it('extracts a static protocol property requirement as static (F75)', () => { + const node = declNode(`protocol P { + static var shared: P { get } +}`); + const result = extractor.extract(node, mockContext); + const shared = result!.fields.find((f) => f.name === 'shared'); + expect(shared).toBeDefined(); + expect(shared!.type).toBe('P'); + expect(shared!.isStatic).toBe(true); + }); + + it('extracts all requirements from a multi-property protocol (F75)', () => { + const node = declNode(`protocol P { + var title: String { get } + var count: Int { get set } + static var shared: P { get } +}`); + const result = extractor.extract(node, mockContext); + const names = result!.fields.map((f) => f.name).sort(); + expect(names).toEqual(['count', 'shared', 'title']); + }); + + it('still extracts a class stored property exactly once (regression)', () => { + const node = declNode(`class C { + var name: String = "" +}`); + expect(extractor.isTypeDeclaration(node)).toBe(true); + const result = extractor.extract(node, mockContext); + const matches = result!.fields.filter((f) => f.name === 'name'); + expect(matches).toHaveLength(1); + expect(matches[0].type).toBe('String'); + }); +}); diff --git a/gitnexus/test/unit/method-extraction.test.ts b/gitnexus/test/unit/method-extraction.test.ts index 452ebc0bd..eea3b887e 100644 --- a/gitnexus/test/unit/method-extraction.test.ts +++ b/gitnexus/test/unit/method-extraction.test.ts @@ -727,6 +727,69 @@ describeKotlin('Kotlin MethodExtractor', () => { expect(result!.methods[0].isStatic).toBe(true); }); }); + + // F48 (issue #1919): secondary constructors were dropped — methodNodeTypes + // listed only 'function_declaration'. They are now extracted as members + // named "constructor" with their function_value_parameters. + describe('secondary constructors (F48)', () => { + it('extracts a secondary constructor as a member named "constructor" with its params', () => { + const tree = parseKotlin(` + class C(val x: Int) { + constructor(a: Int, b: String) : this(a) { } + } + `); + const classNode = tree.rootNode.child(0)!; + const result = extractor.extract(classNode, kotlinCtx); + + const ctor = result!.methods.find((m) => m.name === 'constructor'); + expect(ctor).toBeDefined(); + expect(ctor!.parameters.map((p) => p.name)).toEqual(['a', 'b']); + expect(ctor!.parameters[0].type).toBe('Int'); + }); + + it('extracts multiple secondary constructors distinctly (by arity)', () => { + const tree = parseKotlin(` + class C(val x: Int) { + constructor(a: Int, b: String) : this(a) { } + constructor() { } + } + `); + const classNode = tree.rootNode.child(0)!; + const result = extractor.extract(classNode, kotlinCtx); + + const ctors = result!.methods.filter((m) => m.name === 'constructor'); + expect(ctors).toHaveLength(2); + const arities = ctors.map((c) => c.parameters.length).sort(); + expect(arities).toEqual([0, 2]); + }); + + it('still extracts the secondary constructor when it delegates via : this(...)', () => { + const tree = parseKotlin(` + class C(val x: Int) { + constructor(a: Int) : this(a) { } + } + `); + const classNode = tree.rootNode.child(0)!; + const result = extractor.extract(classNode, kotlinCtx); + + const ctor = result!.methods.find((m) => m.name === 'constructor'); + expect(ctor).toBeDefined(); + expect(ctor!.parameters.map((p) => p.name)).toEqual(['a']); + }); + + it('does not synthesize a constructor member for a class with only a primary constructor + methods', () => { + const tree = parseKotlin(` + class C(val x: Int) { + fun normal(): Int = x + } + `); + const classNode = tree.rootNode.child(0)!; + const result = extractor.extract(classNode, kotlinCtx); + + expect(result!.methods.some((m) => m.name === 'constructor')).toBe(false); + expect(result!.methods.map((m) => m.name)).toEqual(['normal']); + }); + }); }); // --------------------------------------------------------------------------- @@ -4535,6 +4598,74 @@ class Child { expect(result!.methods[0].isOverride).toBe(true); }); }); + + // F79: a Swift `enum { ... }` parses to a class_declaration whose body is an + // `enum_class_body` (NOT class_body). With enum_class_body added to + // bodyNodeTypes the factory reaches enum methods via the proper body-node + // path instead of the generic findBodies fallback. + describe('enum members (F79)', () => { + it('extracts a method declared inside an enum', () => { + const tree = parseSwift(` +enum E { + case a + func describe() -> String { + return "x" + } +} + `); + const enumNode = tree.rootNode.child(0)!; + expect(enumNode.type).toBe('class_declaration'); + expect(extractor.isTypeDeclaration(enumNode)).toBe(true); + + const result = extractor.extract(enumNode, swiftCtx); + expect(result!.ownerName).toBe('E'); + const describe = result!.methods.find((m) => m.name === 'describe'); + expect(describe).toBeDefined(); + expect(describe!.returnType).toBe('String'); + }); + + it('extracts a static method inside an enum as static', () => { + const tree = parseSwift(` +enum E { + case a + static func make() -> E { + return .a + } +} + `); + const enumNode = tree.rootNode.child(0)!; + const result = extractor.extract(enumNode, swiftCtx); + const make = result!.methods.find((m) => m.name === 'make'); + expect(make).toBeDefined(); + expect(make!.isStatic).toBe(true); + }); + + it('extracts multiple enum methods, each exactly once', () => { + const tree = parseSwift(` +enum E { + case a + func describe() -> String { return "x" } + static func make() -> E { return .a } +} + `); + const enumNode = tree.rootNode.child(0)!; + const result = extractor.extract(enumNode, swiftCtx); + const names = result!.methods.map((m) => m.name).sort(); + expect(names).toEqual(['describe', 'make']); + }); + + it('still extracts class methods exactly once (regression)', () => { + const tree = parseSwift(` +class Compass { + func heading() -> String { return "n" } +} + `); + const classNode = tree.rootNode.child(0)!; + const result = extractor.extract(classNode, swiftCtx); + const heading = result!.methods.filter((m) => m.name === 'heading'); + expect(heading).toHaveLength(1); + }); + }); }); // --------------------------------------------------------------------------- diff --git a/gitnexus/test/unit/variable-extraction.test.ts b/gitnexus/test/unit/variable-extraction.test.ts index e7a8be1ce..4c39f0165 100644 --- a/gitnexus/test/unit/variable-extraction.test.ts +++ b/gitnexus/test/unit/variable-extraction.test.ts @@ -12,6 +12,9 @@ import { cppVariableConfig, } from '../../src/core/ingestion/variable-extractors/configs/c-cpp.js'; import { rubyVariableConfig } from '../../src/core/ingestion/variable-extractors/configs/ruby.js'; +import { dartVariableConfig } from '../../src/core/ingestion/variable-extractors/configs/dart.js'; +import { kotlinVariableConfig } from '../../src/core/ingestion/variable-extractors/configs/jvm.js'; +import type { SyntaxNode } from '../../src/core/ingestion/utils/ast-helpers.js'; import type { VariableExtractorContext } from '../../src/core/ingestion/variable-types.js'; import { SupportedLanguages } from '../../src/config/supported-languages.js'; import Parser from 'tree-sitter'; @@ -22,6 +25,14 @@ import Rust from 'tree-sitter-rust'; import Cpp from 'tree-sitter-cpp'; import C from 'tree-sitter-c'; import Ruby from 'tree-sitter-ruby'; +import Dart from 'tree-sitter-dart'; + +let Kotlin: unknown; +try { + Kotlin = require('tree-sitter-kotlin'); +} catch { + // Kotlin grammar may not be installed +} const parser = new Parser(); @@ -473,6 +484,41 @@ describe('VariableExtractor — C++', () => { expect(info!.name).toBe('SIZE'); expect(info!.isConst).toBe(true); }); + + // F9 — structured binding declarations emit one Variable per bound name. + it('emits a Variable per name for `auto [a, b] = make_pair();`', () => { + parser.setLanguage(Cpp); + const tree = parser.parse('auto [a, b] = make_pair();'); + const node = tree.rootNode.child(0)!; + expect(extractor.isVariableDeclaration(node)).toBe(true); + + const infos = extractor.extractAll(node, ctx); + expect(infos.map((i) => i.name)).toEqual(['a', 'b']); + // Top-level declaration → module scope (C++ is in the safe set re: the + // determineScope class-body block hazard). + for (const info of infos) expect(info.scope).toBe('module'); + }); + + it('emits a Variable per name for the reference form `auto& [x, y, z] = tup;`', () => { + parser.setLanguage(Cpp); + const tree = parser.parse('auto& [x, y, z] = tup;'); + const node = tree.rootNode.child(0)!; + + const infos = extractor.extractAll(node, ctx); + expect(infos.map((i) => i.name)).toEqual(['x', 'y', 'z']); + for (const info of infos) expect(info.scope).toBe('module'); + }); + + it('does not double-emit for an ordinary single-name declaration `int n = 0;`', () => { + parser.setLanguage(Cpp); + const tree = parser.parse('int n = 0;'); + const node = tree.rootNode.child(0)!; + + const infos = extractor.extractAll(node, ctx); + expect(infos).toHaveLength(1); + expect(infos[0]!.name).toBe('n'); + expect(infos[0]!.scope).toBe('module'); + }); }); // --------------------------------------------------------------------------- @@ -724,3 +770,159 @@ describe('VariableExtractor — block-scoped declarations', () => { expect(info).toBeNull(); }); }); + +// --------------------------------------------------------------------------- +// Dart config — F29: top-level variable declarations +// +// Top-level Dart vars are loose siblings under `program` (no `declaration` +// wrapper). The structure query captures the container node +// (initialized_identifier_list / static_final_declaration_list) as +// @definition.variable, so the extractor is fed that container. The previous +// config read a `type_identifier` that does not exist as a direct child of the +// captured node — a dead read. These tests feed the real captured container. +// --------------------------------------------------------------------------- + +describe('VariableExtractor — Dart (F29 top-level)', () => { + const extractor = createVariableExtractor(dartVariableConfig); + const ctx: VariableExtractorContext = { + filePath: 'test.dart', + language: SupportedLanguages.Dart, + }; + + /** The top-level variable container the structure query captures. */ + function captureContainer(src: string): SyntaxNode { + parser.setLanguage(Dart); + const tree = parser.parse(src); + let found: SyntaxNode | undefined; + const walk = (n: SyntaxNode) => { + if ( + (n.type === 'initialized_identifier_list' || n.type === 'static_final_declaration_list') && + n.parent?.type === 'program' + ) { + found ??= n; + } + for (let i = 0; i < n.namedChildCount; i++) { + const c = n.namedChild(i); + if (c) walk(c); + } + }; + walk(tree.rootNode); + if (!found) throw new Error('no top-level variable container found'); + return found; + } + + it('extracts a typed final top-level variable with the real type (F29)', () => { + const node = captureContainer('final int count = 3;'); + expect(extractor.isVariableDeclaration(node)).toBe(true); + const info = extractor.extract(node, ctx); + expect(info).not.toBeNull(); + expect(info!.name).toBe('count'); + expect(info!.type).toBe('int'); + expect(info!.isConst).toBe(true); + expect(info!.isMutable).toBe(false); + expect(info!.scope).toBe('module'); + }); + + it('extracts an inferred var as untyped (not from a phantom field) (F29)', () => { + const node = captureContainer("var name = 'x';"); + const info = extractor.extract(node, ctx); + expect(info).not.toBeNull(); + expect(info!.name).toBe('name'); + // `var` is inferred → no type annotation, so type is null (not a phantom). + expect(info!.type).toBeNull(); + expect(info!.isMutable).toBe(true); + expect(info!.isConst).toBe(false); + }); + + it('extracts both names from a multi-name top-level const (F29)', () => { + const node = captureContainer('const a = 1, b = 2;'); + const infos = extractor.extractAll(node, ctx); + expect(infos.map((i) => i.name).sort()).toEqual(['a', 'b']); + for (const i of infos) { + expect(i.isConst).toBe(true); + expect(i.isMutable).toBe(false); + } + }); + + it('does not let a neighbouring declaration’s type/modifier bleed in', () => { + // Three declarations under one program. The `var name` container sits after + // `final int count` — its type/modifier must NOT leak onto `name`. + parser.setLanguage(Dart); + const tree = parser.parse("final int count = 3;\nvar name = 'x';\nconst a = 1, b = 2;"); + const containers: SyntaxNode[] = []; + const walk = (n: SyntaxNode) => { + if ( + (n.type === 'initialized_identifier_list' || n.type === 'static_final_declaration_list') && + n.parent?.type === 'program' + ) { + containers.push(n); + } + for (let i = 0; i < n.namedChildCount; i++) { + const c = n.namedChild(i); + if (c) walk(c); + } + }; + walk(tree.rootNode); + + const byName = new Map[number]>(); + for (const c of containers) + for (const info of extractor.extractAll(c, ctx)) byName.set(info.name, info); + + expect(byName.get('name')!.type).toBeNull(); // not 'int' from count + expect(byName.get('name')!.isConst).toBe(false); // not final from count + expect(byName.get('count')!.type).toBe('int'); + expect(byName.get('a')!.isConst).toBe(true); + expect(byName.get('b')!.isConst).toBe(true); + }); +}); + +// --------------------------------------------------------------------------- +// Kotlin destructuring declarations (F51, issue #1919) +// --------------------------------------------------------------------------- + +const describeKotlin = Kotlin ? describe : describe.skip; + +describeKotlin('VariableExtractor — Kotlin (F51 destructuring)', () => { + const extractor = createVariableExtractor(kotlinVariableConfig); + const ctx: VariableExtractorContext = { + filePath: 'test.kt', + language: SupportedLanguages.Kotlin, + }; + + /** The first property_declaration whose text starts with `prefix`. */ + function propertyDecl(src: string, prefix: string): SyntaxNode { + parser.setLanguage(Kotlin as Parser.Language); + const tree = parser.parse(src); + let found: SyntaxNode | undefined; + const walk = (n: SyntaxNode) => { + if (n.type === 'property_declaration' && n.text.trimStart().startsWith(prefix)) { + found ??= n; + } + for (let i = 0; i < n.namedChildCount; i++) { + const c = n.namedChild(i); + if (c) walk(c); + } + }; + walk(tree.rootNode); + if (!found) throw new Error(`no property_declaration starting with ${prefix}`); + return found; + } + + it('emits one Variable per destructured name (`val (a, b) = pair`)', () => { + const node = propertyDecl('fun f() { val (a, b) = pair }', 'val (a'); + const infos = extractor.extractAll(node, ctx); + expect(infos.map((i) => i.name)).toEqual(['a', 'b']); + }); + + it('skips the `_` discard placeholder (`val (_, second) = pair`)', () => { + const node = propertyDecl('fun f() { val (_, second) = pair }', 'val (_'); + const infos = extractor.extractAll(node, ctx); + expect(infos.map((i) => i.name)).toEqual(['second']); + }); + + it('still emits exactly one name for a plain `val x = 1` (no double-count)', () => { + const node = propertyDecl('fun f() { val x = 1 }', 'val x'); + const infos = extractor.extractAll(node, ctx); + expect(infos.map((i) => i.name)).toEqual(['x']); + }); +}); From 1ca4f15267f4cef349925b2d6e492ee75699ba91 Mon Sep 17 00:00:00 2001 From: xianzuyang9-blip Date: Mon, 8 Jun 2026 15:44:05 +0800 Subject: [PATCH 11/17] fix: declare onnxruntime-common runtime dependency (#2074) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Declare onnxruntime-common runtime dependency * Declare onnxruntime-common runtime dependency * Declare onnxruntime-common runtime dependency * Declare onnxruntime-common runtime dependency * Declare onnxruntime-common runtime dependency * Remove package metadata unit test --------- Co-authored-by: Gergő Magyar --- gitnexus/package-lock.json | 1 + gitnexus/package.json | 1 + 2 files changed, 2 insertions(+) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 83534ed77..750979d80 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -27,6 +27,7 @@ "js-yaml": "^4.1.1", "jsonc-parser": "^3.3.1", "mnemonist": "^0.40.3", + "onnxruntime-common": "^1.26.0", "onnxruntime-node": "^1.24.0", "pandemonium": "^2.4.0", "pino": "^10.3.1", diff --git a/gitnexus/package.json b/gitnexus/package.json index 1dbb1ed80..b61a1ca96 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -71,6 +71,7 @@ "js-yaml": "^4.1.1", "jsonc-parser": "^3.3.1", "mnemonist": "^0.40.3", + "onnxruntime-common": "^1.26.0", "onnxruntime-node": "^1.24.0", "pandemonium": "^2.4.0", "pino": "^10.3.1", From 2cf7e7fa88bfa05fff76796fa401f74803977932 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 8 Jun 2026 09:44:28 +0100 Subject: [PATCH 12/17] chore: release v1.6.6 (#2075) Bump gitnexus 1.6.5 -> 1.6.6 and add the 1.6.6 CHANGELOG section covering ~190 PRs merged since v1.6.5 (range v1.6.5..main). Co-authored-by: Claude Opus 4.8 (1M context) --- gitnexus/CHANGELOG.md | 65 ++++++++++++++++++++++++++++++++++++++ gitnexus/package-lock.json | 4 +-- gitnexus/package.json | 2 +- 3 files changed, 68 insertions(+), 3 deletions(-) diff --git a/gitnexus/CHANGELOG.md b/gitnexus/CHANGELOG.md index 39db0eb90..e06601f34 100644 --- a/gitnexus/CHANGELOG.md +++ b/gitnexus/CHANGELOG.md @@ -4,6 +4,71 @@ All notable changes to GitNexus will be documented in this file. ## [Unreleased] +## [1.6.6] - 2026-06-08 + +### Added + +- **Scope-resolution (RFC #909) migrations completed across the language matrix** — Rust (#1639), JavaScript (#1640), Ruby (#1831), Swift (#937, #1948), Vue SFC (#940, #1950), Dart (#939, #1970), COBOL (#941, #1835, #1842), and Kotlin (#1727, #1746, #1782) now run on the registry-primary path; Java reached 100% scope-resolution parity and joined `MIGRATED_LANGUAGES` (#1805); per-language progress reporting added to the scope-resolution phase (#1813) +- **HTTP route & consumer contract extraction (group mode)** — Spring interface routes attributed to controllers (#1743); named/positional Java Spring route args (#1834); Kotlin Spring HTTP route, consumer, and WebClient long-form extraction (#1849, #1855, #1884); Java HTTP consumer contracts (#1872); OpenFeign `@RequestLine` consumer contracts incl. plain interfaces without `@FeignClient` (#1904, #1917); FastAPI `include_router(prefix=...)` cross-file routes (#1877); indirect call patterns via FastAPI `Depends()` and frontend HTTP consumers (#1852); gRPC consumer FQN derivation from Java imports for client-jar consumers (#1889) +- **C++ overload & template resolution** — operator-call resolution (#1754), template partial ordering (#1885), user-defined conversion ranking (#1829), nullptr/ellipsis pointer conversion ranks (#1708), SFINAE filter (#1623), expanded `type_traits` constraint registry (#1648), structured resolver-suppression outcomes (#1785), function-type ADL entities (#1822), and a parameter-type class sidecar (#1642) +- **Go enhancements** — structural interface implementation inference (#1966) and a `builtInNames` set for the Go language provider (#1886) +- **Self-healing worker pool** — automatic worker replacement plus deferred-resolution observability and verbose progress logging (#1741, #1773, #1947) +- **`.gitnexusrc` config file and `gitnexus analyze --default-branch`** (#243, #1996) +- **CLI / MCP impact ergonomics** — `--uid/--file/--kind` disambiguation flags (#1907, #1914), `limit/offset/summaryOnly` pagination on the impact tool (#1818), and a per-symbol `processes` field on `byDepth` items (#1867) +- **`gitnexus analyze --repair-fts`** — enforces FTS verification with hardened repair safeguards (#1720) +- **Web viewer** — Tree View and Circles View (#1799), GitLab repository URLs (#1565), `GITNEXUS_BACKEND_URL` env var for Docker deployments (#1286), and web + CLI internationalization (#1748) +- **Wiki** — local Claude/Codex providers (#1769), an opencode local provider (#2039), and `gitnexus wiki --lang ` for multilanguage wiki generation (#1613) +- **`detect-changes` git-worktree support** (#1654) +- **DeepSeek V4 API support** (#1594) +- **Devcontainer for the Claude / Codex / Cursor CLIs** (#1875) and antigravity integration setup + hook adapter (#1730) +- **Object-literal methods linked to exported bindings** (#1718) +- **`eval-server --host`** for a user-configured bind IP (#1667) +- **PR reviewer swarm agents** (#1851) +- **tree-sitter node-type/field validation gate** — validates against the grammar and removes dead literal handling (#1937) + +### Fixed + +- **Parsing-layer coverage gaps closed across the language matrix** (umbrella #1919) — remaining open gaps (#2072) plus Java F35/F38/F41 (#1928, #2045), PHP F53/F54/F55 (#1931, #1989), COBOL F17–F23 (#1925, #1959), Rust F66/F68/F71/F72 (#1934, #1974), Python F57/F58/F61 (#1932, #1964), JS/TS F44/F83/F85/F86/F87 (#1929, #1968), and Ruby F62 (#1933, #1972) +- **Fully-qualified nested-type identity for C++ and Ruby** — distinct nodes for union-, anonymous-namespace-, and same-tail-nested types (#1978, #1981, #2004, #2005); cross-namespace same-tail inheritance bases resolved (#1993, #2005); Ruby same-tail nested mixin modules qualified with `IMPLEMENTS` routed by scope (#1991, #2006); shared codec for `__heritage__`/`__property__` markers (#1994, #2007); graph nodes materialized for scoped class/module/impl declarations (#1975, #1977); generic Rust inherent-impl methods owned through the mod-qualified `Impl` node (#1992, #2003) +- **C# resolution & memory** — global-namespace `typeBindings` O(files²) OOM eliminated (#1871, #1954) and namespace-siblings OOM with worker-path re-parse removed (#1905); qualified/alias constructor names, `:base`/`:this` initializers, and generic type-arg stripping (#2046); primary-base receiver type normalization (#2036); spurious `IMPORTS` edges from ungated `using` resolution stopped (#1881, #1908) +- **C++ dependent-base and member lookup** — resolution across nested/inline namespaces (#1634, #1814), base-specifier qualifier threading (#1815, #1819), call-site types threaded into qualified member lookup (#1632, #1810), variadic pack dependent lookup (#1909), uninitialized multi-declarators (#1965), and typedef-enum / anonymous-struct declarations (#1941) +- **Kotlin type resolution** — smart-cast refinement for `when/is` and `if/is` (#1758, #1774), overload target-id by parameter types (#1761, #1777), cross-file iterable return propagation (#1759, #1775), method-chain fixpoint receiver types (#1760, #1776), virtual dispatch via constructor type override (#1762, #1778), interface default-method dispatch via implements-split MRO (#1763, #1779), and default-parameter arity detection (#2034) +- **Go declarations** — multi-name declaration capture (#2032), fixed-array parameter binding normalization (#1988), and generic composite-literal constructor inference F33 (#1976) +- **Rust / PHP / Vue / Java parsing** — Rust `struct_expression` name pattern split (#2051); PHP import decomposition, namespace-less `.phtml` module scopes, and Blade-template exclusion (#1801, #1790, #1989); Vue JSDoc, dual-script merge, and lang plumbing F89/F90/F92 (#1936, #2050); Java inherited `RequestMapping` prefix deduplication (#2057) and same-module type resolution for duplicate FQNs (#1712) +- **TypeScript** — HOC pattern false positives fixed with `export default` HOC support (#1943) and suffix-index reuse in the scope resolver (#1840) +- **Inheritance on the worker path** — all languages' inheritance migrated to scope-resolution in worker mode (#1951, #1956); centralized heritage supertype matching (#1921, #1922, #1940); `File->Member` `DEFINES` edges skipped for class members (#1949); phantom `Function` defs for array-method callbacks no longer emitted (#1906) +- **MCP** — sibling-clone repo-ID collisions prevented and generated MCP tool names corrected (#2067); orphan processes avoided by handling stdin close/end and the startup race (#2049); duplicate-name repo resolution disambiguated for worktrees (#1753); Windows setup fallback when global `gitnexus` resolves to a non-spawnable shim (#1694) +- **Worker pool** — resilient zero-copy ingestion worker pool prevents analyze hangs on TS-root-scale loads (#1693); cache-hit native workers no longer abort (#1751, #1833); worker-pool docs drift corrected and worker-side stack surfaced on crash (#2068, #2070) +- **LadybugDB** — FTS loaded in the Windows read pool (#2040) and probed-then-loaded on Windows (#1690, #1692); non-ASCII KuzuDB paths resolved on Windows (#1811, #1817); WAL corruption detected in schema init with recovery surfaced (#1647, #1650); WAL checkpoint-threshold control (#1772); init lock skipped for read-only opens (#1783, #1784); `serve` kept stable when sidecars are missing (#1747) +- **Server / API** — `gitnexus serve` startup restored under Express 5 (#1749); `/api/graph`, `/api/search`, `/api/grep` opened read-only (#1686); native read-only enforcement and prepared statements for Cypher query paths (#1655); `eval-server` localhost binding left to the OS (#1722) +- **Embeddings** — local ONNX runtime guarded on macOS Intel before the transformers.js import (#1987) +- **Web agent** — Nexus AI agent system prompt aligned with registered tools (#1984) and the agent stopped cleanly on user Stop (#1820) +- **Group / contracts** — HTTP graph and source contracts unioned (#1709); `httpx` `AsyncClient` alias imports detected (#1687); Node gRPC `loadPackageDefinition` gate no longer matches every member call (#1916); manifest/workspace extraction moved before `closeLbug` (#1802, #1807) +- **Hooks / install** — `gitnexus` resolved on `PATH` via a pure-Node, all-OS scan (#1938, #1980); offline-first extension installs (#1161); actionable error and docs for the `pnpm dlx`/`pnpx` native-load crash (#307, #1967); `onnxruntime-common` declared as a runtime dependency (#2074); vendored grammars materialized to fix Windows EPERM (#1728, #1729) +- **CLI** — missing LadybugDB native binary detected at startup with actionable guidance (#835, #1837); `--no-stats` applied to the keep-marker stats line (#1706, #1765); skipped large-file paths surfaced by default (#1659, #1661); build.js skipped when running outside the monorepo (#1795, #1816); auto-heap raised to 16 GB with tightened cross-platform OOM guidance for UE5-scale repos (#1652) +- **Wiki** — hidden 60s default timeout removed with timeout/retry flag validation and surfaced timeout errors (#1651); budget-aware grouping to prevent context overflow on large repos (#627, #1832) +- **`detect-changes`** — `resolveWorktreeCwd` guarded against overriding a separately-indexed worktree (#1691) +- **Windows reliability** — `windowsHide:true` passed to every `child_process` spawn-family call (#1794) + +### Changed + +- **Legacy resolution deletion (Ring 4)** — removed the legacy call-resolution DAG + heritage processor (RING4-1, #942, #2023), the legacy resolution-context + tiered-lookup plumbing (RING4-2, #943, #2033), and the shadow-mode parity harness (RING4-3, #944, #2071) +- **CONTRIBUTING** — clarified local development setup (#2024) +- **Tests / CI** — cli-e2e made read-only and eval-server tests hardened under load (#2000, #1786, #1838, #1688); parity shards consolidated and the cross-platform matrix narrowed (#1798); devcontainer smoke build hardened against Docker Hub flakes (#1969); gitleaks stabilized (#2027) + +### Performance + +- **Linux-kernel-scale analysis overhaul** — worker-pool parse, finalize O(n²), and the scope-resolution memory wall (#1983, #2038) +- **Scope-capture linearized across all languages (O(n²)→O(n))** plus Python import-resolution linearization (#1918), the Go-specific re-walk fix (#1848, #1915), and owner-keyed lookup for Step 2 member resolution (#1657) +- **C++ ADL candidates indexed once instead of per-site rescans** (#1990) +- **Inert local value symbols pruned** during ingestion (#2065) + +### Chore / Dependencies + +- `@ladybugdb/core` bump in /gitnexus (#2056) +- Routine dependency bumps across /gitnexus, /gitnexus-web, /eval, and GitHub Actions — incl. `hono`, `vitest`, `@vitest/coverage-v8`, `tsx`, `lru-cache`, `express`/`@types/express`, `express-rate-limit`, `qs`, `node-addon-api`, `brace-expansion`, `langchain`, `i18next`, `dompurify`, `lucide-react`, `axios`, `zod`, `@langchain/langgraph`, `@vercel/node`, `langsmith`, `aiohttp`, `idna`, and the `docker/*` / `github/codeql-action` / `release-drafter` / `dependency-review-action` actions (#2056, #2044, #2043, #2042, #2016, #2015, #2013, #2012, #2011, #2010, #2009, #2008, #2018, #2019, #2017, #2020, #1986, #1911, #1864, #1863, #1861, #1860, #1866, #1844, #1845, #1826, #1825, #1824, #1791, #1789, #1768, #1767, #1739, #1740, #1738, #1736, #1735, #1734, #1731, #1713, #1698, #1697, #1696, #1689, #1604, #1552, #1464, #872) +- **Security** — `@vercel/node` upgraded in /gitnexus-web with transitive advisories remediated (#1705) + ## [1.6.5] - 2026-05-16 ### Added diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 750979d80..c5f4aa7af 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -1,12 +1,12 @@ { "name": "gitnexus", - "version": "1.6.5", + "version": "1.6.6", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "gitnexus", - "version": "1.6.5", + "version": "1.6.6", "hasInstallScript": true, "license": "PolyForm-Noncommercial-1.0.0", "dependencies": { diff --git a/gitnexus/package.json b/gitnexus/package.json index b61a1ca96..cd065cef0 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -1,6 +1,6 @@ { "name": "gitnexus", - "version": "1.6.5", + "version": "1.6.6", "description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.", "author": "Abhigyan Patwari", "license": "PolyForm-Noncommercial-1.0.0", From 689e6ef1f8289e83147e0000eb6d964ce0badf54 Mon Sep 17 00:00:00 2001 From: Copilot <198982749+Copilot@users.noreply.github.com> Date: Mon, 8 Jun 2026 16:50:09 +0100 Subject: [PATCH 13/17] chore: Sync Claude plugin manifests with the 1.6.6 release (#2090) * Initial plan * fix: sync Claude plugin manifest versions * test: fold manifest sync check into existing node suite * chore(autofix): apply prettier + eslint fixes via /autofix command * test: run manifest sync guard in always-on suite --------- Co-authored-by: copilot-swe-agent[bot] <198982749+Copilot@users.noreply.github.com> Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- .claude-plugin/marketplace.json | 2 +- CONTRIBUTING.md | 7 ++++- .../.claude-plugin/plugin.json | 2 +- gitnexus/test/unit/cli-commands.test.ts | 29 +++++++++++++++++++ 4 files changed, 37 insertions(+), 3 deletions(-) diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 719473def..4dc2f3260 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -11,7 +11,7 @@ "plugins": [ { "name": "gitnexus", - "version": "1.3.3", + "version": "1.6.6", "source": "./gitnexus-claude-plugin", "description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase." } diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index ddefad384..848884be4 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -157,7 +157,12 @@ routes between two modes based on the triggering event: suffix; RC tags are excluded at trigger via a negative glob). Publishes to the `latest` dist-tag with a changelog-backed GitHub release. Maintainers are expected to tag from `main` as a convention; the workflow itself does - not enforce branch reachability. No Docker build (RC-only). + not enforce branch reachability. No Docker build (RC-only). Before cutting a + stable release, keep `gitnexus/package.json`, + `gitnexus-claude-plugin/.claude-plugin/plugin.json`, + `.claude-plugin/marketplace.json`, and the matching `CHANGELOG.md` entry in + lockstep — the always-on `gitnexus` unit suite now fails if those manifest + versions drift. - **Release-candidate mode** — runs on every push to `main` (typically a merged PR) plus manual `workflow_dispatch`. Docs-only changes are skipped via `paths-ignore`. Publishes to the `rc` dist-tag with version diff --git a/gitnexus-claude-plugin/.claude-plugin/plugin.json b/gitnexus-claude-plugin/.claude-plugin/plugin.json index bd4b8c426..da6b42a7b 100644 --- a/gitnexus-claude-plugin/.claude-plugin/plugin.json +++ b/gitnexus-claude-plugin/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "gitnexus", "description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase.", - "version": "1.3.6", + "version": "1.6.6", "author": { "name": "GitNexus" }, diff --git a/gitnexus/test/unit/cli-commands.test.ts b/gitnexus/test/unit/cli-commands.test.ts index 26b0b6f66..fcae5d0bd 100644 --- a/gitnexus/test/unit/cli-commands.test.ts +++ b/gitnexus/test/unit/cli-commands.test.ts @@ -1,5 +1,14 @@ +import fs from 'node:fs/promises'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; import { describe, it, expect, vi } from 'vitest'; +const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..', '..'); + +async function readRepoJson(relativePath: string): Promise { + return JSON.parse(await fs.readFile(path.join(REPO_ROOT, relativePath), 'utf8')) as T; +} + // Mock all the heavy imports before importing index vi.mock('../../src/cli/analyze.js', () => ({ analyzeCommand: vi.fn(), @@ -20,6 +29,26 @@ describe('CLI commands', () => { const pkg = await import('../../package.json', { with: { type: 'json' } }); expect(pkg.default.version).toMatch(/^\d+\.\d+\.\d+/); }); + + it('keeps Claude plugin manifests aligned with the gitnexus release version', async () => { + const pkg = await import('../../package.json', { with: { type: 'json' } }); + const pluginManifest = await readRepoJson<{ version: string }>( + 'gitnexus-claude-plugin/.claude-plugin/plugin.json', + ); + const marketplaceManifest = await readRepoJson<{ + plugins?: Array<{ name: string; version: string }>; + }>('.claude-plugin/marketplace.json'); + + expect(Array.isArray(marketplaceManifest.plugins)).toBe(true); + + const gitnexusEntries = (marketplaceManifest.plugins ?? []).filter( + (plugin) => plugin.name === 'gitnexus', + ); + + expect(gitnexusEntries).toHaveLength(1); + expect(pluginManifest.version).toBe(pkg.default.version); + expect(gitnexusEntries[0]?.version).toBe(pkg.default.version); + }); }); describe('package.json scripts', () => { From f2c9e6979223d8b36fd88be802a8c81d628c053b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 8 Jun 2026 18:56:10 +0100 Subject: [PATCH 14/17] =?UTF-8?q?feat(ingestion):=20M0=20=E2=80=94=20taint?= =?UTF-8?q?/PDG=20substrate=20(schema=20+=20seams=20+=20spikes)=20(#2080)?= =?UTF-8?q?=20(#2092)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- gitnexus-shared/src/graph/types.ts | 30 +++- gitnexus-shared/src/lbug/schema-constants.ts | 10 ++ gitnexus-web/src/lib/constants.ts | 2 + gitnexus/CHANGELOG.md | 4 + .../spikes/s1-reaching-def-index-bench.ts | 151 ++++++++++++++++ .../scripts/spikes/s2-postdom-prototype.ts | 162 ++++++++++++++++++ .../ingestion/model/registration-table.ts | 3 + .../core/ingestion/pipeline-phases/index.ts | 1 + .../ingestion/pipeline-phases/registry.ts | 73 ++++++++ gitnexus/src/core/ingestion/pipeline.ts | 51 +++--- .../ingestion/taint/source-sink-config.ts | 38 ++++ .../ingestion/taint/source-sink-registry.ts | 42 +++++ gitnexus/src/core/lbug/csv-generator.ts | 20 +++ gitnexus/src/core/lbug/lbug-adapter.ts | 10 ++ gitnexus/src/core/lbug/schema.ts | 23 +++ gitnexus/src/server/api.ts | 19 +- .../integration/basicblock-roundtrip.test.ts | 154 +++++++++++++++++ .../test/unit/api-graph-streaming.test.ts | 49 ++++++ .../ingestion/pipeline-phase-registry.test.ts | 102 +++++++++++ .../unit/model/registration-table.test.ts | 29 ++++ gitnexus/test/unit/schema.test.ts | 39 ++++- .../unit/taint/source-sink-registry.test.ts | 45 +++++ 22 files changed, 1027 insertions(+), 30 deletions(-) create mode 100644 gitnexus/scripts/spikes/s1-reaching-def-index-bench.ts create mode 100644 gitnexus/scripts/spikes/s2-postdom-prototype.ts create mode 100644 gitnexus/src/core/ingestion/pipeline-phases/registry.ts create mode 100644 gitnexus/src/core/ingestion/taint/source-sink-config.ts create mode 100644 gitnexus/src/core/ingestion/taint/source-sink-registry.ts create mode 100644 gitnexus/test/integration/basicblock-roundtrip.test.ts create mode 100644 gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts create mode 100644 gitnexus/test/unit/taint/source-sink-registry.test.ts diff --git a/gitnexus-shared/src/graph/types.ts b/gitnexus-shared/src/graph/types.ts index ede8bc906..86abc9eba 100644 --- a/gitnexus-shared/src/graph/types.ts +++ b/gitnexus-shared/src/graph/types.ts @@ -44,7 +44,10 @@ export type NodeLabel = | 'Template' | 'Section' | 'Route' - | 'Tool'; + | 'Tool' + // Taint/PDG substrate (issue #2080). Intra-procedural control-flow node. + // Emitted by no phase yet — M1 (#2081) populates these behind an opt-in. + | 'BasicBlock'; export type NodeProperties = { name: string; @@ -89,6 +92,8 @@ export type NodeProperties = { responseKeys?: string[]; errorKeys?: string[]; middleware?: string[]; + // BasicBlock (taint/PDG substrate, issue #2080) — reuses filePath/startLine/endLine. + text?: string; // Extensible [key: string]: unknown; }; @@ -131,7 +136,28 @@ export type RelationshipType = * `reason` encodes the event name: `vue-emit: `. * Complements `BINDS_EVENT_HANDLER`; a Cypher query joining on the * component File node reveals all (emitter, handler) pairs. */ - | 'EMITS_EVENT'; + | 'EMITS_EVENT' + // ── Taint/PDG substrate (issue #2080) ──────────────────────────────────── + // Reserved edge types for the taint-first PDG substrate. No phase emits any + // of these yet; they are populated behind an opt-in by later milestones + // (CFG → M1 #2081, REACHING_DEF → M2 #2082, TAINTED/SANITIZES/TAINT_PATH → + // M3/M4 #2083/#2084). Adding them here keeps the shared schema stable so + // downstream work does not re-ripple the exhaustiveness sites. + /** Control-flow edge between two BasicBlock nodes (intra-procedural CFG). */ + | 'CFG' + /** Data-dependence edge: a definition of `variable` reaches a use of it. + * The `variable` name is stored in the relation's existing `reason` column + * (M0/S1 verdict: LadybugDB has no secondary index on relationship + * properties, so a dedicated indexed column would not speed the + * variable-filtered path query). */ + | 'REACHING_DEF' + /** A tainted value flows from source toward sink. */ + | 'TAINTED' + /** A sanitizer clears taint along a flow. */ + | 'SANITIZES' + /** Materialized source→sink taint path. Working name — final name/representation + * is confirmed when M3/M4 emits it; no persisted edge exists before then. */ + | 'TAINT_PATH'; export interface GraphNode { id: string; diff --git a/gitnexus-shared/src/lbug/schema-constants.ts b/gitnexus-shared/src/lbug/schema-constants.ts index 656ffe552..d022ba5c4 100644 --- a/gitnexus-shared/src/lbug/schema-constants.ts +++ b/gitnexus-shared/src/lbug/schema-constants.ts @@ -40,6 +40,8 @@ export const NODE_TABLES = [ 'Module', 'Route', 'Tool', + // Taint/PDG substrate (issue #2080) — inert until M1 (#2081) emits blocks. + 'BasicBlock', ] as const; export type NodeTableName = (typeof NODE_TABLES)[number]; @@ -67,6 +69,14 @@ export const REL_TYPES = [ 'ENTRY_POINT_OF', 'WRAPS', 'QUERIES', + // Taint/PDG substrate (issue #2080) — reserved edge types, emitted by no + // phase yet (CFG → M1, REACHING_DEF → M2, TAINTED/SANITIZES/TAINT_PATH → + // M3/M4). REACHING_DEF's variable name rides the relation's `reason` column. + 'CFG', + 'REACHING_DEF', + 'TAINTED', + 'SANITIZES', + 'TAINT_PATH', ] as const; export type RelType = (typeof REL_TYPES)[number]; diff --git a/gitnexus-web/src/lib/constants.ts b/gitnexus-web/src/lib/constants.ts index fe0505483..2f717cab3 100644 --- a/gitnexus-web/src/lib/constants.ts +++ b/gitnexus-web/src/lib/constants.ts @@ -38,6 +38,7 @@ export const NODE_COLORS: Record = { Template: '#a78bfa', // Violet light - like Type Route: '#f43f5e', // Rose - like Process Tool: '#a855f7', // Purple - like Project + BasicBlock: '#475569', // Slate darker - control-flow node (muted, taint/PDG substrate) }; // Node sizes by type - clear visual hierarchy with dramatic size differences @@ -79,6 +80,7 @@ export const NODE_SIZES: Record = { Template: 3, // Like Type Route: 5, // Like Enum Tool: 5, // Like Enum + BasicBlock: 2, // Tiny - control-flow node (taint/PDG substrate) }; // Community color palette for cluster-based coloring diff --git a/gitnexus/CHANGELOG.md b/gitnexus/CHANGELOG.md index e06601f34..5fe3e70f2 100644 --- a/gitnexus/CHANGELOG.md +++ b/gitnexus/CHANGELOG.md @@ -4,6 +4,10 @@ All notable changes to GitNexus will be documented in this file. ## [Unreleased] +### Added + +- **Taint/PDG substrate (M0)** — foundational schema + seams for reliable taint analysis on a PDG-expandable substrate (#2080, Epic #2087). Adds the `BasicBlock` node label and `CFG` / `REACHING_DEF` / `TAINTED` / `SANITIZES` / `TAINT_PATH` relationship types to the graph schema (round-trip through the bulk-COPY path), a phase-registry seam (`registerPhase` / `enabledWhen`) generalising the graph-phase opt-in guard, and a per-language source/sink/sanitizer config registry seam. All additive and inert — no phase emits the new nodes/edges yet, and a default `analyze` run is byte-identical to before. De-risking spikes (LadybugDB rel-property indexing, post-dominator feasibility) recorded on the issue. + ## [1.6.6] - 2026-06-08 ### Added diff --git a/gitnexus/scripts/spikes/s1-reaching-def-index-bench.ts b/gitnexus/scripts/spikes/s1-reaching-def-index-bench.ts new file mode 100644 index 000000000..98eb40bfc --- /dev/null +++ b/gitnexus/scripts/spikes/s1-reaching-def-index-bench.ts @@ -0,0 +1,151 @@ +/** + * Spike S1 (issue #2080, M0) — THROWAWAY benchmark. Not part of the build + * (scripts/ is excluded from tsconfig) or the test suite. + * + * Question: can LadybugDB serve the headline REACHING_DEF query + * [:REACHING_DEF*1..5 {variable}] + * fast enough, and what is the right storage shape for the `variable`? + * + * What it does: + * 1. Builds a synthetic ~100K-edge graph of BasicBlock nodes + REACHING_DEF + * edges (variable carried in the CodeRelation `reason` column) with a + * realistic per-variable fan-out distribution, and loads it through the + * real bulk-COPY path (loadGraphToLbug). + * 2. Probes whether LadybugDB supports a secondary index on a relationship + * property (the crux of the "edge property vs side table" decision). + * 3. Times the variable-filtered bounded var-length path query. + * + * Run: npx tsx scripts/spikes/s1-reaching-def-index-bench.ts [edgeCount] + */ +import fs from 'fs/promises'; +import path from 'path'; +import os from 'os'; +import { performance } from 'node:perf_hooks'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import type { KnowledgeGraph } from '../../src/core/graph/types.js'; + +const EDGE_COUNT = Number(process.argv[2] ?? 30_000); +// Realistic-ish def-use shape: many short chains, variables reused across them. +const CHAIN_LEN = 6; // blocks per function-ish chain +const DISTINCT_VARS = Math.max(1, Math.floor(EDGE_COUNT / 20)); // ~20 edges/variable fan-out + +const log = (m: string) => process.stdout.write(m + '\n'); + +function buildSynthGraph(edgeCount: number): KnowledgeGraph { + const g = createKnowledgeGraph(); + let edges = 0; + let chain = 0; + while (edges < edgeCount) { + const base = `BasicBlock:synth/f${chain}.ts`; + for (let i = 0; i <= CHAIN_LEN; i++) { + g.addNode({ + id: `${base}:${i}`, + label: 'BasicBlock', + properties: { + name: '', + filePath: `synth/f${chain}.ts`, + startLine: i, + endLine: i, + text: '', + }, + }); + } + for (let i = 0; i < CHAIN_LEN && edges < edgeCount; i++) { + const variable = `v${edges % DISTINCT_VARS}`; + g.addRelationship({ + id: `${base}:${i}->${i + 1}:${variable}`, + sourceId: `${base}:${i}`, + targetId: `${base}:${i + 1}`, + type: 'REACHING_DEF', + confidence: 1.0, + reason: variable, // M0 storage: variable rides `reason` + }); + edges++; + } + chain++; + } + return g; +} + +async function main() { + const tmp = path.join(os.tmpdir(), `s1-spike-${Date.now()}`); + const storagePath = path.join(tmp, '.gitnexus'); + const dbPath = path.join(storagePath, 'lbug'); + await fs.mkdir(dbPath, { recursive: true }); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.initLbug(dbPath); + + log( + `[S1] building synthetic graph: ~${EDGE_COUNT} REACHING_DEF edges, ` + + `${DISTINCT_VARS} distinct variables (~20 edges/var fan-out), chains of ${CHAIN_LEN}`, + ); + const g = buildSynthGraph(EDGE_COUNT); + + let t = performance.now(); + await adapter.loadGraphToLbug(g, tmp, storagePath); + const loadMs = performance.now() - t; + const stats = await adapter.getLbugStats(); + log(`[S1] bulk-COPY load: ${loadMs.toFixed(0)}ms (nodes=${stats.nodes}, edges=${stats.edges})`); + + // (2) Probe: does LadybugDB support a secondary index on a REL property? + let relIndexSupported = false; + let relIndexErr = ''; + for (const stmt of [ + "CALL CREATE_REL_INDEX('CodeRelation', 'cr_reason_idx', 'reason')", + 'CREATE INDEX cr_reason_idx ON CodeRelation(reason)', + ]) { + try { + await adapter.executeQuery(stmt); + relIndexSupported = true; + break; + } catch (e: any) { + relIndexErr = String(e?.message ?? e).split('\n')[0]; + } + } + log( + `[S1] rel-property secondary index supported? ${relIndexSupported} ` + + `(last error: ${relIndexErr})`, + ); + + // (3a) Single-hop variable filter — the common case M3 runs most. + const probeVar = 'v0'; + t = performance.now(); + const single = await adapter.executeQuery( + `MATCH (a:BasicBlock)-[r:CodeRelation {type: 'REACHING_DEF', reason: '${probeVar}'}]->(b:BasicBlock) + RETURN count(r) AS c`, + ); + const singleMs = performance.now() - t; + log(`[S1] single-hop variable filter → ${single[0]?.c} edges in ${singleMs.toFixed(0)}ms`); + + // (3b) SOURCE-ANCHORED bounded var-length path — the realistic taint query + // (anchor the source block, then walk REACHING_DEF up to 5 hops). The + // UNANCHORED global form ([:REACHING_DEF*1..5] from every block) is + // impractical at scale (path explosion) — that is itself an S1 finding: + // taint queries MUST be scoped to a source block, not run graph-wide. + const srcId = 'BasicBlock:synth/f0.ts:0'; + t = performance.now(); + const anchored = await adapter.executeQuery( + `MATCH p = (a:BasicBlock)-[:CodeRelation*1..5 {type: 'REACHING_DEF'}]->(b:BasicBlock) + WHERE a.id = '${srcId}' AND all(rel IN relationships(p) WHERE rel.reason = '${probeVar}') + RETURN count(p) AS paths`, + ); + const pathMs = performance.now() - t; + log( + `[S1] source-anchored [:REACHING_DEF*1..5 {reason='${probeVar}'}] from one block → ` + + `${anchored[0]?.paths} paths in ${pathMs.toFixed(0)}ms`, + ); + + await adapter.closeLbug(); + await fs.rm(tmp, { recursive: true, force: true }); + + log('\n[S1] VERDICT INPUTS:'); + log( + ` load_ms=${loadMs.toFixed(0)} single_hop_ms=${singleMs.toFixed(0)} anchored_path_ms=${pathMs.toFixed(0)} rel_index=${relIndexSupported}`, + ); +} + +main().catch((e) => { + console.error('[S1] FAILED:', e); + process.exit(1); +}); diff --git a/gitnexus/scripts/spikes/s2-postdom-prototype.ts b/gitnexus/scripts/spikes/s2-postdom-prototype.ts new file mode 100644 index 000000000..da06d8e7e --- /dev/null +++ b/gitnexus/scripts/spikes/s2-postdom-prototype.ts @@ -0,0 +1,162 @@ +/** + * Spike S2 (issue #2080, M0) — THROWAWAY post-dominator feasibility prototype. + * Not part of the build (scripts/ excluded from tsconfig) or the test suite. + * + * Question (per maintainer review): does the post-dominator algorithm Epic B + * (#2085, CDG) depends on hold up on real TS/JS control-flow shapes — the + * classic CFG hazards — before Epic B commits to it? + * + * Scope boundary: post-dominators operate on a CFG, not on the AST directly. + * This prototype validates the ALGORITHM (iterative dataflow on the reverse + * CFG, EXIT-rooted, → immediate-post-dominator tree) against CFGs that model + * each hazard's real TS control flow (the TS source each CFG represents is + * shown inline). Building the CFG from a tree-sitter AST is M1's job (#2081); + * this spike deliberately does not reimplement it. + * + * Run: npx tsx scripts/spikes/s2-postdom-prototype.ts + */ + +type CFG = { + name: string; + tsSource: string; + entry: string; + exit: string; + // adjacency: block -> successors + succ: Record; + hazard: string; +}; + +// Iterative post-dominator dataflow on the reverse CFG. +// PostDom(EXIT) = {EXIT}; PostDom(n) = {n} ∪ (⋂ PostDom(s) for s ∈ succ(n)). +// Monotone over a finite lattice (powerset of blocks) ⇒ guaranteed to converge. +function postDominators(cfg: CFG): { pdom: Record>; iterations: number } { + const blocks = Object.keys(cfg.succ); + const all = new Set(blocks); + const pdom: Record> = {}; + for (const b of blocks) pdom[b] = b === cfg.exit ? new Set([cfg.exit]) : new Set(all); + + let changed = true; + let iterations = 0; + while (changed) { + changed = false; + iterations++; + for (const b of blocks) { + if (b === cfg.exit) continue; + const succs = cfg.succ[b] ?? []; + let inter: Set | null = null; + for (const s of succs) { + if (inter === null) inter = new Set(pdom[s]); + else inter = new Set([...inter].filter((x) => pdom[s].has(x))); + } + const next = new Set(inter ?? []); + next.add(b); + if (next.size !== pdom[b].size || [...next].some((x) => !pdom[b].has(x))) { + pdom[b] = next; + changed = true; + } + } + if (iterations > blocks.length + 5) + throw new Error('post-dom did not converge (suspected bug)'); + } + return { pdom, iterations }; +} + +// Immediate post-dominator: the closest strict post-dominator. +function ipdom(cfg: CFG, pdom: Record>): Record { + const res: Record = {}; + for (const b of Object.keys(cfg.succ)) { + if (b === cfg.exit) { + res[b] = null; + continue; + } + const strict = [...pdom[b]].filter((x) => x !== b); + // ipdom = the strict post-dom that does not post-dominate any other strict post-dom. + res[b] = + strict.find((cand) => strict.every((other) => other === cand || !pdom[other].has(cand))) ?? + null; + } + return res; +} + +const CFGS: CFG[] = [ + { + name: 'early-return', + hazard: 'early return / multiple paths to EXIT', + tsSource: `function f(x){ if (x) { return 1; } g(); return 2; }`, + entry: 'ENTRY', + exit: 'EXIT', + succ: { ENTRY: ['ret1', 'g'], ret1: ['EXIT'], g: ['ret2'], ret2: ['EXIT'], EXIT: [] }, + }, + { + name: 'try-throw-finally', + hazard: 'try/throw/finally with multiple exits through finally', + tsSource: `function f(){ try { risky(); } catch(e){ handle(e); } finally { cleanup(); } done(); }`, + entry: 'ENTRY', + exit: 'EXIT', + // try → (normal | throw→catch) → finally → done → EXIT; finally also reached on rethrow + succ: { + ENTRY: ['try'], + try: ['finally', 'catch'], + catch: ['finally'], + finally: ['done', 'EXIT'], + done: ['EXIT'], + EXIT: [], + }, + }, + { + name: 'labeled-break', + hazard: 'labeled break/continue across nested loops', + tsSource: `outer: for(;;){ for(;;){ if (a) break outer; if (b) continue outer; work(); } }`, + entry: 'ENTRY', + exit: 'EXIT', + succ: { + ENTRY: ['outerHead'], + outerHead: ['innerHead', 'EXIT'], + innerHead: ['breakOuter', 'afterIf1'], + breakOuter: ['EXIT'], + afterIf1: ['contOuter', 'work'], + contOuter: ['outerHead'], + work: ['innerHead'], + EXIT: [], + }, + }, + { + name: 'if-else-diamond', + hazard: 'baseline reducible diamond (sanity)', + tsSource: `function f(x){ if (x) { a(); } else { b(); } c(); }`, + entry: 'ENTRY', + exit: 'EXIT', + succ: { ENTRY: ['a', 'b'], a: ['c'], b: ['c'], c: ['EXIT'], EXIT: [] }, + }, +]; + +function main() { + let allOk = true; + for (const cfg of CFGS) { + try { + const { pdom, iterations } = postDominators(cfg); + const idom = ipdom(cfg, pdom); + // Sanity invariants: EXIT post-dominates every block; ipdom tree reaches EXIT. + const exitPostDomsAll = Object.keys(cfg.succ).every((b) => pdom[b].has(cfg.exit)); + console.log(`\n[S2] ${cfg.name} — ${cfg.hazard}`); + console.log(` TS: ${cfg.tsSource}`); + console.log( + ` converged in ${iterations} iters; EXIT post-dominates all blocks: ${exitPostDomsAll}`, + ); + console.log( + ` ipdom tree: ${Object.entries(idom) + .map(([b, p]) => `${b}->${p ?? '∅'}`) + .join(' ')}`, + ); + if (!exitPostDomsAll) allOk = false; + } catch (e) { + allOk = false; + console.log(`\n[S2] ${cfg.name} FAILED: ${(e as Error).message}`); + } + } + console.log( + `\n[S2] VERDICT INPUT: all hazard CFGs converged + EXIT post-dominates all = ${allOk}`, + ); +} + +main(); diff --git a/gitnexus/src/core/ingestion/model/registration-table.ts b/gitnexus/src/core/ingestion/model/registration-table.ts index 6e0e2646d..2de59e966 100644 --- a/gitnexus/src/core/ingestion/model/registration-table.ts +++ b/gitnexus/src/core/ingestion/model/registration-table.ts @@ -182,6 +182,9 @@ const LABEL_BEHAVIOR = { Section: 'inert', Route: 'inert', Tool: 'inert', + // Taint/PDG substrate (issue #2080) — a control-flow node, never a + // symbol-resolution target. Inert: file index only, no owner scope. + BasicBlock: 'inert', } as const satisfies Record & // Cross-invariant 1 — every class-like label (participates in // qualifiedName fallback in `SymbolTable.add()`) MUST be classified as diff --git a/gitnexus/src/core/ingestion/pipeline-phases/index.ts b/gitnexus/src/core/ingestion/pipeline-phases/index.ts index 60f41d87d..4d8fa93bd 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/index.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/index.ts @@ -30,3 +30,4 @@ export { processesPhase, type ProcessesOutput } from './processes.js'; export { runPipeline } from './runner.js'; export type { PipelinePhase, PipelineContext, PhaseResult } from './types.js'; export { getPhaseOutput } from './types.js'; +export { PhaseRegistry, type RegisterPhaseOptions } from './registry.js'; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/registry.ts b/gitnexus/src/core/ingestion/pipeline-phases/registry.ts new file mode 100644 index 000000000..c64988208 --- /dev/null +++ b/gitnexus/src/core/ingestion/pipeline-phases/registry.ts @@ -0,0 +1,73 @@ +/** + * Phase registry seam (issue #2080, taint/PDG substrate M0). + * + * A small, behaviour-preserving abstraction over phase-list *assembly*. Today + * `buildPhaseList` is a hand-maintained array with a single ad-hoc + * `if (!skipGraphPhases)` guard; this registry generalises that guard into a + * per-phase `enabledWhen` predicate so later milestones can register opt-in + * phases (e.g. CFG → M1 #2081) without editing the array each time. + * + * M0 wires the seam with **no behaviour change**: `build(options)` must return + * a phase list identical in membership and order to the legacy array for every + * options combination. The registry covers only list assembly — the runner, + * topological sort, `PipelinePhase.execute`, and any result-extraction guards + * (e.g. the `skipGraphPhases` check in `runPipelineFromRepo`) are untouched. + * + * Generic over the options type so this module depends only on `PipelinePhase` + * (no import of `PipelineOptions`, which lives in `pipeline.ts` and would + * otherwise create an import cycle). + */ + +import type { PipelinePhase } from './types.js'; + +/** Options accepted when registering a phase. */ +export interface RegisterPhaseOptions { + /** + * Predicate deciding whether this phase is included for a given options + * object. Absent ⇒ the phase is always enabled. This is the generalised + * form of the legacy `if (!skipGraphPhases)` guard. + * + * `options` is required, not optional: callers normalize an absent options + * object once at `build()` (e.g. `buildPhaseList` passes `options ?? {}`), so + * individual predicates read `(o) => !o.skipGraphPhases` without a defensive + * `?.` on every phase (#2080 review S1). + */ + readonly enabledWhen?: (options: TOptions) => boolean; +} + +interface PhaseRegistration { + readonly phase: PipelinePhase; + readonly enabledWhen?: (options: TOptions) => boolean; +} + +/** + * Ordered registry of pipeline phases. Not a global singleton — callers + * construct a fresh registry (so registration order is deterministic and there + * is no import-order or test-isolation hazard) and `build()` it per run. + */ +export class PhaseRegistry { + private readonly registrations: PhaseRegistration[] = []; + + /** + * Register a phase. This is the `registerPhase(phase, { enabledWhen })` seam + * named in issue #2080. Returns `this` for fluent chaining. Registration + * order is preserved by `build()`. + */ + register(phase: PipelinePhase, options?: RegisterPhaseOptions): this { + this.registrations.push({ phase, enabledWhen: options?.enabledWhen }); + return this; + } + + /** + * Build the ordered phase list for the given options. A phase is included + * iff it has no `enabledWhen` predicate or its predicate returns `true`. + * Order matches registration order. `options` is required — callers that may + * have no options normalize once at the call site (`options ?? {}`) so the + * predicates never see `undefined`. + */ + build(options: TOptions): PipelinePhase[] { + return this.registrations + .filter((r) => r.enabledWhen === undefined || r.enabledWhen(options)) + .map((r) => r.phase); + } +} diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index 611b7ac7a..0e610e647 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -35,6 +35,7 @@ import { mroPhase, communitiesPhase, processesPhase, + PhaseRegistry, type ScopeResolutionOutput, type PipelinePhase, type CommunitiesOutput, @@ -142,28 +143,36 @@ export interface PipelineOptions { * → mro → communities → processes * * To add a new phase: create a file in pipeline-phases/, export the phase - * object, and add it to the appropriate position in this array. + * object, and `.register()` it at the appropriate position below. Opt-in + * phases pass an `enabledWhen` predicate (issue #2080 phase-registry seam) — + * the legacy `if (!skipGraphPhases)` guard is now expressed that way on the + * three graph phases, with no change in behaviour. + * + * Exported for the parity test (`pipeline-phase-registry.test.ts`), which + * asserts the produced list is byte-identical to the legacy array for every + * options combination. */ -function buildPhaseList(options?: PipelineOptions): PipelinePhase[] { - const phases: PipelinePhase[] = [ - scanPhase, - structurePhase, - markdownPhase, - cobolPhase, - parsePhase, - routesPhase, - toolsPhase, - ormPhase, - crossFilePhase, - scopeResolutionPhase, - pruneLocalSymbolsPhase, - ]; - - if (!options?.skipGraphPhases) { - phases.push(mroPhase, communitiesPhase, processesPhase); - } - - return phases; +export function buildPhaseList(options?: PipelineOptions): PipelinePhase[] { + return ( + new PhaseRegistry() + .register(scanPhase) + .register(structurePhase) + .register(markdownPhase) + .register(cobolPhase) + .register(parsePhase) + .register(routesPhase) + .register(toolsPhase) + .register(ormPhase) + .register(crossFilePhase) + .register(scopeResolutionPhase) + .register(pruneLocalSymbolsPhase) + .register(mroPhase, { enabledWhen: (o) => !o.skipGraphPhases }) + .register(communitiesPhase, { enabledWhen: (o) => !o.skipGraphPhases }) + .register(processesPhase, { enabledWhen: (o) => !o.skipGraphPhases }) + // Normalize a missing options object once here so phase predicates above + // take a required PipelineOptions and need no `?.` guard (#2080 review S1). + .build(options ?? {}) + ); } // ── Pipeline orchestrator ───────────────────────────────────────────────── diff --git a/gitnexus/src/core/ingestion/taint/source-sink-config.ts b/gitnexus/src/core/ingestion/taint/source-sink-config.ts new file mode 100644 index 000000000..2b03bf289 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/source-sink-config.ts @@ -0,0 +1,38 @@ +/** + * Source/sink/sanitizer config model (issue #2080, taint/PDG substrate M0). + * + * The per-language taint configuration *shape*. M0 ships only the type and an + * (empty) registry seam — no analysis consumes it yet. M3 (#2083, intra-proc + * taint) populates per-language specs and reads them when emitting TAINTED / + * SANITIZES edges. + * + * Kept deliberately minimal: enough for M3 to express "callable X is a + * source / sink / sanitizer, optionally for argument position N" without M0 + * committing to matcher semantics it cannot yet validate. The shape is + * expected to grow (e.g. sanitizer escape conditions, return-position taint) + * when M3 makes contact with real flows; that is a forward-declared-interface + * design choice, not a finished contract. + */ + +/** + * Identifies a callable that participates in taint flow. `name` is matched + * against a resolved callable (simple or qualified name — exact matching + * semantics are M3's call). `args` optionally narrows to specific 0-based + * argument positions that carry taint (for a source/sink) or clear it (for a + * sanitizer); omit to mean "unspecified / all". + */ +export interface TaintCallableMatcher { + readonly name: string; + readonly args?: readonly number[]; +} + +/** + * The taint configuration for a single language: which callables introduce + * taint (sources), which are dangerous to reach with tainted input (sinks), + * and which clear taint (sanitizers). + */ +export interface SourceSinkSanitizerSpec { + readonly sources: readonly TaintCallableMatcher[]; + readonly sinks: readonly TaintCallableMatcher[]; + readonly sanitizers: readonly TaintCallableMatcher[]; +} diff --git a/gitnexus/src/core/ingestion/taint/source-sink-registry.ts b/gitnexus/src/core/ingestion/taint/source-sink-registry.ts new file mode 100644 index 000000000..7fe1f1229 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/source-sink-registry.ts @@ -0,0 +1,42 @@ +/** + * Per-language source/sink/sanitizer registry seam (issue #2080). + * + * A keyed registry of {@link SourceSinkSanitizerSpec} by language id. M0 stands + * up the empty seam — no language is registered and nothing in the pipeline + * reads it. M3 (#2083) registers per-language specs and queries this registry + * when emitting taint edges. + * + * The store is module-level (matching the codebase's other per-language + * registries). {@link clearSourceSinkRegistry} resets it for test isolation. + */ + +import type { SourceSinkSanitizerSpec } from './source-sink-config.js'; + +const registry = new Map(); + +/** + * Register the taint config for a language. Last-write-wins: re-registering + * the same `languageId` overwrites the previous spec (so M3 can override a + * built-in default). Returns nothing. + */ +export function registerSourceSinkConfig(languageId: string, spec: SourceSinkSanitizerSpec): void { + registry.set(languageId, spec); +} + +/** + * Look up the taint config for a language. Returns `undefined` when no spec is + * registered (the M0 default for every language) — never throws. + */ +export function getSourceSinkConfig(languageId: string): SourceSinkSanitizerSpec | undefined { + return registry.get(languageId); +} + +/** Language ids that currently have a registered spec. Empty in M0. */ +export function registeredTaintLanguages(): string[] { + return [...registry.keys()]; +} + +/** Reset the registry. Primarily for test isolation. */ +export function clearSourceSinkRegistry(): void { + registry.clear(); +} diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index e03db4b91..c7d8413b5 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -305,6 +305,13 @@ export const streamAllCSVsToDisk = async ( 'id,name,filePath,description', ); + // BasicBlock nodes — taint/PDG substrate (issue #2080). No `name` column; + // blocks are identified by id + source span. Emitted by no phase yet. + const basicBlockWriter = new BufferedCSVWriter( + path.join(csvDir, 'basicblock.csv'), + 'id,filePath,startLine,endLine,text', + ); + // Multi-language node types share the same CSV shape (no isExported column) const multiLangHeader = 'id,name,filePath,startLine,endLine,content,description'; const MULTI_LANG_TYPES = [ @@ -478,6 +485,17 @@ export const streamAllCSVsToDisk = async ( ].join(','), ); break; + case 'BasicBlock': + await basicBlockWriter.addRow( + [ + escapeCSVField(node.id), + escapeCSVField(node.properties.filePath || ''), + escapeCSVNumber(node.properties.startLine, -1), + escapeCSVNumber(node.properties.endLine, -1), + escapeCSVField(node.properties.text || ''), + ].join(','), + ); + break; default: { // Code element nodes (Function, Class, Interface, CodeElement) const writer = codeWriterMap[node.label]; @@ -535,6 +553,7 @@ export const streamAllCSVsToDisk = async ( sectionWriter, routeWriter, toolWriter, + basicBlockWriter, ...multiLangWriters.values(), ]; await Promise.all(allWriters.map((w) => w.finish())); @@ -571,6 +590,7 @@ export const streamAllCSVsToDisk = async ( ['Section' as NodeTableName, sectionWriter], ['Route' as NodeTableName, routeWriter], ['Tool' as NodeTableName, toolWriter], + ['BasicBlock' as NodeTableName, basicBlockWriter], ...Array.from(multiLangWriters.entries()).map( ([name, w]) => [name as NodeTableName, w] as [NodeTableName, BufferedCSVWriter], ), diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index 51f5ce860..101783bcb 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -1124,6 +1124,10 @@ const getCopyQuery = (table: NodeTableName, filePath: string): string => { if (table === 'Tool') { return `COPY ${t}(id, name, filePath, description) FROM "${filePath}" ${COPY_CSV_OPTS}`; } + if (table === 'BasicBlock') { + // Taint/PDG substrate (issue #2080) — no name column. + return `COPY ${t}(id, filePath, startLine, endLine, text) FROM "${filePath}" ${COPY_CSV_OPTS}`; + } if (table === 'Method') { return `COPY ${t}(id, name, filePath, startLine, endLine, isExported, content, description, parameterCount, returnType) FROM "${filePath}" ${COPY_CSV_OPTS}`; } @@ -1176,6 +1180,9 @@ export const insertNodeToLbug = async ( ? `, description: ${escapeValue(properties.description)}` : ''; query = `CREATE (n:Section {id: ${escapeValue(properties.id)}, name: ${escapeValue(properties.name)}, filePath: ${escapeValue(properties.filePath)}, startLine: ${properties.startLine || 0}, endLine: ${properties.endLine || 0}, level: ${properties.level || 1}, content: ${escapeValue(properties.content || '')}${descPart}})`; + } else if (label === 'BasicBlock') { + // Taint/PDG substrate (issue #2080) — no name column. + query = `CREATE (n:BasicBlock {id: ${escapeValue(properties.id)}, filePath: ${escapeValue(properties.filePath)}, startLine: ${properties.startLine || 0}, endLine: ${properties.endLine || 0}, text: ${escapeValue(properties.text || '')}})`; } else if (TABLES_WITH_EXPORTED.has(label)) { const descPart = properties.description ? `, description: ${escapeValue(properties.description)}` @@ -1259,6 +1266,9 @@ export const batchInsertNodesToLbug = async ( ? `, n.description = ${escapeValue(properties.description)}` : ''; query = `MERGE (n:Section {id: ${escapeValue(properties.id)}}) SET n.name = ${escapeValue(properties.name)}, n.filePath = ${escapeValue(properties.filePath)}, n.startLine = ${properties.startLine || 0}, n.endLine = ${properties.endLine || 0}, n.level = ${properties.level || 1}, n.content = ${escapeValue(properties.content || '')}${descPart}`; + } else if (label === 'BasicBlock') { + // Taint/PDG substrate (issue #2080) — no name column. + query = `MERGE (n:BasicBlock {id: ${escapeValue(properties.id)}}) SET n.filePath = ${escapeValue(properties.filePath)}, n.startLine = ${properties.startLine || 0}, n.endLine = ${properties.endLine || 0}, n.text = ${escapeValue(properties.text || '')}`; } else if (TABLES_WITH_EXPORTED.has(label)) { const descPart = properties.description ? `, n.description = ${escapeValue(properties.description)}` diff --git a/gitnexus/src/core/lbug/schema.ts b/gitnexus/src/core/lbug/schema.ts index cd6838dd8..4c8cc7cc5 100644 --- a/gitnexus/src/core/lbug/schema.ts +++ b/gitnexus/src/core/lbug/schema.ts @@ -221,6 +221,23 @@ CREATE NODE TABLE Section ( PRIMARY KEY (id) )`; +// Taint/PDG substrate (issue #2080) — intra-procedural control-flow node. +// Emitted by no phase yet; M1 (#2081) populates these behind an opt-in. +// REACHING_DEF carries its variable name in the relation's existing `reason` +// column (see RELATION_SCHEMA) — LadybugDB has no secondary index on rel +// properties, so a dedicated indexed column would buy nothing for the +// variable-filtered path query (M0/S1 verdict). No `name` column: blocks are +// identified by id + source span, not a symbol name. +export const BASICBLOCK_SCHEMA = ` +CREATE NODE TABLE BasicBlock ( + id STRING, + filePath STRING, + startLine INT64, + endLine INT64, + text STRING, + PRIMARY KEY (id) +)`; + // ============================================================================ // RELATION TABLE SCHEMA // Single table with 'type' property - connects all node tables @@ -431,6 +448,7 @@ CREATE REL TABLE ${REL_TABLE_NAME} ( FROM CodeElement TO Process, FROM Route TO Process, FROM Tool TO Process, + FROM BasicBlock TO BasicBlock, type STRING, confidence DOUBLE, reason STRING, @@ -521,6 +539,11 @@ export const NODE_SCHEMA_QUERIES = [ ROUTE_SCHEMA, // MCP tools TOOL_SCHEMA, + // Taint/PDG substrate (issue #2080) — must be appended here, not just + // declared above: SCHEMA_QUERIES (the list initLbug actually runs) is built + // from NODE_SCHEMA_QUERIES. Omitting this leaves the BasicBlock table + // uncreated and the bulk-COPY round-trip fails with "table does not exist". + BASICBLOCK_SCHEMA, ]; export const REL_SCHEMA_QUERIES = [RELATION_SCHEMA]; diff --git a/gitnexus/src/server/api.ts b/gitnexus/src/server/api.ts index da99ed93b..6ab9cf1c3 100644 --- a/gitnexus/src/server/api.ts +++ b/gitnexus/src/server/api.ts @@ -344,9 +344,17 @@ const GRAPH_RELATIONSHIP_QUERY = const quoteNodeTable = (table: string): string => `\`${table.replace(/`/g, '``')}\``; -const getNodeQuery = (table: string, includeContent: boolean): string => { +export const getNodeQuery = (table: string, includeContent: boolean): string => { const tableLabel = quoteNodeTable(table); + if (table === 'BasicBlock') { + // Taint/PDG substrate (issue #2080) — BasicBlock has no name/content + // columns. Project only its declared columns: a default `n.name` + // projection raises a Ladybug "Cannot find property name" binder error + // (not matched by isIgnorableGraphQueryError), which would 500 the graph + // endpoint the moment BasicBlock joins NODE_TABLES, even on an empty table. + return `MATCH (n:${tableLabel}) RETURN n.id AS id, n.filePath AS filePath, n.startLine AS startLine, n.endLine AS endLine, n.text AS text`; + } if (table === 'File') { return includeContent ? `MATCH (n:${tableLabel}) RETURN n.id AS id, n.name AS name, n.filePath AS filePath, n.content AS content` @@ -376,10 +384,17 @@ const mapGraphNodeRow = (table: string, row: any, includeContent: boolean): Grap id: row.id ?? row[0], label: table as GraphNode['label'], properties: { - name: row.name ?? row.label ?? row[1], + // `?? ''` keeps NodeProperties.name a `string` even for label rows that + // project no name/label column (BasicBlock — taint/PDG substrate #2080). + // Without it, BasicBlock rows carry name:undefined (masked by the cast + // below) and the web layer (Header search, circles/tree layout) derefs + // `.name` unguarded → TypeError once M1 emits blocks. `row.text` gives a + // BasicBlock a sensible fallback name before the empty-string floor. + name: row.name ?? row.label ?? row.text ?? row[1] ?? '', filePath: row.filePath ?? row[2], startLine: row.startLine, endLine: row.endLine, + text: row.text, content: includeContent ? row.content : undefined, responseKeys: row.responseKeys, errorKeys: row.errorKeys, diff --git a/gitnexus/test/integration/basicblock-roundtrip.test.ts b/gitnexus/test/integration/basicblock-roundtrip.test.ts new file mode 100644 index 000000000..88f9a60c9 --- /dev/null +++ b/gitnexus/test/integration/basicblock-roundtrip.test.ts @@ -0,0 +1,154 @@ +/** + * Integration test: BasicBlock + taint/PDG edge types round-trip the + * bulk-COPY load path (issue #2080, U5 / R4 / AC2). + * + * Exercises the real csv-generator → loadGraphToLbug → COPY → query path: + * - a BasicBlock node (id/filePath/startLine/endLine/text) round-trips + * - one edge of each new type (CFG/REACHING_DEF/TAINTED/SANITIZES/TAINT_PATH) + * between two BasicBlocks round-trips (asserts the new FROM/TO DDL pair + + * REL_TYPES load through bulk COPY) + * - REACHING_DEF carries its `variable` in the existing `reason` column + * (M0/S1 storage decision) and a variable-filtered query returns it + * - the DDL (BASICBLOCK_SCHEMA wired into NODE_SCHEMA_QUERIES) loads on a + * fresh DB — if BASICBLOCK_SCHEMA were not in SCHEMA_QUERIES, initLbug would + * never create the table and these COPYs would fail (F1 guard, end-to-end) + */ +import { describe, it, expect, beforeAll, afterAll } from 'vitest'; +import fs from 'fs/promises'; +import path from 'path'; +import os from 'os'; +import { NODE_TABLES } from 'gitnexus-shared'; +import { buildTestGraph } from '../helpers/test-graph.js'; +import { getNodeQuery } from '../../src/server/api.js'; + +let tmpBase: string; +let storagePath: string; +let dbPath: string; + +const BB1 = 'BasicBlock:src/a.ts:0'; +const BB2 = 'BasicBlock:src/a.ts:1'; +const NEW_EDGE_TYPES = ['CFG', 'REACHING_DEF', 'TAINTED', 'SANITIZES', 'TAINT_PATH'] as const; + +beforeAll(async () => { + tmpBase = path.join(os.tmpdir(), `gitnexus-bb-roundtrip-${Date.now()}-${process.pid}`); + storagePath = path.join(tmpBase, '.gitnexus'); + dbPath = path.join(storagePath, 'lbug'); + await fs.mkdir(dbPath, { recursive: true }); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.initLbug(dbPath); + + // Two BasicBlock nodes + one edge of each new type between them. The + // REACHING_DEF edge stores its variable name ('x') in `reason`. + const graph = buildTestGraph( + [ + { + id: BB1, + label: 'BasicBlock', + name: '', // BasicBlock has no name column; ignored by the writer + filePath: 'src/a.ts', + startLine: 1, + endLine: 3, + extra: { text: 'const x = req.body;' }, + }, + { + id: BB2, + label: 'BasicBlock', + name: '', + filePath: 'src/a.ts', + startLine: 4, + endLine: 6, + extra: { text: 'sink(x);' }, + }, + ], + NEW_EDGE_TYPES.map((type) => ({ + sourceId: BB1, + targetId: BB2, + type, + reason: type === 'REACHING_DEF' ? 'x' : `${type.toLowerCase()}-edge`, + })), + ); + + await adapter.loadGraphToLbug(graph, tmpBase, storagePath); +}); + +afterAll(async () => { + try { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.closeLbug(); + } catch { + /* may not have opened */ + } + if (tmpBase) { + for (let attempt = 0; attempt < 5; attempt++) { + try { + await fs.rm(tmpBase, { recursive: true, force: true }); + return; + } catch { + if (attempt < 4) await new Promise((r) => setTimeout(r, 200 * (attempt + 1))); + } + } + } +}); + +describe('BasicBlock + taint/PDG edge round-trip (#2080)', () => { + it('BasicBlock nodes round-trip with their source span and text', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const rows = await adapter.executeQuery( + 'MATCH (n:BasicBlock) RETURN n.id AS id, n.text AS text, n.filePath AS filePath, n.startLine AS startLine, n.endLine AS endLine ORDER BY n.id', + ); + expect(rows).toHaveLength(2); + expect(rows[0].id).toBe(BB1); + expect(rows[0].text).toBe('const x = req.body;'); + expect(rows[0].filePath).toBe('src/a.ts'); + expect(Number(rows[0].startLine)).toBe(1); + expect(Number(rows[0].endLine)).toBe(3); + expect(rows[1].id).toBe(BB2); + expect(rows[1].text).toBe('sink(x);'); + expect(rows[1].filePath).toBe('src/a.ts'); + expect(Number(rows[1].endLine)).toBe(6); + }); + + // Regression guard: adding a node table whose columns differ from the + // default (BasicBlock has no name/content) must not break the server's + // graph read path. getNodeQuery is what /api/graph's buildGraph + + // streamGraphNdjson run per NODE_TABLE; a default `n.name` projection on + // BasicBlock raises a non-ignorable Ladybug binder error → HTTP 500 on + // every analyzed repo. Assert every NODE_TABLE's query binds + runs, and + // that BasicBlock returns its loaded rows. + it('getNodeQuery binds + runs for every NODE_TABLE against the real schema', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + for (const table of NODE_TABLES) { + for (const includeContent of [false, true]) { + const q = getNodeQuery(table, includeContent); + await expect( + adapter.executeQuery(q), + `getNodeQuery(${table}, includeContent=${includeContent}) should bind`, + ).resolves.toBeDefined(); + } + } + const bbRows = await adapter.executeQuery(getNodeQuery('BasicBlock', false)); + expect(bbRows).toHaveLength(2); + }); + + it('each new edge type round-trips between the two BasicBlocks', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + // All edges live in the single CodeRelation table, keyed by `type`. + for (const type of NEW_EDGE_TYPES) { + const rows = await adapter.executeQuery( + `MATCH (:BasicBlock)-[r:CodeRelation {type: '${type}'}]->(:BasicBlock) RETURN count(r) AS c`, + ); + expect(Number(rows[0].c), `${type} edge should round-trip`).toBe(1); + } + }); + + it('REACHING_DEF carries its variable in reason and is queryable by it', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const rows = await adapter.executeQuery( + "MATCH (a:BasicBlock)-[r:CodeRelation {type: 'REACHING_DEF', reason: 'x'}]->(b:BasicBlock) RETURN a.id AS from, b.id AS to", + ); + expect(rows).toHaveLength(1); + expect(rows[0].from).toBe(BB1); + expect(rows[0].to).toBe(BB2); + }); +}); diff --git a/gitnexus/test/unit/api-graph-streaming.test.ts b/gitnexus/test/unit/api-graph-streaming.test.ts index 0cc69833c..3db16f51b 100644 --- a/gitnexus/test/unit/api-graph-streaming.test.ts +++ b/gitnexus/test/unit/api-graph-streaming.test.ts @@ -232,4 +232,53 @@ describe('streamGraphNdjson', () => { }, }); }); + + // Taint/PDG substrate (#2080): BasicBlock has no name/content columns, so its + // getNodeQuery projects none — mapGraphNodeRow must still yield a `string` + // name (NodeProperties.name contract) or the web layer derefs undefined. + it('emits a string name for BasicBlock nodes (no name column)', async () => { + lbugMocks.streamQuery.mockImplementation( + async (query: string, onRow: (row: any) => Promise) => { + if (query.includes('MATCH (n:`BasicBlock`)')) { + expect(query).not.toContain('n.name'); // BasicBlock projects no name column + await onRow({ + id: 'BasicBlock:src/a.ts:0', + filePath: 'src/a.ts', + startLine: 1, + endLine: 3, + text: 'const x = req.body;', + }); + // a block with no text must still map to a string name, not undefined + await onRow({ + id: 'BasicBlock:src/a.ts:1', + filePath: 'src/a.ts', + startLine: 4, + endLine: 4, + }); + return 2; + } + return 0; + }, + ); + + const writes: string[] = []; + const response = createMockResponse((chunk) => { + writes.push(chunk); + return true; + }); + + await expect(streamGraphNdjson(response, false)).resolves.toBeUndefined(); + + const blocks = writes + .map((chunk) => JSON.parse(chunk)) + .filter((r) => r.type === 'node' && r.data.label === 'BasicBlock'); + expect(blocks).toHaveLength(2); + for (const b of blocks) { + expect(typeof b.data.properties.name).toBe('string'); // never undefined + } + // falls back to the block text when present, else the empty-string floor + expect(blocks[0].data.properties.name).toBe('const x = req.body;'); + expect(blocks[0].data.properties.text).toBe('const x = req.body;'); + expect(blocks[1].data.properties.name).toBe(''); + }); }); diff --git a/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts b/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts new file mode 100644 index 000000000..5a0932f15 --- /dev/null +++ b/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts @@ -0,0 +1,102 @@ +import { describe, it, expect } from 'vitest'; +import { PhaseRegistry } from '../../../src/core/ingestion/pipeline-phases/registry.js'; +import type { PipelinePhase } from '../../../src/core/ingestion/pipeline-phases/types.js'; +import { buildPhaseList } from '../../../src/core/ingestion/pipeline.js'; + +// --------------------------------------------------------------------------- +// PhaseRegistry — the issue #2080 phase-registry seam, tested in isolation +// with lightweight fake phases (no real pipeline dependencies). +// --------------------------------------------------------------------------- + +const fakePhase = (name: string): PipelinePhase => ({ + name, + deps: [], + execute: async () => ({}), +}); + +describe('PhaseRegistry', () => { + it('preserves registration order in build()', () => { + const list = new PhaseRegistry() + .register(fakePhase('a')) + .register(fakePhase('b')) + .register(fakePhase('c')) + .build({}); + expect(list.map((p) => p.name)).toEqual(['a', 'b', 'c']); + }); + + it('includes a phase with no enabledWhen predicate unconditionally', () => { + const list = new PhaseRegistry<{ flag?: boolean }>() + .register(fakePhase('always')) + .build({ flag: true }); + expect(list.map((p) => p.name)).toEqual(['always']); + }); + + it('excludes a phase whose enabledWhen returns false', () => { + const reg = new PhaseRegistry<{ skip?: boolean }>() + .register(fakePhase('core')) + .register(fakePhase('optional'), { enabledWhen: (o) => !o.skip }); + + expect(reg.build({ skip: true }).map((p) => p.name)).toEqual(['core']); + expect(reg.build({ skip: false }).map((p) => p.name)).toEqual(['core', 'optional']); + // empty (normalized) options → predicate sees a real object, phase enabled + expect(reg.build({}).map((p) => p.name)).toEqual(['core', 'optional']); + }); + + it('enabledWhen filtering does not reorder surviving phases', () => { + const list = new PhaseRegistry<{ drop?: boolean }>() + .register(fakePhase('first')) + .register(fakePhase('gated'), { enabledWhen: (o) => !o.drop }) + .register(fakePhase('last')) + .build({ drop: true }); + expect(list.map((p) => p.name)).toEqual(['first', 'last']); + }); +}); + +// --------------------------------------------------------------------------- +// buildPhaseList parity — the registry refactor must produce a phase list +// byte-identical (names + order) to the legacy hand-maintained array for +// every options combination. This is the U6 characterization gate (R7/R8). +// +// Note: the second `skipGraphPhases` guard in runPipelineFromRepo (the +// result-extraction path) is intentionally NOT routed through the registry +// (KTD5); it remains keyed on the same option, so membership here stays +// consistent with output consumption there. +// --------------------------------------------------------------------------- + +const FULL_ORDER = [ + 'scan', + 'structure', + 'markdown', + 'cobol', + 'parse', + 'routes', + 'tools', + 'orm', + 'crossFile', + 'scopeResolution', + 'pruneLocalSymbols', + 'mro', + 'communities', + 'processes', +]; + +const WITHOUT_GRAPH_PHASES = FULL_ORDER.filter( + (n) => n !== 'mro' && n !== 'communities' && n !== 'processes', +); + +describe('buildPhaseList parity (registry refactor, #2080)', () => { + it('default options → full phase list in legacy order', () => { + expect(buildPhaseList(undefined).map((p) => p.name)).toEqual(FULL_ORDER); + expect(buildPhaseList({}).map((p) => p.name)).toEqual(FULL_ORDER); + }); + + it('skipGraphPhases:false → full phase list (graph phases included)', () => { + expect(buildPhaseList({ skipGraphPhases: false }).map((p) => p.name)).toEqual(FULL_ORDER); + }); + + it('skipGraphPhases:true → omits exactly mro/communities/processes', () => { + expect(buildPhaseList({ skipGraphPhases: true }).map((p) => p.name)).toEqual( + WITHOUT_GRAPH_PHASES, + ); + }); +}); diff --git a/gitnexus/test/unit/model/registration-table.test.ts b/gitnexus/test/unit/model/registration-table.test.ts index fe37d0bbb..e16ed152d 100644 --- a/gitnexus/test/unit/model/registration-table.test.ts +++ b/gitnexus/test/unit/model/registration-table.test.ts @@ -9,6 +9,11 @@ import { createTypeRegistry } from '../../../src/core/ingestion/model/type-regis import { createMethodRegistry } from '../../../src/core/ingestion/model/method-registry.js'; import { createFieldRegistry } from '../../../src/core/ingestion/model/field-registry.js'; import { ALL_NODE_LABELS } from '../../../src/core/ingestion/model/index.js'; +import { + CLASS_TYPES_TUPLE, + FREE_CALLABLE_TUPLE, +} from '../../../src/core/ingestion/model/symbol-table.js'; +import { EMBEDDABLE_LABELS } from '../../../src/core/embeddings/types.js'; import type { SymbolDefinition } from 'gitnexus-shared'; import { makeDef as makeBaseDef } from './helpers.js'; @@ -101,6 +106,30 @@ describe('NodeLabel taxonomy coverage', () => { }); }); +// --------------------------------------------------------------------------- +// BasicBlock — taint/PDG substrate node (issue #2080). It is a control-flow +// node, never a symbol-resolution or embedding target (KTD4). These guards +// fail if a future change accidentally promotes it into a dispatch/callable +// tuple or the embeddable set. +// --------------------------------------------------------------------------- + +describe('BasicBlock taint/PDG substrate label (issue #2080)', () => { + it('is classified inert — not a dispatch or callable resolution target', () => { + expect(INERT_LABELS.has('BasicBlock')).toBe(true); + expect(DISPATCH_LABELS.has('BasicBlock')).toBe(false); + expect(CALLABLE_ONLY_LABELS.has('BasicBlock')).toBe(false); + }); + + it('is excluded from the class-like and free-callable tuples', () => { + expect((CLASS_TYPES_TUPLE as readonly string[]).includes('BasicBlock')).toBe(false); + expect((FREE_CALLABLE_TUPLE as readonly string[]).includes('BasicBlock')).toBe(false); + }); + + it('is not embeddable', () => { + expect((EMBEDDABLE_LABELS as readonly string[]).includes('BasicBlock')).toBe(false); + }); +}); + // --------------------------------------------------------------------------- // Behavior group coverage — every label in a behavior group routes to the // group's registry write, regardless of how hooks are implemented (shared diff --git a/gitnexus/test/unit/schema.test.ts b/gitnexus/test/unit/schema.test.ts index 17db9bdf9..33f18ed01 100644 --- a/gitnexus/test/unit/schema.test.ts +++ b/gitnexus/test/unit/schema.test.ts @@ -17,6 +17,7 @@ import { CODE_ELEMENT_SCHEMA, COMMUNITY_SCHEMA, PROCESS_SCHEMA, + BASICBLOCK_SCHEMA, RELATION_SCHEMA, EMBEDDING_SCHEMA, CREATE_VECTOR_INDEX_QUERY, @@ -68,9 +69,13 @@ describe('LadybugDB Schema', () => { } }); + it('includes the BasicBlock taint/PDG substrate node (issue #2080)', () => { + expect(NODE_TABLES).toContain('BasicBlock'); + }); + it('has expected total count', () => { - // 9 core + 19 multi-language + Route + Tool = 31 - expect(NODE_TABLES).toHaveLength(31); + // 9 core + 19 multi-language + Route + Tool + BasicBlock = 32 + expect(NODE_TABLES).toHaveLength(32); }); }); @@ -90,6 +95,12 @@ describe('LadybugDB Schema', () => { expect(REL_TYPES).toContain(t); } }); + + it('includes the taint/PDG substrate edge types (issue #2080)', () => { + for (const t of ['CFG', 'REACHING_DEF', 'TAINTED', 'SANITIZES', 'TAINT_PATH']) { + expect(REL_TYPES).toContain(t); + } + }); }); describe('node schema DDL', () => { @@ -123,6 +134,19 @@ describe('LadybugDB Schema', () => { expect(PROPERTY_SCHEMA).toContain('declaredType STRING'); }); + it('BasicBlock schema is wired into SCHEMA_QUERIES (issue #2080, F1 guard)', () => { + // Defining BASICBLOCK_SCHEMA is not enough — it must be appended to + // NODE_SCHEMA_QUERIES (→ SCHEMA_QUERIES) or initLbug never creates the + // table and the bulk-COPY round-trip fails with "table does not exist". + expect(SCHEMA_QUERIES).toContain(BASICBLOCK_SCHEMA); + expect(BASICBLOCK_SCHEMA).toContain('CREATE NODE TABLE BasicBlock'); + expect(BASICBLOCK_SCHEMA).toContain('filePath STRING'); + expect(BASICBLOCK_SCHEMA).toContain('startLine INT64'); + expect(BASICBLOCK_SCHEMA).toContain('endLine INT64'); + expect(BASICBLOCK_SCHEMA).toContain('text STRING'); + expect(BASICBLOCK_SCHEMA).toContain('PRIMARY KEY (id)'); + }); + it('Community schema has heuristicLabel and cohesion', () => { expect(COMMUNITY_SCHEMA).toContain('heuristicLabel STRING'); expect(COMMUNITY_SCHEMA).toContain('cohesion DOUBLE'); @@ -164,6 +188,10 @@ describe('LadybugDB Schema', () => { expect(RELATION_SCHEMA).toContain('FROM Method TO Process'); }); + it('connects BasicBlock to BasicBlock (taint/PDG substrate edges, #2080)', () => { + expect(RELATION_SCHEMA).toContain('FROM BasicBlock TO BasicBlock'); + }); + it('has all FROM/TO pairs needed for HAS_METHOD edges', () => { // HAS_METHOD sources: Class, Interface, Struct, Trait, Impl, Record // HAS_METHOD targets: Method, Constructor (Property is now HAS_PROPERTY) @@ -208,7 +236,8 @@ describe('LadybugDB Schema', () => { describe('schema query ordering', () => { it('NODE_SCHEMA_QUERIES has correct count', () => { - expect(NODE_SCHEMA_QUERIES).toHaveLength(31); + // 31 + BasicBlock = 32 + expect(NODE_SCHEMA_QUERIES).toHaveLength(32); }); it('REL_SCHEMA_QUERIES has one relation table', () => { @@ -216,8 +245,8 @@ describe('LadybugDB Schema', () => { }); it('SCHEMA_QUERIES includes all node + rel + embedding schemas', () => { - // 31 node + 1 rel + 1 embedding = 33 - expect(SCHEMA_QUERIES).toHaveLength(33); + // 32 node + 1 rel + 1 embedding = 34 + expect(SCHEMA_QUERIES).toHaveLength(34); }); it('node schemas come before relation schemas in SCHEMA_QUERIES', () => { diff --git a/gitnexus/test/unit/taint/source-sink-registry.test.ts b/gitnexus/test/unit/taint/source-sink-registry.test.ts new file mode 100644 index 000000000..43b85d970 --- /dev/null +++ b/gitnexus/test/unit/taint/source-sink-registry.test.ts @@ -0,0 +1,45 @@ +import { describe, it, expect, beforeEach } from 'vitest'; +import { + registerSourceSinkConfig, + getSourceSinkConfig, + registeredTaintLanguages, + clearSourceSinkRegistry, +} from '../../../src/core/ingestion/taint/source-sink-registry.js'; +import type { SourceSinkSanitizerSpec } from '../../../src/core/ingestion/taint/source-sink-config.js'; + +const spec = (over: Partial = {}): SourceSinkSanitizerSpec => ({ + sources: [], + sinks: [], + sanitizers: [], + ...over, +}); + +describe('source/sink/sanitizer registry seam (#2080)', () => { + beforeEach(() => clearSourceSinkRegistry()); + + it('is empty by default — no language registered (guards default-run parity)', () => { + expect(registeredTaintLanguages()).toEqual([]); + expect(getSourceSinkConfig('typescript')).toBeUndefined(); + }); + + it('register then get round-trips the spec', () => { + const ts = spec({ sinks: [{ name: 'eval' }], sources: [{ name: 'req.body', args: [0] }] }); + registerSourceSinkConfig('typescript', ts); + expect(getSourceSinkConfig('typescript')).toBe(ts); + expect(registeredTaintLanguages()).toEqual(['typescript']); + }); + + it('getSourceSinkConfig returns undefined for an unregistered language (never throws)', () => { + registerSourceSinkConfig('typescript', spec()); + expect(getSourceSinkConfig('python')).toBeUndefined(); + }); + + it('re-registering the same language id overwrites (last-write-wins)', () => { + const first = spec({ sinks: [{ name: 'eval' }] }); + const second = spec({ sinks: [{ name: 'exec' }] }); + registerSourceSinkConfig('typescript', first); + registerSourceSinkConfig('typescript', second); + expect(getSourceSinkConfig('typescript')).toBe(second); + expect(registeredTaintLanguages()).toEqual(['typescript']); + }); +}); From 4de4d205dd67eea70130fc31f09168bfbd90d5b6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 9 Jun 2026 04:58:01 +0100 Subject: [PATCH 15/17] fix(ingestion): lazy-load optional grammars so analyze never crashes when one is missing (#2091, #2093) (#2101) --- gitnexus/src/cli/analyze.ts | 12 +- gitnexus/src/cli/optional-grammars.ts | 78 ++++++++-- .../core/ingestion/languages/dart/query.ts | 19 ++- .../core/ingestion/languages/kotlin/query.ts | 19 ++- .../core/ingestion/languages/swift/query.ts | 19 ++- .../ingestion/pipeline-phases/parse-impl.ts | 23 ++- .../scope-resolution/pipeline/phase.ts | 10 ++ .../core/ingestion/workers/parse-worker.ts | 13 ++ .../src/core/tree-sitter/parser-loader.ts | 104 +++++++++++++ .../registry-import-closure.test.ts | 130 ++++++++++++++++ .../skip-optional-pipeline.test.ts | 110 ++++++++++++++ .../unit/parser-loader-skip-optional.test.ts | 141 ++++++++++++++++++ 12 files changed, 648 insertions(+), 30 deletions(-) create mode 100644 gitnexus/test/integration/optional-grammars/registry-import-closure.test.ts create mode 100644 gitnexus/test/integration/optional-grammars/skip-optional-pipeline.test.ts create mode 100644 gitnexus/test/unit/parser-loader-skip-optional.test.ts diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index 3f196b11d..a3bf6cd5a 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -37,7 +37,7 @@ import { } from './analyze-config.js'; import { runFullAnalysis } from '../core/run-analyze.js'; import { getMaxFileSizeBannerMessage } from '../core/ingestion/utils/max-file-size.js'; -import { warnMissingOptionalGrammars } from './optional-grammars.js'; +import { warnMissingOptionalGrammars, getOptionalGrammarExtensions } from './optional-grammars.js'; import { glob } from 'glob'; import fs from 'fs/promises'; import { cliError } from './cli-message.js'; @@ -943,11 +943,13 @@ const analyzeCommandImpl = async ( } // If the target repo contains files an optional grammar would parse but - // that grammar's native binding is absent, warn before analysis so users - // learn why those files end up unparsed instead of silently getting a - // degraded index. + // that grammar's native binding is absent (or disabled via + // GITNEXUS_SKIP_OPTIONAL_GRAMMARS), warn before analysis so users learn why + // those files end up unparsed instead of silently getting a degraded index. + // The extension set is derived from OPTIONAL_GRAMMARS so it can't drift. try { - const matches = await glob(['**/*.dart', '**/*.proto'], { + const optionalGlobs = getOptionalGrammarExtensions().map((e) => `**/*${e}`); + const matches = await glob(optionalGlobs, { cwd: repoPath, ignore: ['**/node_modules/**', '**/.git/**', '**/dist/**', '**/build/**'], dot: false, diff --git a/gitnexus/src/cli/optional-grammars.ts b/gitnexus/src/cli/optional-grammars.ts index e12b471e0..994016a56 100644 --- a/gitnexus/src/cli/optional-grammars.ts +++ b/gitnexus/src/cli/optional-grammars.ts @@ -4,18 +4,22 @@ * tree-sitter-dart, tree-sitter-proto, and tree-sitter-swift are vendored * under vendor/ and materialized into node_modules/ at postinstall. Dart * and Proto are built from source with node-gyp; Swift ships platform - * prebuilds activated via node-gyp-build. All three can be skipped via + * prebuilds activated via node-gyp-build. tree-sitter-kotlin is a declared + * optionalDependency (not vendored). All can be skipped via * GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1 (postinstall scripts), or can silently - * soft-fail when the toolchain is missing (Dart/Proto) or no prebuild - * matches the host platform (Swift). + * soft-fail when the toolchain is missing (Dart/Proto), when no prebuild + * matches the host platform (Swift), or when the optional install was + * skipped or its native build failed (Kotlin). * * Either path produces the same observable: the .node binding is absent * at runtime. This helper detects that condition and surfaces a single - * stderr line per missing grammar so users learn why .dart/.proto/.swift + * stderr line per missing grammar so users learn why .dart/.proto/.swift/.kt * support is unavailable instead of silently getting a degraded index. */ import { createRequire } from 'module'; +import { SupportedLanguages } from 'gitnexus-shared'; +import { isGrammarRuntimeSkipped } from '../core/tree-sitter/parser-loader.js'; import { cliWarn } from './cli-message.js'; const _require = createRequire(import.meta.url); @@ -27,17 +31,55 @@ interface OptionalGrammar { pkg: string; /** File extensions this grammar parses */ extensions: string[]; + /** + * SupportedLanguages id, when this grammar backs an ingestion language. + * Used to ask `isGrammarRuntimeSkipped` whether the grammar was disabled via + * `GITNEXUS_SKIP_OPTIONAL_GRAMMARS` (vs. genuinely missing). Omitted for + * `.proto`, which is a gRPC-extractor concern, not a SupportedLanguages. + */ + language?: SupportedLanguages; } const OPTIONAL_GRAMMARS: OptionalGrammar[] = [ - { name: 'tree-sitter-dart', pkg: 'tree-sitter-dart', extensions: ['.dart'] }, + { + name: 'tree-sitter-dart', + pkg: 'tree-sitter-dart', + extensions: ['.dart'], + language: SupportedLanguages.Dart, + }, { name: 'tree-sitter-proto', pkg: 'tree-sitter-proto', extensions: ['.proto'] }, - { name: 'tree-sitter-swift', pkg: 'tree-sitter-swift', extensions: ['.swift'] }, + { + name: 'tree-sitter-swift', + pkg: 'tree-sitter-swift', + extensions: ['.swift'], + language: SupportedLanguages.Swift, + }, + { + name: 'tree-sitter-kotlin', + pkg: 'tree-sitter-kotlin', + extensions: ['.kt', '.kts'], + language: SupportedLanguages.Kotlin, + }, ]; +/** + * The file extensions backed by an optional grammar — the single source for + * the `analyze` preflight glob (so the glob can't drift from this list). + */ +export function getOptionalGrammarExtensions(): string[] { + return [...new Set(OPTIONAL_GRAMMARS.flatMap((g) => g.extensions))]; +} + export interface MissingGrammar { name: string; extensions: string[]; + /** + * `missing` — the native binding could not be loaded (not installed / build + * soft-failed / no prebuild). `skipped` — the binding is fine but the user + * disabled it via `GITNEXUS_SKIP_OPTIONAL_GRAMMARS`. Drives the warning text + * so a deliberate opt-out is not told to reinstall. + */ + reason: 'missing' | 'skipped'; } /** @@ -59,6 +101,13 @@ export interface MissingGrammar { export function detectMissingOptionalGrammars(): MissingGrammar[] { const missing: MissingGrammar[] = []; for (const g of OPTIONAL_GRAMMARS) { + // Deliberate runtime opt-out comes first: even an installed binding is + // treated as unavailable, with a `skipped` reason so the warning says so + // instead of suggesting a reinstall (#2101 review). + if (g.language !== undefined && isGrammarRuntimeSkipped(g.language)) { + missing.push({ name: g.name, extensions: g.extensions, reason: 'skipped' }); + continue; + } try { _require(g.pkg); } catch (err) { @@ -80,7 +129,7 @@ export function detectMissingOptionalGrammars(): MissingGrammar[] { { grammar: g.name, extensions: g.extensions, error: msg }, ); } - missing.push({ name: g.name, extensions: g.extensions }); + missing.push({ name: g.name, extensions: g.extensions, reason: 'missing' }); } } return missing; @@ -110,9 +159,16 @@ export function warnMissingOptionalGrammars(opts?: { if (relevantExtensions && !g.extensions.some((e) => relevantExtensions.has(e))) { continue; } - cliWarn( - `GitNexus${ctx}: optional grammar "${g.name}" is unavailable — ${g.extensions.join('/')} files will not be parsed. Reinstall without GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1 (and ensure python3, make, g++) to enable.`, - { grammar: g.name, extensions: g.extensions, context: opts?.context }, - ); + const exts = g.extensions.join('/'); + const message = + g.reason === 'skipped' + ? `GitNexus${ctx}: optional grammar "${g.name}" is disabled via GITNEXUS_SKIP_OPTIONAL_GRAMMARS — ${exts} files will not be parsed. Unset the variable to re-enable.` + : `GitNexus${ctx}: optional grammar "${g.name}" is unavailable — ${exts} files will not be parsed. Reinstall without GITNEXUS_SKIP_OPTIONAL_GRAMMARS=1 (and ensure python3, make, g++) to enable.`; + cliWarn(message, { + grammar: g.name, + extensions: g.extensions, + reason: g.reason, + context: opts?.context, + }); } } diff --git a/gitnexus/src/core/ingestion/languages/dart/query.ts b/gitnexus/src/core/ingestion/languages/dart/query.ts index 7d66b1c4a..5314b0c8b 100644 --- a/gitnexus/src/core/ingestion/languages/dart/query.ts +++ b/gitnexus/src/core/ingestion/languages/dart/query.ts @@ -23,7 +23,15 @@ */ import Parser from 'tree-sitter'; -import Dart from 'tree-sitter-dart'; +import { SupportedLanguages } from 'gitnexus-shared'; +// `tree-sitter-dart` is an optional/vendored grammar that may be absent on a +// default install. Loaded lazily + guarded via parser-loader rather than +// statically imported: this module is pulled onto the main thread eagerly by +// the scope-resolution registry and the language-provider index, so a top-level +// `import Dart from 'tree-sitter-dart'` would throw ERR_MODULE_NOT_FOUND at +// module-load and crash `analyze` even for repos with no Dart files (#2091, +// #2093). The grammar is only ever needed inside the lazy getters below. +import { getLanguageGrammar } from '../../../tree-sitter/parser-loader.js'; const DART_SCOPE_QUERY = ` ; ── Scopes ─────────────────────────────────────────────────────────────────── @@ -134,14 +142,19 @@ let _query: Parser.Query | null = null; export function getDartParser(): Parser { if (_parser === null) { _parser = new Parser(); - _parser.setLanguage(Dart as Parameters[0]); + _parser.setLanguage( + getLanguageGrammar(SupportedLanguages.Dart) as Parameters[0], + ); } return _parser; } export function getDartScopeQuery(): Parser.Query { if (_query === null) { - _query = new Parser.Query(Dart as Parameters[0], DART_SCOPE_QUERY); + _query = new Parser.Query( + getLanguageGrammar(SupportedLanguages.Dart) as Parameters[0], + DART_SCOPE_QUERY, + ); } return _query; } diff --git a/gitnexus/src/core/ingestion/languages/kotlin/query.ts b/gitnexus/src/core/ingestion/languages/kotlin/query.ts index 31613a38e..f3a749086 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/query.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/query.ts @@ -1,5 +1,13 @@ import Parser from 'tree-sitter'; -import Kotlin from 'tree-sitter-kotlin'; +import { SupportedLanguages } from 'gitnexus-shared'; +// `tree-sitter-kotlin` is an optionalDependency that may be absent on a default +// install (or fail its native build). Loaded lazily + guarded via parser-loader +// rather than statically imported: this module is pulled onto the main thread +// eagerly by the scope-resolution registry and the language-provider index, so +// a top-level `import Kotlin from 'tree-sitter-kotlin'` would throw +// ERR_MODULE_NOT_FOUND at module-load and crash `analyze` even for repos with no +// Kotlin files (#2091, #2093). The grammar is only ever needed in the getters. +import { getLanguageGrammar } from '../../../tree-sitter/parser-loader.js'; const KOTLIN_SCOPE_QUERY = ` ;; Scopes @@ -179,14 +187,19 @@ let query: Parser.Query | null = null; export function getKotlinParser(): Parser { if (parser === null) { parser = new Parser(); - parser.setLanguage(Kotlin as Parameters[0]); + parser.setLanguage( + getLanguageGrammar(SupportedLanguages.Kotlin) as Parameters[0], + ); } return parser; } export function getKotlinScopeQuery(): Parser.Query { if (query === null) { - query = new Parser.Query(Kotlin as Parameters[0], KOTLIN_SCOPE_QUERY); + query = new Parser.Query( + getLanguageGrammar(SupportedLanguages.Kotlin) as Parameters[0], + KOTLIN_SCOPE_QUERY, + ); } return query; } diff --git a/gitnexus/src/core/ingestion/languages/swift/query.ts b/gitnexus/src/core/ingestion/languages/swift/query.ts index cf2699423..78c1efa41 100644 --- a/gitnexus/src/core/ingestion/languages/swift/query.ts +++ b/gitnexus/src/core/ingestion/languages/swift/query.ts @@ -43,7 +43,15 @@ */ import Parser from 'tree-sitter'; -import Swift from 'tree-sitter-swift'; +import { SupportedLanguages } from 'gitnexus-shared'; +// `tree-sitter-swift` is an optional/vendored grammar that may be absent on a +// default install. It is loaded lazily + guarded via parser-loader rather than +// statically imported: this module is pulled onto the main thread eagerly by +// the scope-resolution registry and the language-provider index, so a top-level +// `import Swift from 'tree-sitter-swift'` would throw ERR_MODULE_NOT_FOUND at +// module-load and crash `analyze` even for repos with no Swift files (#2091, +// #2093). The grammar is only ever needed inside the lazy getters below. +import { getLanguageGrammar } from '../../../tree-sitter/parser-loader.js'; const SWIFT_SCOPE_QUERY = ` ;; ── Scopes ────────────────────────────────────────────────────────── @@ -186,14 +194,19 @@ let _query: Parser.Query | null = null; export function getSwiftParser(): Parser { if (_parser === null) { _parser = new Parser(); - _parser.setLanguage(Swift as Parameters[0]); + _parser.setLanguage( + getLanguageGrammar(SupportedLanguages.Swift) as Parameters[0], + ); } return _parser; } export function getSwiftScopeQuery(): Parser.Query { if (_query === null) { - _query = new Parser.Query(Swift as Parameters[0], SWIFT_SCOPE_QUERY); + _query = new Parser.Query( + getLanguageGrammar(SupportedLanguages.Swift) as Parameters[0], + SWIFT_SCOPE_QUERY, + ); } return _query; } diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts index 244f178dd..ddc8c902e 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts @@ -43,9 +43,13 @@ import { type ExportedTypeMap, } from '../call-processor.js'; import { createSemanticModel, type MutableSemanticModel } from '../model/index.js'; -import { type PipelineProgress, getLanguageFromFilename } from 'gitnexus-shared'; +import { + type PipelineProgress, + getLanguageFromFilename, + SupportedLanguages, +} from 'gitnexus-shared'; import { readFileContents } from '../filesystem-walker.js'; -import { isLanguageAvailable } from '../../tree-sitter/parser-loader.js'; +import { isLanguageAvailable, isGrammarRuntimeSkipped } from '../../tree-sitter/parser-loader.js'; import { createWorkerPool, workerPoolDisabledByEnv, @@ -274,9 +278,18 @@ export async function runChunkedParseAndResolve( } } for (const [lang, count] of skippedByLang) { - logger.warn( - `Skipping ${count} ${lang} file(s) — ${lang} parser not available (native binding may not have built). Try: npm rebuild tree-sitter-${lang}`, - ); + // Distinguish a deliberate runtime opt-out from a genuinely-missing binding + // so we don't tell a user who set GITNEXUS_SKIP_OPTIONAL_GRAMMARS to + // `npm rebuild` a grammar that built fine (#2091/#2093 review). + if (isGrammarRuntimeSkipped(lang as SupportedLanguages)) { + logger.warn( + `Skipping ${count} ${lang} file(s) — ${lang} parsing disabled via GITNEXUS_SKIP_OPTIONAL_GRAMMARS.`, + ); + } else { + logger.warn( + `Skipping ${count} ${lang} file(s) — ${lang} parser not available (native binding may not have built). Try: npm rebuild tree-sitter-${lang}`, + ); + } } // Sort parseableScanned alphabetically for stable chunk membership diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts index 3c3f688ec..d686bdd43 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts @@ -31,6 +31,7 @@ import type { ParseOutput } from '../../pipeline-phases/parse.js'; import { SupportedLanguages, getLanguageFromFilename } from 'gitnexus-shared'; import { readFileContents } from '../../filesystem-walker.js'; import { runScopeResolution, type ScopeResolutionSubPhase } from './run.js'; +import { isLanguageAvailable } from '../../../tree-sitter/parser-loader.js'; import { buildGraphNodeLookup } from '../graph-bridge/node-lookup.js'; import { SCOPE_RESOLVERS } from './registry.js'; import { isDev, isSemanticModelValidatorEnabled } from '../../utils/env.js'; @@ -170,6 +171,15 @@ export const scopeResolutionPhase: PipelinePhase = { for (const f of scannedFiles) { const fileLang = getLanguageFromFilename(f.path); if (fileLang === null) continue; + // Skip files whose grammar isn't available (optional grammars like + // swift/dart/kotlin on an install where the binding is absent or the + // user set GITNEXUS_SKIP_OPTIONAL_GRAMMARS). The parse phase already + // excluded and warned about these (parse-impl.ts); without this guard the + // file would fall through to the main-thread re-extract in run.ts and + // throw "Unsupported language" (caught, but noisy, and it needlessly + // loads the grammar on the main thread). `isLanguageAvailable` is + // memoized, so this stays O(1) per language. (#2091, #2093) + if (!isLanguageAvailable(fileLang)) continue; let bucket = filesByLang.get(fileLang); if (bucket === undefined) { bucket = []; diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index b0f787855..b32d95758 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -36,6 +36,19 @@ import type { /** Language grammar type accepted by Parser.setLanguage(). */ type TreeSitterLanguage = Parameters[0]; +// ── Worker grammar loading — enforcement boundary (#2091/#2093, #2101) ─────── +// The worker maintains its own grammar table (the guarded `_require`s below + +// `languageMap`) and intentionally does NOT consult the runtime +// `GITNEXUS_SKIP_OPTIONAL_GRAMMARS` opt-out. It does not need to: the MAIN +// THREAD's `parseableScanned` filter (pipeline-phases/parse-impl.ts, gated on +// `parser-loader.isLanguageAvailable`, which honors the runtime opt-out and a +// genuinely-absent binding alike) excludes files of an unavailable/opted-out +// language BEFORE any chunk is dispatched, so the worker never receives them. +// That main-thread filter is the single enforcement point. Any future change +// that dispatches files to the worker WITHOUT first passing them through +// `isLanguageAvailable` must re-introduce the gate here. (The cleaner end-state +// — routing this table through `parser-loader.getLanguageGrammar` so there is +// one loader — is the deferred Tier-1 consolidation.) // tree-sitter-swift is an optionalDependency — may not be installed const _require = createRequire(import.meta.url); let Swift: TreeSitterLanguage | null = null; diff --git a/gitnexus/src/core/tree-sitter/parser-loader.ts b/gitnexus/src/core/tree-sitter/parser-loader.ts index a51634fbe..3d378f2f0 100644 --- a/gitnexus/src/core/tree-sitter/parser-loader.ts +++ b/gitnexus/src/core/tree-sitter/parser-loader.ts @@ -39,6 +39,15 @@ interface GrammarSource { unavailableNote: string; optional?: boolean; severity?: 'warn' | 'error'; + /** + * When true, this grammar may be disabled at runtime via + * `GITNEXUS_SKIP_OPTIONAL_GRAMMARS`. Set ONLY on genuinely-optional grammars + * (optionalDependencies / vendored — swift/dart/kotlin). Required dependencies + * routed through the optional machinery for ABI safety (e.g. C, which is + * `optional: true` + `severity: 'error'`) must NOT set this — opting out of a + * required parser is always an install/platform problem, never a user choice. + */ + userSkippable?: boolean; } const ISSUES_URL = 'https://github.com/abhigyanpatwari/GitNexus/issues'; @@ -139,6 +148,7 @@ const SOURCES: Record = { [SupportedLanguages.Swift]: { load: () => _require('tree-sitter-swift'), optional: true, + userSkippable: true, unavailableNote: 'Swift parsing disabled: vendored `tree-sitter-swift` (under ' + '`gitnexus/vendor/tree-sitter-swift`) failed to load. ' + @@ -148,6 +158,7 @@ const SOURCES: Record = { [SupportedLanguages.Dart]: { load: () => _require('tree-sitter-dart'), optional: true, + userSkippable: true, unavailableNote: 'Dart parsing disabled: vendored `tree-sitter-dart` (under ' + '`gitnexus/vendor/tree-sitter-dart`) failed to load. ' + @@ -157,6 +168,7 @@ const SOURCES: Record = { [SupportedLanguages.Kotlin]: { load: () => _require('tree-sitter-kotlin'), optional: true, + userSkippable: true, unavailableNote: 'Kotlin parsing disabled: `tree-sitter-kotlin` is an optionalDependency ' + 'and is not installed (or its native binding failed to build).', @@ -189,6 +201,63 @@ type LoadResult = const loadCache = new Map(); const logged = new Set(); +/** + * Runtime opt-out for genuinely-optional grammars (Swift/Dart/Kotlin). + * + * `GITNEXUS_SKIP_OPTIONAL_GRAMMARS` has historically been an *install-time* + * env only — the postinstall build scripts read it to skip building the + * vendored grammars. There was no way to disable an optional grammar at + * analyze time, so users on a platform with a broken/partial binding had no + * escape hatch short of uninstalling the package (#2091, #2093). This honors + * the same env name at runtime: when set, the named optional grammars report + * as unavailable and the pipeline skips their files (mirroring a genuinely + * absent binding) instead of attempting to load them. + * + * Accepts `1` / `true` / `all` / `*` (every skippable grammar), or a + * comma-separated list of language ids and/or package names + * (e.g. `swift,tree-sitter-dart`). Only grammars flagged `userSkippable` (the + * genuinely-optional swift/dart/kotlin) can be skipped — required dependencies + * routed through the optional machinery for ABI safety (C) carry no + * `userSkippable` and are never skippable here. + */ +type SkipDirective = 'all' | Set | null; + +// Parsed form of GITNEXUS_SKIP_OPTIONAL_GRAMMARS, resolved lazily ONCE per +// process. The env is set before analyze runs, so re-reading + re-allocating a +// Set on every call was wasted work (and a latent trap for any future per-file +// caller). `vi.resetModules()` gives the unit tests a fresh module — and thus a +// fresh memo — per case, so this stays test-friendly. +// 'all' → every userSkippable grammar; Set → only the named ids +// (and `tree-sitter-` spellings); null → env unset/empty (nothing). +let _skipDirective: SkipDirective | undefined; +const skipDirective = (): SkipDirective => { + if (_skipDirective !== undefined) return _skipDirective; + const raw = (process.env.GITNEXUS_SKIP_OPTIONAL_GRAMMARS ?? '').trim().toLowerCase(); + if (raw === '') return (_skipDirective = null); + if (raw === '1' || raw === 'true' || raw === 'all' || raw === '*') + return (_skipDirective = 'all'); + return (_skipDirective = new Set( + raw + .split(',') + .map((s) => s.trim()) + .filter(Boolean) + .flatMap((s) => [s, s.replace(/^tree-sitter-/, '')]), + )); +}; + +const isRuntimeSkippedGrammar = (key: string, source: GrammarSource): boolean => { + // Only grammars explicitly flagged user-skippable (swift/dart/kotlin) — never + // required deps that use the optional machinery for ABI safety (C carries no + // `userSkippable`). + if (source.userSkippable !== true) return false; + const directive = skipDirective(); + if (directive === null) return false; + if (directive === 'all') return true; + // `key` is the SupportedLanguages value (e.g. `swift`); the directive Set + // already holds both the bare id and the `tree-sitter-` spelling. + return directive.has(key) || directive.has(`tree-sitter-${key}`); +}; + const logFailure = (key: string, result: LoadResult): void => { if (result.ok === true) return; if (logged.has(key)) return; @@ -227,6 +296,25 @@ const loadGrammar = (key: string): LoadResult => { return result; } + // Runtime opt-out: treat a user-skipped optional grammar exactly like an + // absent binding (non-fatal unavailable + one warning), without attempting + // the native load. See `isRuntimeSkippedGrammar`. + if (isRuntimeSkippedGrammar(key, source)) { + // Deliberate opt-out: emit an accurate "disabled on purpose" note rather + // than `source.unavailableNote` (which blames a missing/unbuilt binding and + // would mislead a user who set the env intentionally — #2101 review). + const result: LoadResult = { + ok: false, + error: new Error('runtime opt-out'), + note: `${key} parsing disabled via GITNEXUS_SKIP_OPTIONAL_GRAMMARS (unset it to re-enable).`, + fatal: false, + severity: 'warn', + }; + loadCache.set(key, result); + logFailure(key, result); + return result; + } + let result: LoadResult; try { result = { ok: true, grammar: source.load() }; @@ -248,6 +336,22 @@ const loadGrammar = (key: string): LoadResult => { export const isLanguageAvailable = (language: SupportedLanguages, filePath?: string): boolean => loadGrammar(resolveLanguageKey(language, filePath)).ok; +/** + * True when `language`'s grammar is being treated as unavailable specifically + * because of the runtime GITNEXUS_SKIP_OPTIONAL_GRAMMARS opt-out — as opposed + * to a genuinely-missing/broken native binding. Lets callers surface an + * accurate "skipped on purpose" message instead of a spurious "npm rebuild" + * recovery hint. Returns false for required grammars and for an absent env. + */ +export const isGrammarRuntimeSkipped = ( + language: SupportedLanguages, + filePath?: string, +): boolean => { + const key = resolveLanguageKey(language, filePath); + const source = SOURCES[key]; + return source !== undefined && isRuntimeSkippedGrammar(key, source); +}; + export const getLanguageGrammar = (language: SupportedLanguages, filePath?: string): unknown => { const key = resolveLanguageKey(language, filePath); const result = loadGrammar(key); diff --git a/gitnexus/test/integration/optional-grammars/registry-import-closure.test.ts b/gitnexus/test/integration/optional-grammars/registry-import-closure.test.ts new file mode 100644 index 000000000..09acea026 --- /dev/null +++ b/gitnexus/test/integration/optional-grammars/registry-import-closure.test.ts @@ -0,0 +1,130 @@ +/** + * Optional-grammar static-import-closure regression test (#2091, #2093). + * + * The scope-resolution registry (`scope-resolution/pipeline/registry.ts`) and + * the language-provider index statically import all 16 language providers. Each + * per-language `query.ts` used to do a top-level `import X from 'tree-sitter-Y'`. + * For the OPTIONAL grammars (swift/dart/kotlin) that import resolved — and on a + * default install where the vendored/optional binding is absent, THREW + * `ERR_MODULE_NOT_FOUND` — at module-load on the main thread, before any runtime + * gate, crashing `gitnexus analyze` regardless of the repo's actual languages. + * + * The fix routes those three `query.ts` modules through the lazy, guarded + * `parser-loader.getLanguageGrammar()` so the grammar binding is only required + * at first use (inside the worker, for a file of that language) — never at + * module-load. + * + * This test locks the fix in WITHOUT needing to simulate a missing grammar: + * spawn a child Node process, import the built scope-resolution `registry.js` + * (the crash-chain root), and assert no OPTIONAL tree-sitter binding + * (swift/dart/kotlin) appears in the module cache. Pre-fix the static imports + * loaded those bindings at import time (this assertion fails); post-fix they + * are lazy (it passes). Required grammars (python/typescript/...) still load + * eagerly via their own `query.ts` — that is expected and NOT asserted against. + * + * Characterization-first: this MUST fail against the pre-fix code (run against + * the parent commit to verify the regression signal works). + */ + +import { describe, it, expect } from 'vitest'; +import { spawnSync } from 'node:child_process'; +import path from 'node:path'; +import fs from 'node:fs'; +import { fileURLToPath, pathToFileURL } from 'node:url'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const REPO_ROOT = path.resolve(__dirname, '..', '..', '..'); +const DIST_REGISTRY = path.join( + REPO_ROOT, + 'dist', + 'core', + 'ingestion', + 'scope-resolution', + 'pipeline', + 'registry.js', +); +const DIST_REGISTRY_URL = pathToFileURL(DIST_REGISTRY).href; + +// Import the registry, then report every newly-loaded CJS-cache key. The cache +// tracks native/.node bindings loaded by either ESM or CJS importers, which is +// exactly how a tree-sitter grammar binding surfaces. +const PROBE = ` + import { createRequire } from 'node:module'; + const req = createRequire(import.meta.url); + const before = new Set(Object.keys(req.cache)); + await import(process.env.PROBE_TARGET); + const after = new Set(Object.keys(req.cache)); + process.stdout.write(JSON.stringify([...after].filter((k) => !before.has(k)))); +`; + +const OPTIONAL_GRAMMAR_RE = /tree-sitter-(swift|dart|kotlin)[\\/]/; + +describe('optional-grammar static-import closure (#2091/#2093)', () => { + it('importing the scope-resolution registry loads NO optional grammar binding', () => { + if (!fs.existsSync(DIST_REGISTRY)) { + throw new Error( + `${DIST_REGISTRY} missing — run \`npm run build\` first (or \`npm run test:integration\`, ` + + `which builds via pretest:integration).`, + ); + } + + const result = spawnSync(process.execPath, ['--input-type=module', '-e', PROBE], { + cwd: REPO_ROOT, + // NODE_OPTIONS cleared so a session-pinned --max-old-space-size etc. can't + // perturb the child. The skip env is cleared so install state is probed. + env: { + ...process.env, + PROBE_TARGET: DIST_REGISTRY_URL, + NODE_OPTIONS: '', + GITNEXUS_SKIP_OPTIONAL_GRAMMARS: '', + }, + timeout: 60_000, + encoding: 'utf8', + }); + + // Post-fix, importing the registry must not throw even though the chain + // reaches swift/dart/kotlin query.ts. (Pre-fix on a machine missing a + // grammar this would be ERR_MODULE_NOT_FOUND; here the grammar is present + // so pre-fix it would instead surface as a loaded binding below.) + if (result.status !== 0) { + // status is null when the child was killed by a signal (e.g. a native + // addon SIGSEGV) — surface the signal so that's distinguishable from a + // non-zero exit / module-not-found. + const exit = + result.status !== null ? `status ${result.status}` : `signal ${result.signal ?? 'unknown'}`; + throw new Error( + `importing the scope-resolution registry failed (${exit}):\n` + + `stderr:\n${result.stderr}\nstdout:\n${result.stdout}`, + ); + } + + const newlyLoaded = JSON.parse(result.stdout) as string[]; + + // Non-vacuity guard: the registry's static-import closure MUST still reach + // the per-language query.ts modules (which is what makes "no optional + // binding loaded" meaningful). The REQUIRED grammars (python/typescript/…) + // still import their binding eagerly in their own query.ts, so at least one + // non-optional tree-sitter binding must appear. If a future refactor severs + // the registry→query.ts edge, this fails loudly instead of letting the + // optional-binding assertion pass green on a no-longer-exercised path. + const requiredLoaded = newlyLoaded.filter( + (p) => /tree-sitter-[a-z-]+[\\/]/.test(p) && !OPTIONAL_GRAMMAR_RE.test(p), + ); + expect( + requiredLoaded.length, + `Expected the registry import closure to load at least one REQUIRED tree-sitter ` + + `binding (proving the chain still reaches the per-language query.ts modules). ` + + `Newly-loaded (${newlyLoaded.length}):\n${newlyLoaded.join('\n')}`, + ).toBeGreaterThan(0); + + // Headline assertion: no OPTIONAL grammar binding (swift/dart/kotlin) is + // loaded at registry static-import time — they must load lazily. + const optionalLoaded = newlyLoaded.filter((p) => OPTIONAL_GRAMMAR_RE.test(p)); + expect( + optionalLoaded, + `Optional tree-sitter grammar binding(s) loaded at registry static-import time. ` + + `query.ts must load swift/dart/kotlin lazily via parser-loader, not via a ` + + `top-level \`import\`. Offending paths:\n${optionalLoaded.join('\n')}`, + ).toEqual([]); + }); +}); diff --git a/gitnexus/test/integration/optional-grammars/skip-optional-pipeline.test.ts b/gitnexus/test/integration/optional-grammars/skip-optional-pipeline.test.ts new file mode 100644 index 000000000..d60fca427 --- /dev/null +++ b/gitnexus/test/integration/optional-grammars/skip-optional-pipeline.test.ts @@ -0,0 +1,110 @@ +/** + * Pipeline-level regression for the optional-grammar exclusion (#2091, #2093). + * + * Locks the scope-resolution phase guard added alongside the lazy query.ts + * load: `scopeResolutionPhase` filters its `filesByLang` partition by + * `isLanguageAvailable`, so a file of an unavailable optional grammar never + * falls through to the main-thread re-extract in `run.ts` (which would throw + * "Unsupported language" — caught, but noisy, and it needlessly loads the + * grammar on the main thread). + * + * Drives the REAL pipeline over a mixed Python+Swift repo with the runtime + * `GITNEXUS_SKIP_OPTIONAL_GRAMMARS` opt-out set (so Swift is treated as + * unavailable even though its binding is installed). This is the automated + * analog of the manual end-to-end verification: Python indexes, Swift is + * cleanly skipped, and the "scope extraction failed for …swift" noise never + * appears. + * + * `parser-loader` memoizes availability per process, so we `vi.resetModules()` + * BEFORE setting the env and dynamically import the pipeline + logger from the + * same fresh registry. That makes the first `isLanguageAvailable` call observe + * our env regardless of import order, and keeps the logger capture wired to the + * loader's logger instance (a static import would not survive resetModules). + */ + +import { describe, it, expect, beforeAll, afterAll, vi } from 'vitest'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import type { PipelineResult } from '../resolvers/helpers.js'; + +const ENV = 'GITNEXUS_SKIP_OPTIONAL_GRAMMARS'; + +describe('optional-grammar pipeline exclusion (#2091/#2093)', () => { + let repoDir = ''; + let result: PipelineResult; + let messages: string[] = []; + let prevEnv: string | undefined; + let getNodesByLabel: (r: PipelineResult, label: string) => string[]; + + beforeAll(async () => { + prevEnv = process.env[ENV]; + vi.resetModules(); + process.env[ENV] = 'swift'; + const helpers = await import('../resolvers/helpers.js'); + const loggerMod = await import('../../../src/core/logger.js'); + getNodesByLabel = helpers.getNodesByLabel; + + repoDir = fs.mkdtempSync(path.join(os.tmpdir(), 'og-skip-pipeline-')); + fs.writeFileSync( + path.join(repoDir, 'app.py'), + 'def greet(name):\n return f"hi {name}"\n\n\nclass Service:\n def run(self):\n return greet("world")\n', + ); + fs.writeFileSync( + path.join(repoDir, 'Foo.swift'), + 'struct Foo {\n func bar() -> Int { return 42 }\n}\n', + ); + + const cap = loggerMod._captureLogger(); + try { + result = await helpers.runPipelineFromRepo(repoDir, () => {}, { skipGraphPhases: true }); + messages = cap + .records() + .map((r) => (typeof r.msg === 'string' ? r.msg : '')) + .filter(Boolean); + } finally { + cap.restore(); + } + }, 60_000); + + afterAll(() => { + if (prevEnv === undefined) delete process.env[ENV]; + else process.env[ENV] = prevEnv; + if (repoDir) fs.rmSync(repoDir, { recursive: true, force: true }); + }); + + it('completes without crashing when an optional grammar is opted out', () => { + expect(result).toBeDefined(); + }); + + it('skips the Swift file at the parse phase (non-vacuity: Swift was present)', () => { + expect(messages.some((m) => /Skipping 1 swift file\(s\)/.test(m))).toBe(true); + }); + + it('routes the opt-out message, not the missing-binding "npm rebuild" hint', () => { + // The "Skipping N swift file(s)" prefix is shared by BOTH the opt-out and + // the missing-binding branches — so assert the opt-out branch specifically: + // a message naming the env var, and NO "npm rebuild" hint anywhere. This is + // what proves the isGrammarRuntimeSkipped routing in parse-impl.ts fired. + expect(messages.some((m) => /GITNEXUS_SKIP_OPTIONAL_GRAMMARS/.test(m))).toBe(true); + expect( + messages.every((m) => !/npm rebuild/i.test(m)), + messages.join('\n'), + ).toBe(true); + }); + + it('never falls through to the main-thread re-extract (no "scope extraction failed")', () => { + // This is the precise signal the scope-resolution phase guard eliminates. + // Without the `if (!isLanguageAvailable(fileLang)) continue;` in phase.ts + // the Swift file would reach run.ts's extractParsedFile and log this. + const offending = messages.filter((m) => /scope extraction failed/i.test(m)); + expect(offending, offending.join('\n')).toEqual([]); + }); + + it('indexes the available Python language and excludes Swift symbols', () => { + // Python indexed (proves the pipeline actually ran end-to-end). + expect(getNodesByLabel(result, 'Class')).toContain('Service'); + // Swift's struct must not be in the graph — it was excluded, not parsed. + expect(getNodesByLabel(result, 'Struct')).not.toContain('Foo'); + }); +}); diff --git a/gitnexus/test/unit/parser-loader-skip-optional.test.ts b/gitnexus/test/unit/parser-loader-skip-optional.test.ts new file mode 100644 index 000000000..ae8cff396 --- /dev/null +++ b/gitnexus/test/unit/parser-loader-skip-optional.test.ts @@ -0,0 +1,141 @@ +import { describe, it, expect, afterEach, vi } from 'vitest'; +import { SupportedLanguages } from '../../src/config/supported-languages.js'; + +/** + * Runtime opt-out for optional grammars (#2091, #2093). + * + * `GITNEXUS_SKIP_OPTIONAL_GRAMMARS` used to be an install-time-only env (the + * postinstall build scripts read it). `parser-loader` now also honors it at + * analyze time: when set, genuinely-optional grammars (swift/dart/kotlin) + * report unavailable so the ingestion pipeline skips their files, mirroring a + * genuinely-absent binding. Grammars that are required `dependencies` routed + * through the optional machinery for ABI safety (C — `severity: 'error'`) are + * NEVER skippable this way. + * + * `parser-loader` memoizes load results at module scope, so each case loads a + * fresh copy via `vi.resetModules()` after setting the env. These assertions + * are install-state-robust: they only assert the SKIP direction (skip → false) + * and that required grammars are unaffected (true) — never that an optional + * grammar is positively available, which depends on the install/platform. + */ + +const ENV = 'GITNEXUS_SKIP_OPTIONAL_GRAMMARS'; + +async function freshLoader(skipValue: string | undefined) { + vi.resetModules(); + if (skipValue === undefined) delete process.env[ENV]; + else process.env[ENV] = skipValue; + return import('../../src/core/tree-sitter/parser-loader.js'); +} + +afterEach(() => { + delete process.env[ENV]; + vi.resetModules(); +}); + +describe('parser-loader GITNEXUS_SKIP_OPTIONAL_GRAMMARS runtime gate', () => { + it('skip=1 reports every optional grammar as unavailable', async () => { + const { isLanguageAvailable } = await freshLoader('1'); + expect(isLanguageAvailable(SupportedLanguages.Swift)).toBe(false); + expect(isLanguageAvailable(SupportedLanguages.Dart)).toBe(false); + expect(isLanguageAvailable(SupportedLanguages.Kotlin)).toBe(false); + }); + + it('skip=all/true/* also skip every optional grammar', async () => { + for (const v of ['all', 'true', '*']) { + const { isLanguageAvailable } = await freshLoader(v); + expect(isLanguageAvailable(SupportedLanguages.Swift), `value=${v}`).toBe(false); + expect(isLanguageAvailable(SupportedLanguages.Dart), `value=${v}`).toBe(false); + expect(isLanguageAvailable(SupportedLanguages.Kotlin), `value=${v}`).toBe(false); + } + }); + + it('does NOT skip required grammars — skip=all is a no-op for C / Python', async () => { + // Compare availability WITH skip=all against the baseline (no skip). The + // runtime opt-out must never change a required grammar's availability: + // C is `optional: true` + `severity: 'error'` (a required dep routed + // through the optional machinery for ABI safety, #1242), and Python is a + // plain required dep. Asserting EQUALITY (not positive truth) keeps this + // install-state-robust — C's native binding is intentionally fallible, so + // a positive assertion could flake on an ABI-mismatched matrix. + const base = await freshLoader(undefined); + const cBase = base.isLanguageAvailable(SupportedLanguages.C); + const pyBase = base.isLanguageAvailable(SupportedLanguages.Python); + const skipped = await freshLoader('all'); + expect(skipped.isLanguageAvailable(SupportedLanguages.C)).toBe(cBase); + expect(skipped.isLanguageAvailable(SupportedLanguages.Python)).toBe(pyBase); + }); + + it('a comma list skips ONLY the named grammars — un-named ones unaffected', async () => { + // Baseline (no skip) so the isolation check is install-state-robust. + const base = await freshLoader(undefined); + const dartBase = base.isLanguageAvailable(SupportedLanguages.Dart); + const kotlinBase = base.isLanguageAvailable(SupportedLanguages.Kotlin); + const { isLanguageAvailable } = await freshLoader('swift'); + expect(isLanguageAvailable(SupportedLanguages.Swift)).toBe(false); + // A prefix/union bug would skip these too — assert they match baseline. + expect(isLanguageAvailable(SupportedLanguages.Dart)).toBe(dartBase); + expect(isLanguageAvailable(SupportedLanguages.Kotlin)).toBe(kotlinBase); + }); + + it('accepts the tree-sitter- package spelling — others unaffected', async () => { + const base = await freshLoader(undefined); + const swiftBase = base.isLanguageAvailable(SupportedLanguages.Swift); + const kotlinBase = base.isLanguageAvailable(SupportedLanguages.Kotlin); + const { isLanguageAvailable } = await freshLoader('tree-sitter-dart'); + expect(isLanguageAvailable(SupportedLanguages.Dart)).toBe(false); + expect(isLanguageAvailable(SupportedLanguages.Swift)).toBe(swiftBase); + expect(isLanguageAvailable(SupportedLanguages.Kotlin)).toBe(kotlinBase); + }); + + it('accepts a multi-entry list', async () => { + const { isLanguageAvailable } = await freshLoader('kotlin, dart'); + expect(isLanguageAvailable(SupportedLanguages.Kotlin)).toBe(false); + expect(isLanguageAvailable(SupportedLanguages.Dart)).toBe(false); + }); + + it('getLanguageGrammar throws a clean "Unsupported language" for a skipped optional grammar', async () => { + const { getLanguageGrammar } = await freshLoader('all'); + expect(() => getLanguageGrammar(SupportedLanguages.Swift)).toThrow(/Unsupported language/); + }); + + it('an empty / unset env does not skip (required grammars load)', async () => { + const { isLanguageAvailable } = await freshLoader(undefined); + expect(isLanguageAvailable(SupportedLanguages.Python)).toBe(true); + }); + + it('isGrammarRuntimeSkipped reflects the opt-out, never for required grammars', async () => { + const swiftOnly = await freshLoader('swift'); + expect(swiftOnly.isGrammarRuntimeSkipped(SupportedLanguages.Swift)).toBe(true); + expect(swiftOnly.isGrammarRuntimeSkipped(SupportedLanguages.Dart)).toBe(false); + const all = await freshLoader('all'); + expect(all.isGrammarRuntimeSkipped(SupportedLanguages.Swift)).toBe(true); + // C is not `userSkippable` (required dep via the optional machinery) — even + // `all` must not mark it runtime-skipped. + expect(all.isGrammarRuntimeSkipped(SupportedLanguages.C)).toBe(false); + }); + + it('logs an accurate runtime-skip note, not a missing-binding message', async () => { + // Import the logger AND parser-loader from the SAME fresh module registry so + // the capture intercepts the loader's logger instance. + vi.resetModules(); + process.env[ENV] = 'swift'; + const { _captureLogger } = await import('../../src/core/logger.js'); + const { isLanguageAvailable } = await import('../../src/core/tree-sitter/parser-loader.js'); + const cap = _captureLogger(); + try { + isLanguageAvailable(SupportedLanguages.Swift); // triggers the one-time skip log + const msgs = cap + .records() + .map((r) => (typeof r.msg === 'string' ? r.msg : '')) + .filter(Boolean); + const skipMsg = msgs.find((m) => m.includes('GITNEXUS_SKIP_OPTIONAL_GRAMMARS')); + expect(skipMsg, `captured:\n${msgs.join('\n')}`).toBeTruthy(); + // The opt-out note must NOT borrow the install/platform "missing binding" + // language — that would tell a deliberate opt-out to reinstall/rebuild. + expect(skipMsg).not.toMatch(/no prebuilt|failed to load|npm rebuild/i); + } finally { + cap.restore(); + } + }); +}); From 774cd4d56826c62f389a4a1019de0dd8e61af3e2 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 9 Jun 2026 05:59:57 +0100 Subject: [PATCH 16/17] chore(deps)(deps): bump @ladybugdb/core in /gitnexus (#2098) --- gitnexus/package-lock.json | 46 +++++++++++++++++++------------------- 1 file changed, 23 insertions(+), 23 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index c5f4aa7af..e4fee45fc 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -1159,9 +1159,9 @@ } }, "node_modules/@ladybugdb/core": { - "version": "0.17.0", - "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.17.0.tgz", - "integrity": "sha512-fg7EGEJUj6H5JJLpU3iD5R0pAEc9u2OkU7ufX10krph9bCIhQyt/Tk6y/g4+x9zi/IHXegOjAVfSpWZp1gpoAA==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.17.1.tgz", + "integrity": "sha512-K1bHnQrRy3bxkyrFHlxGqKUyIUS1LsRXKOSt14XGY/msBZHaDat/uBrlHiWpM4/24OtfOq/qwTqcTCXannnEjw==", "hasInstallScript": true, "license": "MIT", "dependencies": { @@ -1170,17 +1170,17 @@ "node-addon-api": "^6.0.0" }, "optionalDependencies": { - "@ladybugdb/core-darwin-arm64": "0.17.0", - "@ladybugdb/core-darwin-x64": "0.17.0", - "@ladybugdb/core-linux-arm64": "0.17.0", - "@ladybugdb/core-linux-x64": "0.17.0", - "@ladybugdb/core-win32-x64": "0.17.0" + "@ladybugdb/core-darwin-arm64": "0.17.1", + "@ladybugdb/core-darwin-x64": "0.17.1", + "@ladybugdb/core-linux-arm64": "0.17.1", + "@ladybugdb/core-linux-x64": "0.17.1", + "@ladybugdb/core-win32-x64": "0.17.1" } }, "node_modules/@ladybugdb/core-darwin-arm64": { - "version": "0.17.0", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.17.0.tgz", - "integrity": "sha512-wghUBEmcQ9U10QOyOxXVQTZ6SHtkB8QV3sJHwCj2Dn5B/SsaB36kA4l2oKbgXpKDgjmSYiz3DfMs5yfDqen9UA==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.17.1.tgz", + "integrity": "sha512-JG/uzmolEh3wXJ/ME1EaTH5LTDQ9Cs+Q3Czul8pW2eWbWQZghQU3jjM++7ST7Bla5BX/WITqwPqPoC+sL+slfA==", "cpu": [ "arm64" ], @@ -1191,9 +1191,9 @@ ] }, "node_modules/@ladybugdb/core-darwin-x64": { - "version": "0.17.0", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.17.0.tgz", - "integrity": "sha512-f0QRhmDY8NEMjAT3IbFELMEFxAnKq6trppn1vgAFIk9wTD/MKakfX/gtXOazpOZQHkwlfNgs+WSkJFfFvQ4JaQ==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.17.1.tgz", + "integrity": "sha512-Enjm+/V9/jpKmtzF2PB0muVkgpFUGHEvA7r16eJWxVRA/BeO8VPmngTKy9rf/4Yc6TWexjoHRug04BbTXEmerg==", "cpu": [ "x64" ], @@ -1204,9 +1204,9 @@ ] }, "node_modules/@ladybugdb/core-linux-arm64": { - "version": "0.17.0", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.17.0.tgz", - "integrity": "sha512-TS9nbkvLJZt3Tgm6zzd/QuZmTiVoyQTPfbbwPej0VfcCVAfa1sDpZtoMlwse9EVHOrowQ0SorJVPg194lYVxdg==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.17.1.tgz", + "integrity": "sha512-P+xM9o4I3JAQtXpX19ZuLj9EeO2gppa+IdmAqhpI8tuhyA3/a85Eaxby1fXOjsbrnOAEyFJczUdyoDkhCPSyiw==", "cpu": [ "arm64" ], @@ -1217,9 +1217,9 @@ ] }, "node_modules/@ladybugdb/core-linux-x64": { - "version": "0.17.0", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.17.0.tgz", - "integrity": "sha512-T/C0QKDoBCs8s/NQ2Udip8lZgJ8MzLqs2rRgreDd2dCP3aNnXu8eOBe10RSoRO5PiZZIZihXoz4Q8KMM6FtBGQ==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.17.1.tgz", + "integrity": "sha512-N2ujE0CrsToBpVBpou1iWwEkK7CgVxucnUNxteySrnDccZwICXFP5BlcFpKE0qq3Eqmqszh4ptR4GuSi6rKPGw==", "cpu": [ "x64" ], @@ -1230,9 +1230,9 @@ ] }, "node_modules/@ladybugdb/core-win32-x64": { - "version": "0.17.0", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.17.0.tgz", - "integrity": "sha512-XrQrbPD3h+MhP94jVu+4VgNnp8LKSskPll+/au+Ug3yqpXZ0We9lXX5+rW4NHuIYdAYy4iLheYz4OuozUS20qg==", + "version": "0.17.1", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.17.1.tgz", + "integrity": "sha512-9i3xNfFAMqFRuQG3F1hOCWYGna6eTg8HJ/XYhWVDGkeFJNUV3IdneEiYttF5B2qAtQYUd4sAikScsImrMRw+6g==", "cpu": [ "x64" ], From 3a4247ec36b5ad86b1123d3bbce8183a643f7434 Mon Sep 17 00:00:00 2001 From: azizur100389 Date: Tue, 9 Jun 2026 06:53:11 +0100 Subject: [PATCH 17/17] feat(cpp): resolve inheritance-lattice member lookup (#2077) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(cpp): resolve inheritance-lattice member lookup * fix(cpp): harden inheritance-lattice lookup --------- Co-authored-by: Gergő Magyar --- gitnexus/bench/scope-capture/baselines.json | 4 +- .../languages/cpp/capture-side-channel.ts | 16 +- .../core/ingestion/languages/cpp/captures.ts | 2 + .../languages/cpp/import-decomposer.ts | 6 + .../ingestion/languages/cpp/member-lookup.ts | 616 ++++++++++++++++++ .../ingestion/languages/cpp/scope-resolver.ts | 11 +- .../contract/scope-resolver.ts | 22 +- .../passes/receiver-bound-calls.ts | 91 +++ .../scope-resolution/resolution-outcome.ts | 1 + .../lang-resolution/cpp-member-lattice/base.h | 5 + .../cpp-member-lattice/main.cpp | 142 ++++ .../test/integration/resolvers/cpp.test.ts | 103 +++ .../scope-resolution/cpp/cpp-imports.test.ts | 17 +- .../cpp-member-lookup-side-channel.test.ts | 41 ++ 14 files changed, 1066 insertions(+), 11 deletions(-) create mode 100644 gitnexus/src/core/ingestion/languages/cpp/member-lookup.ts create mode 100644 gitnexus/test/fixtures/lang-resolution/cpp-member-lattice/base.h create mode 100644 gitnexus/test/fixtures/lang-resolution/cpp-member-lattice/main.cpp create mode 100644 gitnexus/test/unit/scope-resolution/cpp/cpp-member-lookup-side-channel.test.ts diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index 79c504677..bb0a3f5f4 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -18,11 +18,11 @@ "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression)." }, "cpp": { - "fingerprint": "fd3d3768cdebbb4767d7cf18b8d2df19d61de969c816d7f4d6b599f947811356", + "fingerprint": "f56625342f73e182170e2c964d538e316c079fa6e9466a7f076bff2ebcf8aac4", "scaling_budget": 1.5, "_added": "#1956: cpp added to the scope-capture bench (was UNBENCHED). Heritage-bearing scale source (: public Base, public Mixin) drives emitCppInheritanceCaptures at scale. Adding it exposed + fixed a pre-existing O(n^2) findNodeAtRange root-walk in cpp/captures.ts (~12 sites, threaded c.node, byte-identical over 263 cpp-* fixtures); scaling 2.30 -> 1.12.", "_rebaselined": "#1919 open-language coverage: new lang-resolution fixtures + intended capture additions (F5/F9 c-cpp, F26/F28/F29 dart, F47/F48/F49/F51/F52 kotlin, F75/F79 swift). Fingerprint-only drift; scaling_ratio ~1.0 (linear, no perf regression).", - "_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift — no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures — pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture — pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae." + "_note": "#1975: + cpp-out-of-line-class fixture, fixture_count 263->265. #1990: + cpp-adl-ns-plus-hidden-friend-same-name fixture (ADL hidden-friend + namespace-callable merge parity test). Pure fixture-corpus drift — no scope-extractor change; existing fixtures' captures byte-identical. fixture_count 265->267. #1995: + cpp-union-nested-tail-collision and cpp-anon-ns-tail-collision fixtures — pure fixture-corpus drift; fixture_count 270->272, fingerprint 538e8be->d63ded6. #1993: + cpp-cross-namespace-same-tail fixture — pure fixture-corpus drift; fixture_count 272->273, fingerprint d63ded6->6d6207ae. #2077 review follow-up: cpp-member-lattice adds cross-file, qualified-base, nested-template, inherited-using, this-receiver, and non-virtual-override regressions; fixture_count 274->275. Capture scaling remains linear (1.134 < 1.5)." }, "csharp": { "_rebaselined": "#1956 synth-widening: + csharp-qualified-base fixture; the synth now walks record_declaration + struct_declaration base_lists and handles alias_qualified_name (matching the #1940 legacy leg), so record/struct heritage now emits. csharp-record-base gains a record inherits capture. (record->record SAME-namespace EXTENDS is a separate registry resolution gap, tracked as follow-up.) Linear (~1.00). (Earlier #1956: heritage-bearing scale source.) | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged. | #1924 F16: record primary-constructor base bindings now exclude constructor arguments; capture fingerprint changes, scaling remains linear. | #2036 review follow-up: csharp-record-base now exercises primary-constructor base dispatch end to end; +2 capture groups, scaling remains linear.", diff --git a/gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts index 74f432b69..ad9766d46 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/capture-side-channel.ts @@ -42,6 +42,11 @@ import { applyCppTwoPhaseSideChannel, type CppTwoPhaseSideChannel, } from './two-phase-lookup.js'; +import { + applyCppMemberLookupSideChannel, + collectCppMemberLookupSideChannel, + type CppMemberLookupSideChannel, +} from './member-lookup.js'; /** * Plain JSON-serializable composite of every C++ capture-time side-channel @@ -62,6 +67,7 @@ export interface CppCaptureSideChannel { readonly inlineNamespaceRanges: readonly string[]; readonly fileLocal: CppFileLocalSideChannel; readonly twoPhase: CppTwoPhaseSideChannel; + readonly memberLookup: CppMemberLookupSideChannel; } /** @@ -74,6 +80,7 @@ export function collectCppCaptureSideChannel(filePath: string): CppCaptureSideCh const inlineNamespaceRanges = collectCppInlineNamespaceSideChannel(filePath); const fileLocal = collectCppFileLocalSideChannel(filePath); const twoPhase = collectCppTwoPhaseSideChannel(filePath); + const memberLookup = collectCppMemberLookupSideChannel(filePath); const isEmpty = adl.argInfoBySite.length === 0 && @@ -82,10 +89,12 @@ export function collectCppCaptureSideChannel(filePath: string): CppCaptureSideCh fileLocal.fileLocalNames.length === 0 && fileLocal.anonymousNamespaceRanges.length === 0 && twoPhase.dependentBases.length === 0 && - twoPhase.dependentPackBaseClasses.length === 0; + twoPhase.dependentPackBaseClasses.length === 0 && + memberLookup.baseEdges.length === 0 && + memberLookup.memberUsings.length === 0; if (isEmpty) return undefined; - return { kind: 'cpp', adl, inlineNamespaceRanges, fileLocal, twoPhase }; + return { kind: 'cpp', adl, inlineNamespaceRanges, fileLocal, twoPhase, memberLookup }; } /** @@ -108,4 +117,7 @@ export function applyCppCaptureSideChannel(parsed: ParsedFile): void { } if (data.fileLocal !== undefined) applyCppFileLocalSideChannel(parsed.filePath, data.fileLocal); if (data.twoPhase !== undefined) applyCppTwoPhaseSideChannel(parsed.filePath, data.twoPhase); + if (data.memberLookup !== undefined) { + applyCppMemberLookupSideChannel(parsed.filePath, data.memberLookup); + } } diff --git a/gitnexus/src/core/ingestion/languages/cpp/captures.ts b/gitnexus/src/core/ingestion/languages/cpp/captures.ts index 265db8d5d..883f571c5 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/captures.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/captures.ts @@ -20,6 +20,7 @@ import { markCppDependentBase, markCppDependentPackBase } from './two-phase-look import { markCppAdlSiteArgs, markCppAdlSiteNoAdl, type CppAdlArgInfo } from './adl.js'; import { markCppInlineNamespaceRange } from './inline-namespaces.js'; import { extractCppTemplateConstraints } from './constraint-extractor.js'; +import { captureCppMemberLookupFacts } from './member-lookup.js'; export function emitCppScopeCaptures( sourceText: string, @@ -464,6 +465,7 @@ export function emitCppScopeCaptures( // and the resolver can suppress unqualified-call binding to those // bases per ISO C++ two-phase lookup. detectCppDependentBases(tree.rootNode, filePath); + captureCppMemberLookupFacts(tree.rootNode, filePath); return out; } diff --git a/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts b/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts index ae6843923..4b0820589 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/import-decomposer.ts @@ -73,6 +73,12 @@ function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMat */ export function splitCppUsingDecl(node: SyntaxNode): CaptureMatch | null { if (node.type !== 'using_declaration') return null; + // A class-scope `using Base::member;` changes the derived class's member + // lookup set; it is not a namespace import. The C++ member-lookup sidecar + // captures it separately, so suppress import decomposition here. + for (let parent = node.parent; parent !== null; parent = parent.parent) { + if (parent.type === 'class_specifier' || parent.type === 'struct_specifier') return null; + } // Check for "namespace" keyword among anonymous children let hasNamespaceKeyword = false; diff --git a/gitnexus/src/core/ingestion/languages/cpp/member-lookup.ts b/gitnexus/src/core/ingestion/languages/cpp/member-lookup.ts new file mode 100644 index 000000000..a681c4952 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/cpp/member-lookup.ts @@ -0,0 +1,616 @@ +import type { ParsedFile, ReferenceSite, SymbolDefinition } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../../../graph/types.js'; +import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import { resolveDefGraphId } from '../../scope-resolution/graph-bridge/ids.js'; +import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; +import type { SemanticModel } from '../../model/semantic-model.js'; +import type { ReceiverMemberResolution } from '../../scope-resolution/contract/scope-resolver.js'; +import { buildMro, defaultLinearize } from '../../scope-resolution/passes/mro.js'; +import { + isOverloadAmbiguousAfterNormalization, + narrowOverloadCandidates, +} from '../../scope-resolution/passes/overload-narrowing.js'; +import { isClassLike } from '../../scope-resolution/scope/walkers.js'; +import type { SyntaxNode } from '../../utils/ast-helpers.js'; +import { cppConstraintCompatibility } from './constraint-filter.js'; +import { cppConversionRank } from './conversion-rank.js'; + +interface CapturedBaseEdge { + readonly childName: string; + readonly childQualifiedName?: string; + readonly baseName: string; + readonly baseQualifiedName?: string; + readonly isVirtual: boolean; +} + +interface CapturedMemberUsing { + readonly childName: string; + readonly childQualifiedName?: string; + readonly baseName: string; + readonly baseQualifiedName?: string; + readonly memberName: string; +} + +export interface CppMemberLookupSideChannel { + readonly baseEdges: readonly CapturedBaseEdge[]; + readonly memberUsings: readonly CapturedMemberUsing[]; +} + +const capturedByFile = new Map(); +let directParentsByDefId = new Map(); +let virtualEdges = new Set(); +let ancestorsByDefId = new Map>(); +let memberUsingsByDefId = new Map< + string, + readonly { readonly baseDefId: string; readonly memberName: string }[] +>(); +let inheritedLookupCache = new Map(); + +const MAX_INHERITANCE_VISITS = 4096; + +type CachedInheritedLookup = + | { readonly kind: 'none' } + | { readonly kind: 'candidates'; readonly definitions: readonly SymbolDefinition[] } + | { readonly kind: 'ambiguous'; readonly candidateIds: readonly string[] }; + +export function clearCppMemberLookupState(): void { + capturedByFile.clear(); + directParentsByDefId = new Map(); + virtualEdges = new Set(); + ancestorsByDefId = new Map(); + memberUsingsByDefId = new Map(); + inheritedLookupCache = new Map(); +} + +export function captureCppMemberLookupFacts(root: SyntaxNode, filePath: string): void { + const baseEdges: CapturedBaseEdge[] = []; + const memberUsings: CapturedMemberUsing[] = []; + const stack: SyntaxNode[] = [root]; + + while (stack.length > 0) { + const node = stack.pop()!; + if (node.type === 'class_specifier' || node.type === 'struct_specifier') { + const childName = classNameOf(node); + const childQualifiedName = classQualifiedNameOf(node); + if (childName !== '') { + const baseClause = directChildOfType(node, 'base_class_clause'); + if (baseClause !== null) { + captureBaseEdges(baseClause, childName, childQualifiedName, baseEdges); + } + const body = directChildOfType(node, 'field_declaration_list'); + if (body !== null) { + for (let i = 0; i < body.namedChildCount; i++) { + const child = body.namedChild(i); + if (child?.type !== 'using_declaration') continue; + const parsed = parseMemberUsing(child, childName, childQualifiedName); + if (parsed !== undefined) memberUsings.push(parsed); + } + } + } + } + for (let i = 0; i < node.childCount; i++) { + const child = node.child(i); + if (child !== null) stack.push(child); + } + } + + if (baseEdges.length === 0 && memberUsings.length === 0) { + capturedByFile.delete(filePath); + } else { + capturedByFile.set(filePath, { baseEdges, memberUsings }); + } +} + +export function collectCppMemberLookupSideChannel(filePath: string): CppMemberLookupSideChannel { + return capturedByFile.get(filePath) ?? { baseEdges: [], memberUsings: [] }; +} + +export function applyCppMemberLookupSideChannel( + filePath: string, + data: CppMemberLookupSideChannel, +): void { + if (!Array.isArray(data.baseEdges) || !Array.isArray(data.memberUsings)) return; + if (data.baseEdges.length === 0 && data.memberUsings.length === 0) { + capturedByFile.delete(filePath); + return; + } + capturedByFile.set(filePath, { + baseEdges: data.baseEdges.slice(), + memberUsings: data.memberUsings.slice(), + }); +} + +export function buildCppMemberLookupMro( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + nodeLookup: GraphNodeLookup, +): Map { + populateResolvedHierarchy(graph, parsedFiles, nodeLookup); + return buildMro(graph, parsedFiles, nodeLookup, defaultLinearize); +} + +export function resolveCppReceiverMember( + ownerDef: SymbolDefinition, + memberName: string, + callsite: ReferenceSite, + _scopes: ScopeResolutionIndexes, + model: SemanticModel, +): ReceiverMemberResolution | undefined { + if (callsite.kind !== 'call') return undefined; + const ownMethods = model.methods.lookupAllByOwner(ownerDef.nodeId, memberName); + const introduced = introducedDefinitions(ownerDef.nodeId, memberName, model); + + if (introduced.length > 0) { + return chooseOverload(uniqueDefinitions([...ownMethods, ...introduced]), callsite); + } + + // Direct declarations hide every base declaration. Let the shared path + // retain its existing overload/static filtering for this common case. + if (ownMethods.length > 0) return undefined; + + const lookup = inheritedLookupSet(ownerDef.nodeId, memberName, model); + if (lookup.kind === 'none') return undefined; + if (lookup.kind === 'ambiguous') return lookup; + return chooseOverload(lookup.definitions, callsite); +} + +interface MemberOccurrence { + readonly ownerDefId: string; + readonly definitions: readonly SymbolDefinition[]; + readonly path: readonly string[]; + readonly virtualAnchor?: string; +} + +function collectInheritedOccurrences( + ownerDefId: string, + memberName: string, + model: SemanticModel, + path: readonly string[], + virtualAnchor: string | undefined, + active: Set, + budget: { remaining: number; truncated: boolean }, +): MemberOccurrence[] { + if (budget.remaining <= 0) { + budget.truncated = true; + return []; + } + budget.remaining--; + if (active.has(ownerDefId)) return []; + const nextActive = new Set(active); + nextActive.add(ownerDefId); + + const definitions = uniqueDefinitions([ + ...model.methods.lookupAllByOwner(ownerDefId, memberName), + ...introducedDefinitions(ownerDefId, memberName, model), + ]); + if (definitions.length > 0) { + return [{ ownerDefId, definitions, path, virtualAnchor }]; + } + + const results: MemberOccurrence[] = []; + for (const parentDefId of directParentsByDefId.get(ownerDefId) ?? []) { + const edgeKey = `${ownerDefId}\0${parentDefId}`; + results.push( + ...collectInheritedOccurrences( + parentDefId, + memberName, + model, + [...path, parentDefId], + virtualEdges.has(edgeKey) ? parentDefId : virtualAnchor, + nextActive, + budget, + ), + ); + } + return results; +} + +function inheritedLookupSet( + ownerDefId: string, + memberName: string, + model: SemanticModel, +): CachedInheritedLookup { + const cacheKey = `${ownerDefId}\0${memberName}`; + const cached = inheritedLookupCache.get(cacheKey); + if (cached !== undefined) return cached; + + const budget = { remaining: MAX_INHERITANCE_VISITS, truncated: false }; + const occurrences = collectInheritedOccurrences( + ownerDefId, + memberName, + model, + [], + undefined, + new Set(), + budget, + ); + if (budget.truncated) { + const conservative: CachedInheritedLookup = { + kind: 'ambiguous', + candidateIds: uniqueDefinitions(occurrences.flatMap((entry) => entry.definitions)).map( + (definition) => definition.nodeId, + ), + }; + inheritedLookupCache.set(cacheKey, conservative); + return conservative; + } + if (occurrences.length === 0) { + const none: CachedInheritedLookup = { kind: 'none' }; + inheritedLookupCache.set(cacheKey, none); + return none; + } + + // A declaration can dominate another lookup set only when the latter is + // reached through a shared virtual subobject. Ordinary ancestry alone is + // insufficient: declarations in one non-virtual branch do not hide members + // reached through a sibling base subobject. + const undominated = occurrences.filter( + (candidate) => + !( + candidate.virtualAnchor !== undefined && + occurrences.some( + (other) => + other.ownerDefId !== candidate.ownerDefId && + isAncestor(candidate.ownerDefId, other.ownerDefId), + ) + ), + ); + const groups = new Map(); + for (const occurrence of undominated) { + const key = + occurrence.virtualAnchor !== undefined + ? `virtual:${occurrence.virtualAnchor}:${occurrence.ownerDefId}` + : `path:${occurrence.path.join('>')}:${occurrence.ownerDefId}`; + const bucket = groups.get(key); + if (bucket === undefined) groups.set(key, [occurrence]); + else bucket.push(occurrence); + } + + let result: CachedInheritedLookup; + if (groups.size !== 1) { + result = { + kind: 'ambiguous', + candidateIds: uniqueDefinitions(undominated.flatMap((entry) => entry.definitions)).map( + (definition) => definition.nodeId, + ), + }; + } else { + result = { + kind: 'candidates', + definitions: groups.values().next().value?.[0]?.definitions ?? [], + }; + } + inheritedLookupCache.set(cacheKey, result); + return result; +} + +function introducedDefinitions( + ownerDefId: string, + memberName: string, + model: SemanticModel, +): SymbolDefinition[] { + const definitions: SymbolDefinition[] = []; + for (const entry of memberUsingsByDefId.get(ownerDefId) ?? []) { + if (entry.memberName !== memberName) continue; + definitions.push(...model.methods.lookupAllByOwner(entry.baseDefId, memberName)); + } + return definitions; +} + +function uniqueDefinitions(definitions: readonly SymbolDefinition[]): SymbolDefinition[] { + return [...new Map(definitions.map((definition) => [definition.nodeId, definition])).values()]; +} + +function chooseOverload( + candidates: readonly SymbolDefinition[], + callsite: ReferenceSite, +): ReceiverMemberResolution | undefined { + if (candidates.length === 0) return undefined; + const narrowed = narrowOverloadCandidates(candidates, callsite.arity, callsite.argumentTypes, { + argumentTypeClasses: callsite.argumentTypeClasses, + conversionRankFn: cppConversionRank, + constraintCompatibility: cppConstraintCompatibility, + }); + if (narrowed.length === 1) return { kind: 'resolved', definition: narrowed[0]! }; + if (narrowed.length > 1 || isOverloadAmbiguousAfterNormalization(narrowed, callsite.arity)) { + return { + kind: 'ambiguous', + candidateIds: narrowed.map((candidate) => candidate.nodeId), + }; + } + return undefined; +} + +function populateResolvedHierarchy( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + nodeLookup: GraphNodeLookup, +): void { + const defByGraphId = new Map(); + const defById = new Map(); + const defsByFileAndName = new Map(); + + for (const parsed of parsedFiles) { + for (const def of parsed.localDefs) { + if (!isClassLike(def.type)) continue; + const graphId = resolveDefGraphId(parsed.filePath, def, nodeLookup); + if (graphId === undefined) continue; + defByGraphId.set(graphId, def); + defById.set(def.nodeId, def); + const names = new Set([simpleName(def), definitionQualifiedName(def)]); + for (const name of names) { + if (name === '') continue; + const key = `${parsed.filePath}\0${name}`; + const bucket = defsByFileAndName.get(key); + if (bucket === undefined) defsByFileAndName.set(key, [def]); + else bucket.push(def); + } + } + } + + const parents = new Map(); + for (const rel of graph.iterRelationshipsByType('EXTENDS')) { + const child = defByGraphId.get(rel.sourceId); + const parent = defByGraphId.get(rel.targetId); + if (child === undefined || parent === undefined) continue; + const bucket = parents.get(child.nodeId); + if (bucket === undefined) parents.set(child.nodeId, [parent.nodeId]); + else bucket.push(parent.nodeId); + } + directParentsByDefId = parents; + ancestorsByDefId = buildAncestorClosure(parents); + inheritedLookupCache = new Map(); + + const nextVirtualEdges = new Set(); + const nextUsings = new Map< + string, + { readonly baseDefId: string; readonly memberName: string }[] + >(); + for (const parsed of parsedFiles) { + const captured = capturedByFile.get(parsed.filePath); + if (captured === undefined) continue; + for (const edge of captured.baseEdges) { + if (!edge.isVirtual) continue; + for (const child of matchingChildren( + parsed.filePath, + edge.childName, + edge.childQualifiedName, + defsByFileAndName, + )) { + const parent = findCapturedParent( + parents.get(child.nodeId) ?? [], + edge.baseName, + edge.baseQualifiedName, + defById, + ); + if (parent !== undefined) nextVirtualEdges.add(`${child.nodeId}\0${parent.nodeId}`); + } + } + for (const using of captured.memberUsings) { + const children = matchingChildren( + parsed.filePath, + using.childName, + using.childQualifiedName, + defsByFileAndName, + ); + for (const child of children) { + const baseDef = findCapturedParent( + parents.get(child.nodeId) ?? [], + using.baseName, + using.baseQualifiedName, + defById, + ); + if (baseDef === undefined) continue; + const bucket = nextUsings.get(child.nodeId); + const entry = { baseDefId: baseDef.nodeId, memberName: using.memberName }; + if (bucket === undefined) nextUsings.set(child.nodeId, [entry]); + else bucket.push(entry); + } + } + } + virtualEdges = nextVirtualEdges; + memberUsingsByDefId = nextUsings; +} + +function captureBaseEdges( + baseClause: SyntaxNode, + childName: string, + childQualifiedName: string, + output: CapturedBaseEdge[], +): void { + let segmentStart = 0; + for (let i = 0; i < baseClause.childCount; i++) { + const child = baseClause.child(i); + if (child === null) continue; + if (child.type === ',' || child.text === ',') { + segmentStart = i + 1; + continue; + } + if ( + child.type !== 'type_identifier' && + child.type !== 'template_type' && + child.type !== 'qualified_identifier' + ) { + continue; + } + let isVirtual = false; + for (let j = segmentStart; j < i; j++) { + const modifier = baseClause.child(j); + if (modifier?.text === 'virtual') isVirtual = true; + } + const baseQualifiedName = qualifiedTypeName(child.text); + const baseName = baseQualifiedName.split('.').at(-1) ?? ''; + if (baseName !== '') { + output.push({ + childName, + ...(childQualifiedName !== childName ? { childQualifiedName } : {}), + baseName, + ...(baseQualifiedName !== baseName ? { baseQualifiedName } : {}), + isVirtual, + }); + } + } +} + +function parseMemberUsing( + node: SyntaxNode, + childName: string, + childQualifiedName: string, +): CapturedMemberUsing | undefined { + const qualified = node.namedChildren.find((child) => child.type === 'qualified_identifier'); + if (qualified === undefined) return undefined; + const parts = splitQualifiedSegments(qualified.text); + if (parts.length < 2) return undefined; + const memberName = stripTemplateSuffix(parts.at(-1) ?? ''); + const baseParts = parts.slice(0, -1).map(stripTemplateSuffix).filter(Boolean); + const baseName = baseParts.at(-1) ?? ''; + const baseQualifiedName = baseParts.join('.'); + if (baseName === '' || memberName === '') return undefined; + return { + childName, + ...(childQualifiedName !== childName ? { childQualifiedName } : {}), + baseName, + ...(baseQualifiedName !== baseName ? { baseQualifiedName } : {}), + memberName, + }; +} + +function classNameOf(node: SyntaxNode): string { + const name = node.childForFieldName?.('name'); + return name === null || name === undefined ? '' : trailingIdentifier(name.text); +} + +function classQualifiedNameOf(node: SyntaxNode): string { + const parts = [classNameOf(node)]; + let current = node.parent; + while (current !== null) { + if (current.type === 'class_specifier' || current.type === 'struct_specifier') { + const name = classNameOf(current); + if (name !== '') parts.unshift(name); + } else if (current.type === 'namespace_definition') { + const name = current.childForFieldName?.('name'); + if (name !== null && name !== undefined) { + parts.unshift( + ...splitQualifiedSegments(name.text).map(stripTemplateSuffix).filter(Boolean), + ); + } + } + current = current.parent; + } + return parts.filter(Boolean).join('.'); +} + +function directChildOfType(node: SyntaxNode, type: string): SyntaxNode | null { + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child?.type === type) return child; + } + return null; +} + +function trailingIdentifier(value: string): string { + return stripTemplateSuffix(splitQualifiedSegments(value).at(-1) ?? ''); +} + +function qualifiedTypeName(value: string): string { + return splitQualifiedSegments(value).map(stripTemplateSuffix).filter(Boolean).join('.'); +} + +function splitQualifiedSegments(value: string): string[] { + const parts: string[] = []; + let angleDepth = 0; + let segmentStart = 0; + for (let i = 0; i < value.length; i++) { + const char = value[i]; + if (char === '<') angleDepth++; + else if (char === '>' && angleDepth > 0) angleDepth--; + else if (char === ':' && value[i + 1] === ':' && angleDepth === 0) { + const segment = value.slice(segmentStart, i).trim(); + if (segment !== '') parts.push(segment); + segmentStart = i + 2; + i++; + } + } + const tail = value.slice(segmentStart).trim(); + if (tail !== '') parts.push(tail); + return parts; +} + +function stripTemplateSuffix(value: string): string { + const templateStart = value.indexOf('<'); + return (templateStart >= 0 ? value.slice(0, templateStart) : value).trim(); +} + +function simpleName(def: SymbolDefinition): string { + return def.qualifiedName?.split('.').at(-1) ?? ''; +} + +function definitionQualifiedName(def: SymbolDefinition): string { + const name = def.qualifiedName ?? ''; + if (name === '' || def.namespacePrefix === undefined || def.namespacePrefix === '') return name; + return name.startsWith(`${def.namespacePrefix}.`) ? name : `${def.namespacePrefix}.${name}`; +} + +function matchingChildren( + filePath: string, + childName: string, + childQualifiedName: string | undefined, + defsByFileAndName: ReadonlyMap, +): readonly SymbolDefinition[] { + if (childQualifiedName !== undefined) { + const qualified = defsByFileAndName.get(`${filePath}\0${childQualifiedName}`) ?? []; + if (qualified.length > 0) return qualified; + } + const simple = defsByFileAndName.get(`${filePath}\0${childName}`) ?? []; + return simple.length === 1 ? simple : []; +} + +function findCapturedParent( + parentIds: readonly string[], + baseName: string, + baseQualifiedName: string | undefined, + defById: ReadonlyMap, +): SymbolDefinition | undefined { + const candidates = parentIds + .map((id) => defById.get(id)) + .filter((definition): definition is SymbolDefinition => definition !== undefined); + if (baseQualifiedName !== undefined) { + const qualified = candidates.filter((definition) => { + const name = definitionQualifiedName(definition); + return name === baseQualifiedName || name.endsWith(`.${baseQualifiedName}`); + }); + if (qualified.length === 1) return qualified[0]; + return undefined; + } + const simple = candidates.filter((definition) => simpleName(definition) === baseName); + return simple.length === 1 ? simple[0] : undefined; +} + +function buildAncestorClosure( + parents: ReadonlyMap, +): Map> { + const closure = new Map>(); + const visiting = new Set(); + + const ancestorsOf = (defId: string): ReadonlySet => { + const cached = closure.get(defId); + if (cached !== undefined) return cached; + if (visiting.has(defId)) return new Set(); + visiting.add(defId); + const ancestors = new Set(); + for (const parent of parents.get(defId) ?? []) { + ancestors.add(parent); + for (const ancestor of ancestorsOf(parent)) ancestors.add(ancestor); + } + visiting.delete(defId); + closure.set(defId, ancestors); + return ancestors; + }; + + for (const defId of parents.keys()) ancestorsOf(defId); + return closure; +} + +function isAncestor(ancestorDefId: string, descendantDefId: string): boolean { + return ancestorsByDefId.get(descendantDefId)?.has(ancestorDefId) === true; +} diff --git a/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts index 5e24e292d..3ef89bc07 100644 --- a/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/cpp/scope-resolver.ts @@ -4,7 +4,6 @@ import { findEnclosingClassDef, } from '../../scope-resolution/scope/walkers.js'; import { SupportedLanguages } from 'gitnexus-shared'; -import { buildMro, defaultLinearize } from '../../scope-resolution/passes/mro.js'; import { populateClassOwnedMembers, tagNamespacePrefixes, @@ -42,6 +41,11 @@ import { clearCppUserDefinedConversions, populateCppUserDefinedConversions, } from './user-defined-conversions.js'; +import { + buildCppMemberLookupMro, + clearCppMemberLookupState, + resolveCppReceiverMember, +} from './member-lookup.js'; /** * Per-pass memo of the augmented `#include`-resolution file set @@ -104,6 +108,7 @@ export const cppScopeResolver: ScopeResolver = { clearCppAdlState(); clearCppInlineNamespaces(); clearCppUserDefinedConversions(); + clearCppMemberLookupState(); return scanCppHeaderFiles(repoPath); }, @@ -137,8 +142,7 @@ export const cppScopeResolver: ScopeResolver = { // `'unknown'` keeps the candidate, preserving "degrade not lie". constraintCompatibility: cppConstraintCompatibility, - buildMro: (graph, parsedFiles, nodeLookup) => - buildMro(graph, parsedFiles, nodeLookup, defaultLinearize), + buildMro: buildCppMemberLookupMro, // Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`). // `emitCppScopeCaptures` records per-file ADL call-site arg shapes @@ -261,6 +265,7 @@ export const cppScopeResolver: ScopeResolver = { hoistTypeBindingsToModule: true, // Enable receiver-bound explicit-`this` fallback only for C++. resolveThisViaEnclosingClass: true, + resolveReceiverMember: resolveCppReceiverMember, // The `isFileLocalDef` hook on the global free-call fallback names // file-local linkage historically, but semantically gates "logically // invisible cross-file" defs. C++ extends this to also reject class- diff --git a/gitnexus/src/core/ingestion/scope-resolution/contract/scope-resolver.ts b/gitnexus/src/core/ingestion/scope-resolution/contract/scope-resolver.ts index 1c85f2383..fc17ac414 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/contract/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/contract/scope-resolver.ts @@ -267,6 +267,7 @@ import type { Callsite, ConstraintContext, ParsedFile, + ReferenceSite, ScopeId, SupportedLanguages, SymbolDefinition, @@ -291,6 +292,10 @@ export type LinearizeStrategy = ( /** Result of `ScopeResolver.arityCompatibility` — mirrors `RegistryProviders.arityCompatibility`. */ export type ArityVerdict = 'compatible' | 'unknown' | 'incompatible'; +export type ReceiverMemberResolution = + | { readonly kind: 'resolved'; readonly definition: SymbolDefinition } + | { readonly kind: 'ambiguous'; readonly candidateIds: readonly string[] }; + /** Re-exported for ScopeResolver consumers — same shape as * `RegistryProviders.constraintCompatibility`'s third parameter. */ export type { ConstraintContext } from 'gitnexus-shared'; @@ -407,7 +412,7 @@ export interface ScopeResolver { * for the Tier-A predicate registry and Kleene 3-valued evaluator. */ readonly constraintCompatibility?: ( - callsite: Callsite, + callsite: ReferenceSite, def: SymbolDefinition, ctx: ConstraintContext, ) => ArityVerdict; @@ -834,6 +839,21 @@ export interface ScopeResolver { callsite?: Callsite, ) => SymbolDefinition | 'ambiguous' | undefined; + /** + * Optional language-specific member-lattice lookup. Runs for a resolved + * simple receiver type before the generic flattened-MRO walk. Languages + * with lookup-set semantics that cannot be represented by one linear MRO + * may resolve a member, report ambiguity (which suppresses fallback), or + * return undefined to retain the shared behavior. + */ + readonly resolveReceiverMember?: ( + ownerDef: SymbolDefinition, + memberName: string, + callsite: Callsite, + scopes: ScopeResolutionIndexes, + model: SemanticModel, + ) => ReceiverMemberResolution | undefined; + /** * Enable the receiver-bound Case 0.5 fallback for explicit `this` * receivers (`this->m()` / `this.m()`) that resolves against the diff --git a/gitnexus/src/core/ingestion/scope-resolution/passes/receiver-bound-calls.ts b/gitnexus/src/core/ingestion/scope-resolution/passes/receiver-bound-calls.ts index d5aa2f788..1b7429708 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/passes/receiver-bound-calls.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/passes/receiver-bound-calls.ts @@ -82,6 +82,7 @@ type ReceiverBoundProviderSubset = Pick< | 'unwrapCollectionAccessor' | 'hoistTypeBindingsToModule' | 'resolveQualifiedReceiverMember' + | 'resolveReceiverMember' | 'resolveThisViaEnclosingClass' | 'conversionRankFn' | 'constraintCompatibility' @@ -375,6 +376,51 @@ export function emitReceiverBoundCalls( if (provider.resolveThisViaEnclosingClass === true && receiverName === 'this') { const enclosingClass = findEnclosingClassDef(site.inScope, scopes); if (enclosingClass !== undefined) { + const languageResolution = provider.resolveReceiverMember?.( + enclosingClass, + memberName, + site, + scopes, + model, + ); + if (languageResolution?.kind === 'ambiguous') { + options.recordResolutionOutcome?.({ + kind: 'suppressed', + phase: 'receiver-bound-calls', + filePath: parsed.filePath, + name: site.name, + range: site.atRange, + reason: 'member-lookup-ambiguous', + candidateIds: languageResolution.candidateIds, + }); + handledSites.add(siteKey); + continue; + } + if (languageResolution?.kind === 'resolved') { + const memberDef = languageResolution.definition; + const reason = + site.kind === 'write' || site.kind === 'read' + ? site.kind + : memberDef.filePath !== parsed.filePath + ? 'import-resolved' + : 'global'; + const confidence = site.kind === 'write' || site.kind === 'read' ? 1.0 : 0.85; + const ok = tryEmitEdge( + graph, + scopes, + nodeLookup, + site, + memberDef, + reason, + seen, + confidence, + collapse, + ); + if (ok) emitted++; + handledSites.add(siteKey); + continue; + } + const chain = [ enclosingClass.nodeId, ...scopes.methodDispatch.mroFor(enclosingClass.nodeId), @@ -722,6 +768,51 @@ export function emitReceiverBoundCalls( ); } if (ownerDef !== undefined) { + const languageResolution = provider.resolveReceiverMember?.( + ownerDef, + memberName, + site, + scopes, + model, + ); + if (languageResolution?.kind === 'ambiguous') { + options.recordResolutionOutcome?.({ + kind: 'suppressed', + phase: 'receiver-bound-calls', + filePath: parsed.filePath, + name: site.name, + range: site.atRange, + reason: 'member-lookup-ambiguous', + candidateIds: languageResolution.candidateIds, + }); + handledSites.add(siteKey); + continue; + } + if (languageResolution?.kind === 'resolved') { + const memberDef = languageResolution.definition; + const reason = + site.kind === 'write' || site.kind === 'read' + ? site.kind + : memberDef.filePath !== parsed.filePath + ? 'import-resolved' + : 'global'; + const confidence = site.kind === 'write' || site.kind === 'read' ? 1.0 : 0.85; + const ok = tryEmitEdge( + graph, + scopes, + nodeLookup, + site, + memberDef, + reason, + seen, + confidence, + collapse, + ); + if (ok) emitted++; + handledSites.add(siteKey); + continue; + } + const chain = [ownerDef.nodeId, ...scopes.methodDispatch.mroFor(ownerDef.nodeId)]; let memberDef: SymbolDefinition | undefined; let ambiguous = false; diff --git a/gitnexus/src/core/ingestion/scope-resolution/resolution-outcome.ts b/gitnexus/src/core/ingestion/scope-resolution/resolution-outcome.ts index 4eabfa29b..b46b47f72 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/resolution-outcome.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/resolution-outcome.ts @@ -4,6 +4,7 @@ export type ResolutionSuppressionReason = | 'adl-ordinary-lookup-blocked' | 'conversion-rank-tied' | 'inline-ns-ambiguous' + | 'member-lookup-ambiguous' | 'overload-ambiguous' | 'overload-ambiguous-normalization'; diff --git a/gitnexus/test/fixtures/lang-resolution/cpp-member-lattice/base.h b/gitnexus/test/fixtures/lang-resolution/cpp-member-lattice/base.h new file mode 100644 index 000000000..b25a94a88 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/cpp-member-lattice/base.h @@ -0,0 +1,5 @@ +#pragma once + +struct CrossFileBase { + void crossFile(); +}; diff --git a/gitnexus/test/fixtures/lang-resolution/cpp-member-lattice/main.cpp b/gitnexus/test/fixtures/lang-resolution/cpp-member-lattice/main.cpp new file mode 100644 index 000000000..85479b94e --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/cpp-member-lattice/main.cpp @@ -0,0 +1,142 @@ +#include "base.h" + +struct Left { + void collide(); +}; + +struct Right { + void collide(); +}; + +struct Ambiguous : Left, Right { + void callThis(); +}; + +void ambiguousCall() { + Ambiguous value; + value.collide(); +} + +void Ambiguous::callThis() { + this->collide(); +} + +struct Dominant : Left, Right { + void collide(); +}; + +void dominantCall() { + Dominant value; + value.collide(); +} + +struct Root { + void shared(); +}; + +struct VirtualLeft : virtual Root {}; +struct VirtualRight : virtual Root {}; +struct VirtualDiamond : VirtualLeft, VirtualRight {}; + +void virtualDiamondCall() { + VirtualDiamond value; + value.shared(); +} + +struct PlainLeft : Root {}; +struct PlainRight : Root {}; +struct PlainDiamond : PlainLeft, PlainRight {}; + +void plainDiamondCall() { + PlainDiamond value; + value.shared(); +} + +struct Base { + void select(int); +}; + +struct Derived : Base { + using Base::select; + void select(double); +}; + +void usingCall() { + Derived value; + value.select(1); +} + +struct OverrideRoot { + void overrideMember(); +}; + +struct OverrideLeft : OverrideRoot { + void overrideMember(); +}; + +struct OverrideRight : OverrideRoot {}; +struct OverrideDiamond : OverrideLeft, OverrideRight {}; + +void nonVirtualOverrideCall() { + OverrideDiamond value; + value.overrideMember(); +} + +struct UsingRoot { + void inheritedUsing(int); +}; + +struct UsingMiddle : UsingRoot { + using UsingRoot::inheritedUsing; + void inheritedUsing(double); +}; + +struct UsingLeaf : UsingMiddle {}; + +void inheritedUsingCall() { + UsingLeaf value; + value.inheritedUsing(1); +} + +namespace alpha { +struct SameNameBase { + void qualified(int); +}; +} + +namespace beta { +struct SameNameBase { + void qualified(double); +}; +} + +struct QualifiedBases : alpha::SameNameBase, beta::SameNameBase { + using alpha::SameNameBase::qualified; +}; + +void qualifiedUsingCall() { + QualifiedBases value; + value.qualified(1); +} + +template +struct TemplatedOuter { + template + struct NestedBase { + void nestedTemplate(); + }; +}; + +struct TemplatedDerived : TemplatedOuter::NestedBase {}; + +void nestedTemplateCall() { + TemplatedDerived value; + value.nestedTemplate(); +} + +struct CrossFileDerived : CrossFileBase {}; + +void crossFileCall() { + CrossFileDerived value; + value.crossFile(); +} diff --git a/gitnexus/test/integration/resolvers/cpp.test.ts b/gitnexus/test/integration/resolvers/cpp.test.ts index bfae951e3..9b6a98bcb 100644 --- a/gitnexus/test/integration/resolvers/cpp.test.ts +++ b/gitnexus/test/integration/resolvers/cpp.test.ts @@ -1851,6 +1851,109 @@ describe('C++ Derived : A, B — diamond inheritance via leftmost-base MRO (SM-1 }); }); +describe('C++ inheritance-lattice member lookup (#1891)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'cpp-member-lattice'), () => {}); + }, 60000); + + it('suppresses same-name members inherited from unrelated bases', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'ambiguousCall' && call.target === 'collide', + ); + expect(calls).toHaveLength(0); + }); + + it('lets a derived declaration hide both base declarations', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'dominantCall' && call.target === 'collide', + ); + expect(calls).toHaveLength(1); + expect(calls[0]?.targetFilePath).toBe('main.cpp'); + }); + + it('merges a shared virtual base into one member subobject', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'virtualDiamondCall' && call.target === 'shared', + ); + expect(calls).toHaveLength(1); + }); + + it('suppresses the same declaration reached through two non-virtual base subobjects', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'plainDiamondCall' && call.target === 'shared', + ); + expect(calls).toHaveLength(0); + }); + + it('adds a member using-declaration to the derived overload set', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'usingCall' && call.target === 'select', + ); + expect(calls).toHaveLength(1); + const target = result.graph.getNode(calls[0]!.rel.targetId); + expect(target?.properties.parameterTypes).toEqual(['int']); + }); + + it('records both conservative ambiguity suppressions', () => { + const outcomes = getResolutionOutcomes(result).filter( + (outcome) => outcome.kind === 'suppressed' && outcome.reason === 'member-lookup-ambiguous', + ); + const names = outcomes.map((outcome) => outcome.name); + expect(names).toContain('collide'); + expect(names).toContain('overrideMember'); + expect(names).toContain('shared'); + }); + + it('keeps sibling non-virtual subobjects ambiguous when one branch overrides the member', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'nonVirtualOverrideCall' && call.target === 'overrideMember', + ); + expect(calls).toHaveLength(0); + }); + + it('merges inherited using-declarations with methods declared by the same intermediate class', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'inheritedUsingCall' && call.target === 'inheritedUsing', + ); + expect(calls).toHaveLength(1); + const target = result.graph.getNode(calls[0]!.rel.targetId); + expect(target?.properties.parameterTypes).toEqual(['int']); + }); + + it('uses qualified base identities when same-simple-name direct bases collide', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'qualifiedUsingCall' && call.target === 'qualified', + ); + expect(calls).toHaveLength(1); + const target = result.graph.getNode(calls[0]!.rel.targetId); + expect(target?.properties.parameterTypes).toEqual(['int']); + }); + + it('normalizes every segment of a nested templated base name', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'nestedTemplateCall' && call.target === 'nestedTemplate', + ); + expect(calls).toHaveLength(1); + }); + + it('applies lattice ambiguity suppression to explicit this receivers', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'callThis' && call.target === 'collide', + ); + expect(calls).toHaveLength(0); + }); + + it('resolves inherited members across files', () => { + const calls = getRelationships(result, 'CALLS').filter( + (call) => call.source === 'crossFileCall' && call.target === 'crossFile', + ); + expect(calls).toHaveLength(1); + expect(calls[0]?.targetFilePath).toBe('base.h'); + }); +}); + // --------------------------------------------------------------------------- // U1: `#include` must not leak class-owned methods as unqualified bindings // --------------------------------------------------------------------------- diff --git a/gitnexus/test/unit/scope-resolution/cpp/cpp-imports.test.ts b/gitnexus/test/unit/scope-resolution/cpp/cpp-imports.test.ts index 6bc6e1b86..3f2761d50 100644 --- a/gitnexus/test/unit/scope-resolution/cpp/cpp-imports.test.ts +++ b/gitnexus/test/unit/scope-resolution/cpp/cpp-imports.test.ts @@ -14,9 +14,14 @@ import type { SyntaxNode } from '../../../../src/core/ingestion/utils/ast-helper function parseNode(src: string, type: string): SyntaxNode | null { const tree = getCppParser().parse(src); - for (let i = 0; i < tree.rootNode.namedChildCount; i++) { - const child = tree.rootNode.namedChild(i); - if (child?.type === type) return child as SyntaxNode; + const stack: SyntaxNode[] = [tree.rootNode as SyntaxNode]; + while (stack.length > 0) { + const node = stack.pop()!; + if (node.type === type) return node; + for (let i = node.namedChildCount - 1; i >= 0; i--) { + const child = node.namedChild(i); + if (child !== null) stack.push(child as SyntaxNode); + } } return null; } @@ -59,6 +64,12 @@ describe('C++ include decomposition (splitCppInclude)', () => { // ── using declaration decomposition ───────────────────────────────────────── describe('C++ using declaration decomposition (splitCppUsingDecl)', () => { + it('does not treat a class-scope member using-declaration as an import', () => { + const node = parseNode('struct Derived : Base { using Base::run; };', 'using_declaration'); + expect(node).not.toBeNull(); + expect(splitCppUsingDecl(node!)).toBeNull(); + }); + it('decomposes "using namespace std;" as wildcard import', () => { const node = parseNode('using namespace std;', 'using_declaration'); expect(node).not.toBeNull(); diff --git a/gitnexus/test/unit/scope-resolution/cpp/cpp-member-lookup-side-channel.test.ts b/gitnexus/test/unit/scope-resolution/cpp/cpp-member-lookup-side-channel.test.ts new file mode 100644 index 000000000..102ce517a --- /dev/null +++ b/gitnexus/test/unit/scope-resolution/cpp/cpp-member-lookup-side-channel.test.ts @@ -0,0 +1,41 @@ +import { beforeEach, describe, expect, it } from 'vitest'; +import { + applyCppMemberLookupSideChannel, + clearCppMemberLookupState, + collectCppMemberLookupSideChannel, + type CppMemberLookupSideChannel, +} from '../../../../src/core/ingestion/languages/cpp/member-lookup.js'; + +describe('C++ member-lookup capture side-channel', () => { + beforeEach(() => { + clearCppMemberLookupState(); + }); + + it('preserves qualified base identities through a worker-style JSON round trip', () => { + const snapshot: CppMemberLookupSideChannel = { + baseEdges: [ + { + childName: 'Derived', + childQualifiedName: 'app.Derived', + baseName: 'Base', + baseQualifiedName: 'detail.Base', + isVirtual: true, + }, + ], + memberUsings: [ + { + childName: 'Derived', + childQualifiedName: 'app.Derived', + baseName: 'Base', + baseQualifiedName: 'detail.Base', + memberName: 'select', + }, + ], + }; + const throughWorker = JSON.parse(JSON.stringify(snapshot)) as CppMemberLookupSideChannel; + + applyCppMemberLookupSideChannel('main.cpp', throughWorker); + + expect(collectCppMemberLookupSideChannel('main.cpp')).toEqual(snapshot); + }); +});