mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-09 03:17:54 +00:00
refactor(ingestion): worker-pool-only parsing; remove sequential parser (#1983)
Completes the #1983 huge-repo parse-OOM effort by making the worker pool GitNexus's sole parse path. Parallel serialization (the perf core): workers serialize their ParsedFiles to a disk store in parallel and stream them back to scope-resolution, so the main thread no longer re-parses every file (the tree-sitter native-memory leak that caused the OOM). Adds chunk merge-pipelining + work-proportional chunk sizing so the pool stays saturated. Remove the sequential parser: `--workers 0`, `GITNEXUS_WORKER_POOL_SIZE=0`, and `skipWorkers` now hard-error (no silent degrade — #1741); the small-repo threshold no longer selects an in-process path; pool creation stays lazy / cache-miss-gated so warm all-hit runs never spawn workers. Worker-path parity fixes — removing sequential surfaced two pre-existing gaps that tiny-fixture tests had masked by running below the worker threshold, both fixed by carrying per-file metadata as DATA across the worker boundary (never re-parsing on the main thread, preserving the OOM fix): - C++: templateConstraints wired into worker node identity (SFINAE overload disambiguation) + ADL / inline-namespace capture side-channel serialized onto the ParsedFile. - Kotlin: companion-scope side-channel serialized the same way (companion / static dispatch). Validation: tsc + build clean; full suite green (10,190 pass — the only deterministic failures were the now-fixed C++/Kotlin worker-path gaps; the 2 remaining full-run failures are pre-existing load flakiness, green in isolation); cpp-pipeline benchmark stays linear on a 1-worker pool. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
fccdadde95
commit
f45d86bb19
69 changed files with 2273 additions and 2038 deletions
|
|
@ -74,4 +74,28 @@ export interface ParsedFile {
|
|||
*/
|
||||
readonly localDefs: readonly SymbolDefinition[];
|
||||
readonly referenceSites: readonly ReferenceSite[];
|
||||
/**
|
||||
* Opaque, language-private serialization of capture-time side-channel
|
||||
* state that a provider's `emitScopeCaptures` populates into module-level
|
||||
* maps as a SIDE EFFECT (not onto the scopes/defs of this `ParsedFile`).
|
||||
*
|
||||
* Such state is computed inside the parse worker (where `emitScopeCaptures`
|
||||
* runs) and would otherwise be lost across the worker→main MessageChannel
|
||||
* and the disk store, because scope-resolution reuses the serialized
|
||||
* `ParsedFile` and SKIPS re-extraction on the main thread (#1983 — the
|
||||
* whole point is to avoid a main-thread tree-sitter re-parse). Carrying the
|
||||
* data here lets the main thread repopulate those maps WITHOUT re-parsing.
|
||||
*
|
||||
* Shared / ingestion code treats this as opaque (`unknown`) per AGENTS.md
|
||||
* (no language names in shared code). The producing language fills it via
|
||||
* the `LanguageProvider.collectCaptureSideChannel` hook (worker side) and
|
||||
* consumes it via the `ScopeResolver.applyCaptureSideChannel` hook
|
||||
* (main-thread resolution side). It MUST be plain JSON-serializable data
|
||||
* (objects / arrays / primitives) so it round-trips through the disk-backed
|
||||
* `parsedfile-store` (JSON.stringify + interning reviver).
|
||||
*
|
||||
* Optional: providers whose `emitScopeCaptures` is pure (no module-level
|
||||
* side effects — the contract default) leave this undefined.
|
||||
*/
|
||||
readonly captureSideChannel?: unknown;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -400,7 +400,7 @@ Values above **32768 KB (32 MB)** are clamped to the tree-sitter parser ceiling;
|
|||
|
||||
### Analyze reports a worker timeout
|
||||
|
||||
Worker parse timeouts are recoverable. GitNexus retries stalled worker jobs with backoff, splits large jobs to isolate slow files, and falls back to the sequential parser when needed. If a large repository needs more time per worker job, use either:
|
||||
Worker parse timeouts are recoverable. GitNexus retries stalled worker jobs with backoff, splits large jobs to isolate slow files, and quarantines a file that repeatedly crashes its worker (respawning the slot so the pool keeps going). If a large repository needs more time per worker job, use either:
|
||||
|
||||
```bash
|
||||
# CLI flag, in seconds
|
||||
|
|
|
|||
|
|
@ -474,8 +474,7 @@ async function ensureHeap(): Promise<boolean> {
|
|||
cliError(
|
||||
` Analysis aborted in a native worker or native binding path.\n` +
|
||||
` Try one of these recovery paths:\n` +
|
||||
` gitnexus analyze --workers 0\n` +
|
||||
` npm uninstall -g gitnexus && npm install -g gitnexus@latest\n` +
|
||||
` npm uninstall -g gitnexus && npm install -g gitnexus@latest (rebuilds native bindings)\n` +
|
||||
` Use Node 22 LTS if you are on a newer non-LTS runtime.\n`,
|
||||
{ recoveryHint: 'native-worker-abort' },
|
||||
);
|
||||
|
|
@ -599,7 +598,7 @@ export interface AnalyzeOptions {
|
|||
workerTimeout?: string;
|
||||
/** Control LadybugDB WAL auto-checkpoint threshold during analyze. */
|
||||
walCheckpointThreshold?: string;
|
||||
/** Parse worker pool size; 0 disables workers (sequential fallback). */
|
||||
/** Parse worker pool size (>=1); 0 is rejected (no sequential mode). */
|
||||
workers?: string;
|
||||
embeddingThreads?: string;
|
||||
embeddingBatchSize?: string;
|
||||
|
|
@ -794,10 +793,11 @@ const analyzeCommandImpl = async (
|
|||
let workerPoolSize: number | undefined;
|
||||
if (options.workers !== undefined) {
|
||||
const parsedWorkers = Number(options.workers);
|
||||
if (!Number.isInteger(parsedWorkers) || parsedWorkers < 0) {
|
||||
if (!Number.isInteger(parsedWorkers) || parsedWorkers < 1) {
|
||||
cliError(
|
||||
' --workers must be a non-negative integer. ' +
|
||||
'Pass 0 to disable the worker pool (sequential fallback).\n',
|
||||
' --workers must be a positive integer (>= 1). ' +
|
||||
'GitNexus parses through a worker pool only — there is no sequential ' +
|
||||
'mode, so 0 is not allowed. Omit --workers for an auto-sized pool.\n',
|
||||
);
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
|
|
|
|||
|
|
@ -175,7 +175,7 @@ export const en = {
|
|||
'help.option.analyze.walCheckpointThreshold':
|
||||
'LadybugDB WAL auto-checkpoint threshold in bytes during analyze (integer >= -1; default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).',
|
||||
'help.option.analyze.workers':
|
||||
'Parse worker pool size. Default: cores-1 capped at 16. Pass 0 to disable workers (sequential).',
|
||||
'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.',
|
||||
'help.option.analyze.embeddingThreads': 'Limit local ONNX embedding CPU threads',
|
||||
'help.option.analyze.embeddingBatchSize': 'Number of nodes per embedding batch',
|
||||
'help.option.analyze.embeddingSubBatchSize': 'Number of chunks per embedding model call',
|
||||
|
|
|
|||
|
|
@ -164,7 +164,7 @@ export const zhCN = {
|
|||
'help.option.analyze.walCheckpointThreshold':
|
||||
'analyze 期间 LadybugDB WAL 自动 checkpoint 阈值(字节,整数 >= -1;默认:67108864 = 64 MiB;-1 保持 Ladybug 默认约 16 MiB)。',
|
||||
'help.option.analyze.workers':
|
||||
'解析 worker 池大小。默认:cores-1,最多 16。传 0 禁用 worker(顺序执行)。',
|
||||
'解析 worker 池大小(>=1)。默认:cores-1,最多 16,按仓库规模自适应。',
|
||||
'help.option.analyze.embeddingThreads': '限制本地 ONNX 嵌入 CPU 线程数',
|
||||
'help.option.analyze.embeddingBatchSize': '每个嵌入批次的节点数',
|
||||
'help.option.analyze.embeddingSubBatchSize': '每次嵌入模型调用的分块数',
|
||||
|
|
|
|||
|
|
@ -87,7 +87,7 @@ program
|
|||
)
|
||||
.option(
|
||||
'--workers <n>',
|
||||
'Parse worker pool size. Default: cores-1 capped at 16. Pass 0 to disable workers (sequential).',
|
||||
'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.',
|
||||
)
|
||||
.option('--embedding-threads <n>', 'Limit local ONNX embedding CPU threads')
|
||||
.option('--embedding-batch-size <n>', 'Number of nodes per embedding batch')
|
||||
|
|
|
|||
|
|
@ -311,6 +311,27 @@ interface LanguageProviderConfig {
|
|||
},
|
||||
) => readonly CaptureMatch[];
|
||||
|
||||
/**
|
||||
* Snapshot the capture-time side-channel state that this provider's
|
||||
* `emitScopeCaptures` just populated for `filePath` into module-level maps,
|
||||
* returning a plain JSON-serializable value (or `undefined` when there is
|
||||
* nothing to carry).
|
||||
*
|
||||
* Called in the parse worker IMMEDIATELY after `emitScopeCaptures` runs for
|
||||
* a file (see `parse-worker.ts`), and the result is stored on the produced
|
||||
* `ParsedFile.captureSideChannel`. Scope-resolution on the main thread reuses
|
||||
* that serialized `ParsedFile` and skips re-extraction (#1983), so this hook
|
||||
* is how the worker-computed marks survive the worker→main boundary and the
|
||||
* disk store WITHOUT a main-thread re-parse. The main thread restores them
|
||||
* via the matching `ScopeResolver.applyCaptureSideChannel` hook.
|
||||
*
|
||||
* MUST return plain data (objects / arrays / primitives) so it round-trips
|
||||
* through `JSON.stringify` + the parsedfile-store interning reviver.
|
||||
*
|
||||
* Default: undefined (provider has no capture-time module-level side effects).
|
||||
*/
|
||||
readonly collectCaptureSideChannel?: (filePath: string) => unknown;
|
||||
|
||||
/**
|
||||
* Interpret a raw `@import.statement` capture group into a `ParsedImport`.
|
||||
* The central finalize algorithm resolves `ParsedImport.targetRaw` to a
|
||||
|
|
|
|||
|
|
@ -62,6 +62,7 @@ import {
|
|||
cppBindingScopeFor,
|
||||
cppImportOwningScope,
|
||||
cppReceiverBinding,
|
||||
collectCppCaptureSideChannel,
|
||||
} from './cpp/index.js';
|
||||
import { extractCppTemplateConstraints } from './cpp/constraint-extractor.js';
|
||||
|
||||
|
|
@ -465,6 +466,11 @@ export const cppProvider = defineLanguage({
|
|||
|
||||
// ── RFC #909 Ring 3: scope-based resolution hooks (RFC §5) ──────────
|
||||
emitScopeCaptures: emitCppScopeCaptures,
|
||||
// Worker-side: snapshot the module-level capture marks `emitCppScopeCaptures`
|
||||
// just populated for this file into plain data on `ParsedFile.captureSideChannel`,
|
||||
// so the main thread can restore them via `applyCaptureSideChannel` WITHOUT a
|
||||
// re-parse (#1983). See `cpp/capture-side-channel.ts`.
|
||||
collectCaptureSideChannel: collectCppCaptureSideChannel,
|
||||
interpretImport: interpretCppImport,
|
||||
interpretTypeBinding: interpretCppTypeBinding,
|
||||
bindingScopeFor: cppBindingScopeFor,
|
||||
|
|
|
|||
|
|
@ -379,6 +379,58 @@ export function markCppAdlSiteNoAdl(filePath: string, line: number, col: number)
|
|||
noAdlSites.add(siteKey(filePath, line, col));
|
||||
}
|
||||
|
||||
/**
|
||||
* Plain-data, JSON-serializable snapshot of the per-file ADL capture state
|
||||
* (`argInfoBySite` entries for this file + `noAdlSites` keys for this file).
|
||||
* Carried on `ParsedFile.captureSideChannel` across the worker→main boundary
|
||||
* (#1983); the call-site key's `line:col` are stored per-entry so the full
|
||||
* `filePath:line:col` key can be reconstructed without parsing.
|
||||
*/
|
||||
export interface CppAdlSideChannel {
|
||||
/** Per-call-site arg info: `[line, col, args]` for sites in this file. */
|
||||
readonly argInfoBySite: readonly [number, number, readonly CppAdlArgInfo[]][];
|
||||
/** ADL-suppressed sites in this file: `[line, col]`. */
|
||||
readonly noAdlSites: readonly [number, number][];
|
||||
}
|
||||
|
||||
const SITE_KEY_RE = /^(.*):(\d+):(\d+)$/;
|
||||
|
||||
/** Split a `filePath:line:col` site key, tolerating colons in the path. */
|
||||
function parseSiteKey(key: string): { filePath: string; line: number; col: number } | undefined {
|
||||
const m = SITE_KEY_RE.exec(key);
|
||||
if (m === null) return undefined;
|
||||
return { filePath: m[1], line: Number(m[2]), col: Number(m[3]) };
|
||||
}
|
||||
|
||||
/** Snapshot this file's ADL capture state for the worker→main side-channel. */
|
||||
export function collectCppAdlSideChannel(filePath: string): CppAdlSideChannel {
|
||||
const args: [number, number, readonly CppAdlArgInfo[]][] = [];
|
||||
for (const [key, value] of argInfoBySite) {
|
||||
const parsed = parseSiteKey(key);
|
||||
if (parsed !== undefined && parsed.filePath === filePath) {
|
||||
args.push([parsed.line, parsed.col, value]);
|
||||
}
|
||||
}
|
||||
const noAdl: [number, number][] = [];
|
||||
for (const key of noAdlSites) {
|
||||
const parsed = parseSiteKey(key);
|
||||
if (parsed !== undefined && parsed.filePath === filePath) {
|
||||
noAdl.push([parsed.line, parsed.col]);
|
||||
}
|
||||
}
|
||||
return { argInfoBySite: args, noAdlSites: noAdl };
|
||||
}
|
||||
|
||||
/** Restore this file's ADL capture state from the side-channel (no parse). */
|
||||
export function applyCppAdlSideChannel(filePath: string, data: CppAdlSideChannel): void {
|
||||
for (const [line, col, value] of data.argInfoBySite) {
|
||||
argInfoBySite.set(siteKey(filePath, line, col), value);
|
||||
}
|
||||
for (const [line, col] of data.noAdlSites) {
|
||||
noAdlSites.add(siteKey(filePath, line, col));
|
||||
}
|
||||
}
|
||||
|
||||
/** Clear ADL state. Called from `cppScopeResolver.loadResolutionConfig`
|
||||
* (alongside `clearFileLocalNames`) so all C++ resolver per-pipeline state is
|
||||
* reset together at the start of each resolution pass. */
|
||||
|
|
|
|||
|
|
@ -0,0 +1,98 @@
|
|||
/**
|
||||
* C++ capture-time side-channel serialization (#1983).
|
||||
*
|
||||
* `emitCppScopeCaptures` populates several MODULE-LEVEL maps as a side effect
|
||||
* that are NOT part of the returned `ParsedFile`'s scopes/defs:
|
||||
*
|
||||
* - `argInfoBySite` / `noAdlSites` (adl.ts)
|
||||
* - `inlineNamespaceRangesByFile` (inline-namespaces.ts)
|
||||
* - `fileLocalNames` / `anonymousNamespaceRangesByFile` (file-local-linkage.ts)
|
||||
* - `dependentBasesByFile` / `dependentPackBaseClassesByFile` (two-phase-lookup.ts)
|
||||
*
|
||||
* On the worker path those maps are filled in the WORKER process and lost
|
||||
* across the worker→main MessageChannel (and the disk-backed parsedfile-store),
|
||||
* because scope-resolution reuses the serialized `ParsedFile` and SKIPS the
|
||||
* main-thread re-extraction — the entire point of #1983 is to avoid a
|
||||
* main-thread tree-sitter re-parse on huge `.h`/`.cpp` repos (the OOM).
|
||||
*
|
||||
* This module snapshots the per-file slice of those maps into a plain,
|
||||
* JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and
|
||||
* restores it on the main thread WITHOUT any parse. It is the data-only
|
||||
* replacement for the removed re-parse `replayCaptureSideChannel` hook.
|
||||
*
|
||||
* The derived state each `populateOwners` / `populateWorkspaceOwners` pass
|
||||
* builds (resolved scope-id Sets, `dependentBaseNodeIds`, etc.) is recomputed
|
||||
* on the main thread from these restored capture-time maps, so only the
|
||||
* capture-time maps need to cross the boundary.
|
||||
*/
|
||||
|
||||
import type { ParsedFile } from 'gitnexus-shared';
|
||||
import { collectCppAdlSideChannel, applyCppAdlSideChannel, type CppAdlSideChannel } from './adl.js';
|
||||
import {
|
||||
collectCppInlineNamespaceSideChannel,
|
||||
applyCppInlineNamespaceSideChannel,
|
||||
} from './inline-namespaces.js';
|
||||
import {
|
||||
collectCppFileLocalSideChannel,
|
||||
applyCppFileLocalSideChannel,
|
||||
type CppFileLocalSideChannel,
|
||||
} from './file-local-linkage.js';
|
||||
import {
|
||||
collectCppTwoPhaseSideChannel,
|
||||
applyCppTwoPhaseSideChannel,
|
||||
type CppTwoPhaseSideChannel,
|
||||
} from './two-phase-lookup.js';
|
||||
|
||||
/**
|
||||
* Plain JSON-serializable composite of every C++ capture-time side-channel
|
||||
* slice for one file. Carried opaquely on `ParsedFile.captureSideChannel`.
|
||||
*/
|
||||
export interface CppCaptureSideChannel {
|
||||
readonly adl: CppAdlSideChannel;
|
||||
/** Inline-namespace source-range keys recorded for this file. */
|
||||
readonly inlineNamespaceRanges: readonly string[];
|
||||
readonly fileLocal: CppFileLocalSideChannel;
|
||||
readonly twoPhase: CppTwoPhaseSideChannel;
|
||||
}
|
||||
|
||||
/**
|
||||
* `LanguageProvider.collectCaptureSideChannel` implementation for C++.
|
||||
* Returns `undefined` when this file recorded no side-channel state at all, so
|
||||
* the produced `ParsedFile` carries the field only when there's data to ship.
|
||||
*/
|
||||
export function collectCppCaptureSideChannel(filePath: string): CppCaptureSideChannel | undefined {
|
||||
const adl = collectCppAdlSideChannel(filePath);
|
||||
const inlineNamespaceRanges = collectCppInlineNamespaceSideChannel(filePath);
|
||||
const fileLocal = collectCppFileLocalSideChannel(filePath);
|
||||
const twoPhase = collectCppTwoPhaseSideChannel(filePath);
|
||||
|
||||
const isEmpty =
|
||||
adl.argInfoBySite.length === 0 &&
|
||||
adl.noAdlSites.length === 0 &&
|
||||
inlineNamespaceRanges.length === 0 &&
|
||||
fileLocal.fileLocalNames.length === 0 &&
|
||||
fileLocal.anonymousNamespaceRanges.length === 0 &&
|
||||
twoPhase.dependentBases.length === 0 &&
|
||||
twoPhase.dependentPackBaseClasses.length === 0;
|
||||
if (isEmpty) return undefined;
|
||||
|
||||
return { adl, inlineNamespaceRanges, fileLocal, twoPhase };
|
||||
}
|
||||
|
||||
/**
|
||||
* `ScopeResolver.applyCaptureSideChannel` implementation for C++. Reads the
|
||||
* worker-serialized snapshot from `parsed.captureSideChannel` and writes it
|
||||
* back into the module-level maps. Tolerant of `undefined` (file carried no
|
||||
* data) and of an unexpected shape (defensive — never throws on a malformed
|
||||
* snapshot). Does NO tree-sitter parse.
|
||||
*/
|
||||
export function applyCppCaptureSideChannel(parsed: ParsedFile): void {
|
||||
const data = parsed.captureSideChannel as CppCaptureSideChannel | undefined;
|
||||
if (data === undefined || data === null || typeof data !== 'object') return;
|
||||
if (data.adl !== undefined) applyCppAdlSideChannel(parsed.filePath, data.adl);
|
||||
if (data.inlineNamespaceRanges !== undefined) {
|
||||
applyCppInlineNamespaceSideChannel(parsed.filePath, data.inlineNamespaceRanges);
|
||||
}
|
||||
if (data.fileLocal !== undefined) applyCppFileLocalSideChannel(parsed.filePath, data.fileLocal);
|
||||
if (data.twoPhase !== undefined) applyCppTwoPhaseSideChannel(parsed.filePath, data.twoPhase);
|
||||
}
|
||||
|
|
@ -95,6 +95,47 @@ export function isCppAnonymousNamespaceScope(scopeId: ScopeId): boolean {
|
|||
return anonymousNamespaceScopeIds.has(scopeId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Plain-data, JSON-serializable snapshot of the per-file capture-time
|
||||
* file-local-linkage state. Carried on `ParsedFile.captureSideChannel` across
|
||||
* the worker→main boundary (#1983). The derived sets (`nonGloballyVisibleNodeIds`,
|
||||
* `anonymousNamespaceScopeIds`) are recomputed by `populateCppNonGloballyVisible`
|
||||
* / `populateCppAnonymousNamespaceScopes` during `populateOwners`, so only the
|
||||
* two capture-time maps cross the boundary.
|
||||
*/
|
||||
export interface CppFileLocalSideChannel {
|
||||
/** File-local symbol names (static / anonymous-namespace) in this file. */
|
||||
readonly fileLocalNames: readonly string[];
|
||||
/** Anonymous-namespace source-range keys recorded for this file. */
|
||||
readonly anonymousNamespaceRanges: readonly string[];
|
||||
}
|
||||
|
||||
/** Snapshot this file's file-local-linkage capture state for the side-channel. */
|
||||
export function collectCppFileLocalSideChannel(filePath: string): CppFileLocalSideChannel {
|
||||
const names = fileLocalNames.get(filePath);
|
||||
const anon = anonymousNamespaceRangesByFile.get(filePath);
|
||||
return {
|
||||
fileLocalNames: names === undefined ? [] : [...names],
|
||||
anonymousNamespaceRanges: anon === undefined ? [] : [...anon],
|
||||
};
|
||||
}
|
||||
|
||||
/** Restore this file's file-local-linkage capture state from the side-channel. */
|
||||
export function applyCppFileLocalSideChannel(
|
||||
filePath: string,
|
||||
data: CppFileLocalSideChannel,
|
||||
): void {
|
||||
for (const name of data.fileLocalNames) markFileLocal(filePath, name);
|
||||
if (data.anonymousNamespaceRanges.length > 0) {
|
||||
let set = anonymousNamespaceRangesByFile.get(filePath);
|
||||
if (set === undefined) {
|
||||
set = new Set();
|
||||
anonymousNamespaceRangesByFile.set(filePath, set);
|
||||
}
|
||||
for (const r of data.anonymousNamespaceRanges) set.add(r);
|
||||
}
|
||||
}
|
||||
|
||||
/** Clear tracked file-local names (call at start of each resolution pass). */
|
||||
export function clearFileLocalNames(): void {
|
||||
fileLocalNames.clear();
|
||||
|
|
|
|||
|
|
@ -14,3 +14,7 @@ export {
|
|||
clearFileLocalNames,
|
||||
expandCppWildcardNames,
|
||||
} from './file-local-linkage.js';
|
||||
export {
|
||||
collectCppCaptureSideChannel,
|
||||
applyCppCaptureSideChannel,
|
||||
} from './capture-side-channel.js';
|
||||
|
|
|
|||
|
|
@ -61,6 +61,30 @@ export function markCppInlineNamespaceRange(filePath: string, range: RangeKey):
|
|||
set.add(rangeKey(range));
|
||||
}
|
||||
|
||||
/** Snapshot this file's captured inline-namespace ranges for the worker→main
|
||||
* side-channel (#1983). `populateCppInlineNamespaceScopes` (in `populateOwners`)
|
||||
* later resolves these range keys to ScopeIds on the main thread, so only the
|
||||
* capture-time ranges need to cross the boundary. Returns the rangeKey strings
|
||||
* as a plain array (empty when this file recorded none). */
|
||||
export function collectCppInlineNamespaceSideChannel(filePath: string): readonly string[] {
|
||||
const set = inlineNamespaceRangesByFile.get(filePath);
|
||||
return set === undefined ? [] : [...set];
|
||||
}
|
||||
|
||||
/** Restore this file's captured inline-namespace ranges from the side-channel. */
|
||||
export function applyCppInlineNamespaceSideChannel(
|
||||
filePath: string,
|
||||
ranges: readonly string[],
|
||||
): void {
|
||||
if (ranges.length === 0) return;
|
||||
let set = inlineNamespaceRangesByFile.get(filePath);
|
||||
if (set === undefined) {
|
||||
set = new Set();
|
||||
inlineNamespaceRangesByFile.set(filePath, set);
|
||||
}
|
||||
for (const r of ranges) set.add(r);
|
||||
}
|
||||
|
||||
/** Clear all inline-namespace state. Called from `clearFileLocalNames`. */
|
||||
export function clearCppInlineNamespaces(): void {
|
||||
inlineNamespaceRangesByFile.clear();
|
||||
|
|
|
|||
|
|
@ -30,6 +30,7 @@ import {
|
|||
isCppDependentBaseMember,
|
||||
} from './two-phase-lookup.js';
|
||||
import { populateCppAssociatedNamespaces, clearCppAdlState, pickCppAdlCandidates } from './adl.js';
|
||||
import { applyCppCaptureSideChannel } from './capture-side-channel.js';
|
||||
import {
|
||||
clearCppInlineNamespaces,
|
||||
populateCppInlineNamespaceScopes,
|
||||
|
|
@ -103,6 +104,24 @@ export const cppScopeResolver: ScopeResolver = {
|
|||
buildMro: (graph, parsedFiles, nodeLookup) =>
|
||||
buildMro(graph, parsedFiles, nodeLookup, defaultLinearize),
|
||||
|
||||
// Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`).
|
||||
// `emitCppScopeCaptures` records per-file ADL call-site arg shapes
|
||||
// (`markCppAdlSiteArgs`/`markCppAdlSiteNoAdl`), inline-/anonymous-namespace
|
||||
// ranges (`markCppInlineNamespaceRange`/`markCppAnonymousNamespaceRange`),
|
||||
// dependent-base names (`markCppDependentBase`/`markCppDependentPackBase`),
|
||||
// and file-local linkage (`markFileLocal`) into module-level maps as a SIDE
|
||||
// EFFECT — none of it is serialized onto the returned ParsedFile's scopes/defs.
|
||||
// On the worker path those marks are populated in the worker process and lost
|
||||
// across the MessageChannel / disk store; the main thread reuses the
|
||||
// serialized ParsedFile and skips `extractParsedFile`, so `populateOwners` +
|
||||
// the ADL / two-phase-lookup passes would see empty maps and emit zero edges.
|
||||
// The worker stashed a plain-data snapshot on `parsed.captureSideChannel` via
|
||||
// `cppProvider.collectCaptureSideChannel`; this restores it into the module
|
||||
// maps WITHOUT any tree-sitter re-parse (the #1983 fix — the old re-parse
|
||||
// replay re-OOM'd huge `.h`/`.cpp` repos). The freshly-extracted leg never
|
||||
// calls this — its marks were just populated in this process.
|
||||
applyCaptureSideChannel: applyCppCaptureSideChannel,
|
||||
|
||||
populateOwners: (parsed: ParsedFile) => {
|
||||
populateClassOwnedMembers(parsed);
|
||||
// #1982: tag namespace-nested defs with their enclosing-namespace prefix so
|
||||
|
|
|
|||
|
|
@ -104,6 +104,56 @@ export function markCppDependentPackBase(filePath: string, className: string): v
|
|||
perFile.add(className);
|
||||
}
|
||||
|
||||
/**
|
||||
* Plain-data, JSON-serializable snapshot of the per-file capture-time
|
||||
* two-phase-lookup state. Carried on `ParsedFile.captureSideChannel` across the
|
||||
* worker→main boundary (#1983). The resolved `dependentBaseNodeIds` index is
|
||||
* rebuilt by `populateCppDependentBases` (workspace pass) after all files have
|
||||
* their `populateOwners` applied, so only the two capture-time maps cross.
|
||||
*
|
||||
* Nested `Map`/`Set` are flattened to arrays here so the snapshot stays plain
|
||||
* JSON (avoids relying on the parsedfile-store's Map/Set replacer for nested
|
||||
* structures): `dependentBases` is `[className, [baseName, qualifiers[]][]][]`.
|
||||
*/
|
||||
export interface CppTwoPhaseSideChannel {
|
||||
readonly dependentBases: readonly [string, readonly [string, readonly string[]][]][];
|
||||
readonly dependentPackBaseClasses: readonly string[];
|
||||
}
|
||||
|
||||
/** Snapshot this file's two-phase-lookup capture state for the side-channel. */
|
||||
export function collectCppTwoPhaseSideChannel(filePath: string): CppTwoPhaseSideChannel {
|
||||
const perFile = dependentBasesByFile.get(filePath);
|
||||
const dependentBases: [string, [string, string[]][]][] = [];
|
||||
if (perFile !== undefined) {
|
||||
for (const [className, bases] of perFile) {
|
||||
const baseEntries: [string, string[]][] = [];
|
||||
for (const [baseName, quals] of bases) {
|
||||
baseEntries.push([baseName, [...quals]]);
|
||||
}
|
||||
dependentBases.push([className, baseEntries]);
|
||||
}
|
||||
}
|
||||
const pack = dependentPackBaseClassesByFile.get(filePath);
|
||||
return {
|
||||
dependentBases,
|
||||
dependentPackBaseClasses: pack === undefined ? [] : [...pack],
|
||||
};
|
||||
}
|
||||
|
||||
/** Restore this file's two-phase-lookup capture state from the side-channel. */
|
||||
export function applyCppTwoPhaseSideChannel(filePath: string, data: CppTwoPhaseSideChannel): void {
|
||||
for (const [className, baseEntries] of data.dependentBases) {
|
||||
for (const [baseName, quals] of baseEntries) {
|
||||
for (const qualifier of quals) {
|
||||
markCppDependentBase(filePath, className, baseName, qualifier);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const className of data.dependentPackBaseClasses) {
|
||||
markCppDependentPackBase(filePath, className);
|
||||
}
|
||||
}
|
||||
|
||||
/** Clear two-phase-lookup state. Called from `clearFileLocalNames`. */
|
||||
export function clearCppDependentBases(): void {
|
||||
dependentBasesByFile.clear();
|
||||
|
|
|
|||
|
|
@ -86,10 +86,11 @@ export function emitCsharpScopeCaptures(
|
|||
_filePath: string,
|
||||
cachedTree?: unknown,
|
||||
): readonly CaptureMatch[] {
|
||||
// Skip the parse when the caller (parse phase's scopeTreeCache)
|
||||
// already produced a Tree for this source. Cache miss = re-parse,
|
||||
// same as before. The cachedTree parameter is typed as `unknown` at
|
||||
// the LanguageProvider contract layer; cast here at the use site.
|
||||
// Reuse a pre-parsed Tree when the caller passes one via `cachedTree`; a
|
||||
// miss re-parses. (The cache is currently always empty — its only producer,
|
||||
// the sequential parser, was removed — so this re-parses in practice.) The
|
||||
// cachedTree parameter is typed `unknown` at the LanguageProvider contract
|
||||
// layer; cast here at the use site.
|
||||
let tree = cachedTree as ReturnType<ReturnType<typeof getCsharpParser>['parse']> | undefined;
|
||||
if (tree === undefined) {
|
||||
tree = parseSourceSafe(getCsharpParser(), sourceText, undefined, {
|
||||
|
|
|
|||
|
|
@ -356,8 +356,9 @@ function extractFileStructure(content: string, cachedTree: unknown): CsharpFileS
|
|||
|
||||
/** Content + (optional) pre-parsed tree-sitter trees keyed by filePath.
|
||||
* The orchestrator builds `fileContents` from the pipeline's file list;
|
||||
* `treeCache` is the same `scopeTreeCache` already populated by the
|
||||
* parse phase, so cache hits avoid a second `parser.parse()`. */
|
||||
* `treeCache` is currently always empty (its only producer, the sequential
|
||||
* parser, was removed), so the providers re-parse. Kept as an extension
|
||||
* point that would let cache hits avoid a second `parser.parse()`. */
|
||||
export interface CsharpSiblingInputs {
|
||||
readonly fileContents: ReadonlyMap<string, string>;
|
||||
readonly treeCache?: { get(filePath: string): unknown };
|
||||
|
|
|
|||
|
|
@ -28,6 +28,7 @@ import { kotlinMethodConfig } from '../method-extractors/configs/jvm.js';
|
|||
import { createVariableExtractor } from '../variable-extractors/generic.js';
|
||||
import { kotlinVariableConfig } from '../variable-extractors/configs/jvm.js';
|
||||
import {
|
||||
collectKotlinCaptureSideChannel,
|
||||
emitKotlinScopeCaptures,
|
||||
interpretKotlinImport,
|
||||
interpretKotlinTypeBinding,
|
||||
|
|
@ -175,6 +176,13 @@ export const kotlinProvider = defineLanguage({
|
|||
|
||||
// ── RFC #909 Ring 3: scope-based resolution hooks ──
|
||||
emitScopeCaptures: emitKotlinScopeCaptures,
|
||||
// Worker-side: snapshot the module-level companion-scope marks
|
||||
// `emitKotlinScopeCaptures` just populated for this file (`markCompanionScope`
|
||||
// → `companionScopesByFile`) into plain data on `ParsedFile.captureSideChannel`,
|
||||
// so the main thread can restore them via `applyCaptureSideChannel` WITHOUT a
|
||||
// re-parse (#1983). Without this, companion/static dispatch emits no CALLS
|
||||
// edges on the worker path. See `kotlin/capture-side-channel.ts`.
|
||||
collectCaptureSideChannel: collectKotlinCaptureSideChannel,
|
||||
interpretImport: interpretKotlinImport,
|
||||
interpretTypeBinding: interpretKotlinTypeBinding,
|
||||
bindingScopeFor: kotlinBindingScopeFor,
|
||||
|
|
|
|||
|
|
@ -0,0 +1,75 @@
|
|||
/**
|
||||
* Kotlin capture-time side-channel serialization (#1983).
|
||||
*
|
||||
* `emitKotlinScopeCaptures` populates one MODULE-LEVEL, per-file map as a side
|
||||
* effect that is NOT part of the returned `ParsedFile`'s scopes/defs:
|
||||
*
|
||||
* - `companionScopesByFile` (companion-scopes.ts) — the `ScopeId`s that came
|
||||
* from a `companion_object` AST node, recorded via `markCompanionScope`
|
||||
* from the `@scope.companion` marker capture.
|
||||
*
|
||||
* On the worker path that map is filled in the WORKER process and lost across
|
||||
* the worker→main MessageChannel (and the disk-backed parsedfile-store),
|
||||
* because scope-resolution reuses the serialized `ParsedFile` and SKIPS the
|
||||
* main-thread re-extraction (the #1983 fix that avoids a main-thread
|
||||
* tree-sitter re-parse / OOM on huge repos). The main thread then reads the map
|
||||
* empty in `isKotlinStaticOnly` / `populateCompanionMembersOnEnclosingClass`
|
||||
* (owners.ts) — so companion methods aren't identified as static and
|
||||
* companion/static dispatch emits no CALLS edges.
|
||||
*
|
||||
* This module snapshots the per-file slice of that map into a plain,
|
||||
* JSON-serializable object (carried on `ParsedFile.captureSideChannel`) and
|
||||
* restores it on the main thread WITHOUT any parse. It mirrors the C++ pattern
|
||||
* in `cpp/capture-side-channel.ts`.
|
||||
*
|
||||
* The single generic `ParsedFile.captureSideChannel` field is shared with C++,
|
||||
* which is safe because each file is one language (a `.kt` file uses the kotlin
|
||||
* provider, a `.cpp` file the cpp provider). The payload is self-describing
|
||||
* (`{ kind: 'kotlin', companionScopes }`) so `applyKotlinCaptureSideChannel`
|
||||
* only restores kotlin state and ignores a foreign-shaped snapshot.
|
||||
*/
|
||||
|
||||
import type { ParsedFile, ScopeId } from 'gitnexus-shared';
|
||||
import { getCompanionScopesForFile, markCompanionScope } from './companion-scopes.js';
|
||||
|
||||
/**
|
||||
* Plain JSON-serializable snapshot of the per-file Kotlin capture-time
|
||||
* side-channel. Carried opaquely on `ParsedFile.captureSideChannel`. The
|
||||
* `kind` tag makes the payload self-describing so `apply` can distinguish a
|
||||
* kotlin snapshot from another language's (C++ shares the same field).
|
||||
*/
|
||||
export interface KotlinCaptureSideChannel {
|
||||
readonly kind: 'kotlin';
|
||||
/** Companion-object scope ids recorded for this file. */
|
||||
readonly companionScopes: readonly ScopeId[];
|
||||
}
|
||||
|
||||
/**
|
||||
* `LanguageProvider.collectCaptureSideChannel` implementation for Kotlin.
|
||||
* Returns `undefined` when this file recorded no companion scopes at all, so
|
||||
* the produced `ParsedFile` carries the field only when there's data to ship.
|
||||
*/
|
||||
export function collectKotlinCaptureSideChannel(
|
||||
filePath: string,
|
||||
): KotlinCaptureSideChannel | undefined {
|
||||
const companionScopes = getCompanionScopesForFile(filePath);
|
||||
if (companionScopes.length === 0) return undefined;
|
||||
return { kind: 'kotlin', companionScopes };
|
||||
}
|
||||
|
||||
/**
|
||||
* `ScopeResolver.applyCaptureSideChannel` implementation for Kotlin. Reads the
|
||||
* worker-serialized snapshot from `parsed.captureSideChannel` and re-populates
|
||||
* the module-level companion-scope map via `markCompanionScope`. Tolerant of
|
||||
* `undefined` (file carried no data) and of an unexpected / foreign shape
|
||||
* (defensive — the `kind` tag guards against restoring a non-kotlin payload).
|
||||
* Does NO tree-sitter parse.
|
||||
*/
|
||||
export function applyKotlinCaptureSideChannel(parsed: ParsedFile): void {
|
||||
const data = parsed.captureSideChannel as KotlinCaptureSideChannel | undefined;
|
||||
if (data === undefined || data === null || typeof data !== 'object') return;
|
||||
if (data.kind !== 'kotlin' || !Array.isArray(data.companionScopes)) return;
|
||||
for (const scopeId of data.companionScopes) {
|
||||
markCompanionScope(parsed.filePath, scopeId);
|
||||
}
|
||||
}
|
||||
|
|
@ -55,6 +55,17 @@ export function isCompanionScope(filePath: string, scopeId: ScopeId): boolean {
|
|||
return companionScopesByFile.get(filePath)?.has(scopeId) ?? false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Snapshot the companion-object scope ids recorded for `filePath` as a plain
|
||||
* array (for the worker→main capture side-channel, #1983). Returns an empty
|
||||
* array when the file recorded no companion scopes. See
|
||||
* `capture-side-channel.ts`.
|
||||
*/
|
||||
export function getCompanionScopesForFile(filePath: string): ScopeId[] {
|
||||
const scopes = companionScopesByFile.get(filePath);
|
||||
return scopes === undefined ? [] : [...scopes];
|
||||
}
|
||||
|
||||
/** Clear all tracked companion scopes (for testing). */
|
||||
export function clearCompanionScopes(): void {
|
||||
companionScopesByFile.clear();
|
||||
|
|
|
|||
|
|
@ -1,4 +1,9 @@
|
|||
export { emitKotlinScopeCaptures } from './captures.js';
|
||||
export {
|
||||
collectKotlinCaptureSideChannel,
|
||||
applyKotlinCaptureSideChannel,
|
||||
type KotlinCaptureSideChannel,
|
||||
} from './capture-side-channel.js';
|
||||
export { getKotlinCaptureCacheStats, resetKotlinCaptureCacheStats } from './cache-stats.js';
|
||||
export { interpretKotlinImport, interpretKotlinTypeBinding } from './interpret.js';
|
||||
export { kotlinArityCompatibility } from './arity.js';
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ import {
|
|||
type KotlinResolveContext,
|
||||
} from './index.js';
|
||||
import { clearCompanionScopes } from './companion-scopes.js';
|
||||
import { applyKotlinCaptureSideChannel } from './capture-side-channel.js';
|
||||
import { isKotlinStaticOnly } from './owners.js';
|
||||
|
||||
/**
|
||||
|
|
@ -84,6 +85,23 @@ export const kotlinScopeResolver: ScopeResolver = {
|
|||
|
||||
buildMro: (graph, parsedFiles, nodeLookup) => buildKotlinMro(graph, parsedFiles, nodeLookup),
|
||||
|
||||
// Worker-boundary restore (see `ScopeResolver.applyCaptureSideChannel`).
|
||||
// `emitKotlinScopeCaptures` records per-file companion-object scope ids
|
||||
// (`markCompanionScope` → `companionScopesByFile`) as a SIDE EFFECT — that
|
||||
// state is NOT serialized onto the returned ParsedFile's scopes/defs. On the
|
||||
// worker path those marks are populated in the worker process and lost across
|
||||
// the MessageChannel / disk store; the main thread reuses the serialized
|
||||
// ParsedFile and skips `extractParsedFile`, so `isKotlinStaticOnly` and
|
||||
// `populateCompanionMembersOnEnclosingClass` (owners.ts) would see an empty
|
||||
// map and companion/static dispatch would emit zero CALLS edges. The worker
|
||||
// stashed a plain-data snapshot on `parsed.captureSideChannel` via
|
||||
// `kotlinProvider.collectCaptureSideChannel`; this restores it into the
|
||||
// module map WITHOUT any tree-sitter re-parse (the #1983 fix). The
|
||||
// freshly-extracted leg never calls this — its marks were just populated in
|
||||
// this process. Runs BEFORE `populateOwners` so the restored companion map is
|
||||
// visible to it.
|
||||
applyCaptureSideChannel: applyKotlinCaptureSideChannel,
|
||||
|
||||
populateOwners: (parsed: ParsedFile) => populateKotlinOwners(parsed),
|
||||
|
||||
isSuperReceiver: (text) => text.trim() === 'super',
|
||||
|
|
|
|||
|
|
@ -163,10 +163,11 @@ export function emitTsScopeCaptures(
|
|||
filePath: string,
|
||||
cachedTree?: unknown,
|
||||
): readonly CaptureMatch[] {
|
||||
// Skip the parse when the caller (parse phase's scopeTreeCache) already
|
||||
// produced a Tree for this source. Cache miss = re-parse, same as before.
|
||||
// The cachedTree parameter is typed as `unknown` at the LanguageProvider
|
||||
// contract layer; cast here at the use site.
|
||||
// Reuse a pre-parsed Tree when the caller passes one via `cachedTree`; a
|
||||
// miss re-parses. (The cache is currently always empty — its only producer,
|
||||
// the sequential parser, was removed — so this re-parses in practice.) The
|
||||
// cachedTree parameter is typed `unknown` at the LanguageProvider contract
|
||||
// layer; cast here at the use site.
|
||||
//
|
||||
// Grammar selection: `.tsx` files are parsed with the TSX grammar,
|
||||
// `.ts` files with the TypeScript grammar. The two grammars have
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -2,8 +2,9 @@
|
|||
* Parse implementation — chunked parse + resolve loop.
|
||||
*
|
||||
* This is the core parsing engine of the ingestion pipeline. It reads
|
||||
* source files in byte-budget chunks (~20MB each), parses via worker
|
||||
* pool (or sequential fallback), and emits route CALLS edges. Import,
|
||||
* source files in byte-budget chunks (~20MB each), parses via the worker
|
||||
* pool (the sole parse path — there is no sequential fallback), and emits
|
||||
* route CALLS edges. Import,
|
||||
* call, and inheritance resolution are owned by the scope-resolution
|
||||
* phase, not here (RING4-1 #942 removed the legacy call DAG; RING4-2 #943
|
||||
* removed the legacy per-file import resolution + wildcard synthesis).
|
||||
|
|
@ -19,13 +20,14 @@ import {
|
|||
enrichExportedTypeMap,
|
||||
type BindingEntry,
|
||||
} from '../binding-accumulator.js';
|
||||
import { processParsing, mergeChunkResults } from '../parsing-processor.js';
|
||||
import { mergeChunkResults, dispatchChunkParse } from '../parsing-processor.js';
|
||||
import {
|
||||
fileContentHash,
|
||||
computeChunkHash,
|
||||
loadParseCacheChunk,
|
||||
persistParseCacheChunk,
|
||||
} from '../../../storage/parse-cache.js';
|
||||
import { clearParsedFileStore, persistParsedFileChunk } from '../../../storage/parsedfile-store.js';
|
||||
import type { ParseWorkerResult } from '../workers/parse-worker.js';
|
||||
import type { WorkerExtractedData } from '../parsing-processor.js';
|
||||
import {
|
||||
|
|
@ -34,14 +36,16 @@ import {
|
|||
type ExportedTypeMap,
|
||||
} from '../call-processor.js';
|
||||
import { createSemanticModel, type MutableSemanticModel } from '../model/index.js';
|
||||
import { ASTCache, createASTCache } from '../ast-cache.js';
|
||||
import { createASTCache } from '../ast-cache.js';
|
||||
import { type PipelineProgress, getLanguageFromFilename } from 'gitnexus-shared';
|
||||
import { readFileContents } from '../filesystem-walker.js';
|
||||
import { isLanguageAvailable } from '../../tree-sitter/parser-loader.js';
|
||||
import {
|
||||
createWorkerPool,
|
||||
workerPoolDisabledByEnv,
|
||||
resolveAutoPoolSize,
|
||||
WorkerPoolInitializationError,
|
||||
WorkerPoolDisabledError,
|
||||
} from '../workers/worker-pool.js';
|
||||
import type { WorkerPool } from '../workers/worker-pool.js';
|
||||
import type {
|
||||
|
|
@ -59,7 +63,6 @@ import type {
|
|||
} from '../route-extractors/fastapi-router-bindings.js';
|
||||
import type { KnowledgeGraph } from '../../graph/types.js';
|
||||
import type { PipelineOptions } from '../pipeline.js';
|
||||
import { extractFetchCallsFromFiles } from '../call-processor.js';
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
import { fileURLToPath, pathToFileURL } from 'node:url';
|
||||
|
|
@ -73,7 +76,6 @@ import {
|
|||
startTimer,
|
||||
} from '../utils/deferred-resolution-profile.js';
|
||||
import { isDebugHeapEnabled, logHeapProbe } from '../utils/heap-probe.js';
|
||||
import { extractORMQueriesInline } from './orm-extraction.js';
|
||||
|
||||
import { logger } from '../../logger.js';
|
||||
// ── Constants ──────────────────────────────────────────────────────────────
|
||||
|
|
@ -99,12 +101,37 @@ import { logger } from '../../logger.js';
|
|||
*/
|
||||
const DEFAULT_CHUNK_BYTE_BUDGET = 2 * 1024 * 1024;
|
||||
|
||||
function resolveChunkByteBudget(options?: PipelineOptions): number {
|
||||
/**
|
||||
* Per-worker share of a chunk's byte budget when auto-scaling (#worker-idle).
|
||||
*
|
||||
* A chunk is a single `WorkerPool.dispatch` unit; the pool fans a chunk's files
|
||||
* into sub-batch jobs and assigns them to idle workers (`wakeIdleSlots`). When
|
||||
* the chunk budget (2 MB) was far below the 8 MB sub-batch cap, every chunk
|
||||
* produced exactly ONE job → ONE busy worker while the other N-1 sat idle. To
|
||||
* keep all workers fed, the auto chunk budget now scales as
|
||||
* `poolSize × CHUNK_BYTES_PER_WORKER`, so each dispatch carries enough work to
|
||||
* fan across the whole pool. Sequential / explicit-budget runs are unaffected.
|
||||
*/
|
||||
const CHUNK_BYTES_PER_WORKER = 2 * 1024 * 1024;
|
||||
|
||||
/**
|
||||
* Target jobs-per-worker per dispatch. More jobs than workers gives the pool's
|
||||
* idle-slot assignment room to load-balance (a slow job doesn't strand a worker
|
||||
* while the rest finish early). Drives the derived `subBatchMaxBytes`.
|
||||
*/
|
||||
const TARGET_JOBS_PER_WORKER = 3;
|
||||
|
||||
/** Floor for a derived sub-batch so jobs don't shrink to per-file IPC churn. */
|
||||
const MIN_SUB_BATCH_BYTES = 256 * 1024;
|
||||
|
||||
function resolveChunkByteBudget(options?: PipelineOptions, effectivePoolSize = 1): number {
|
||||
const opt = options?.chunkByteBudget;
|
||||
if (typeof opt === 'number' && Number.isFinite(opt) && opt > 0) return opt;
|
||||
const env = Number(process.env.GITNEXUS_CHUNK_BYTE_BUDGET);
|
||||
if (Number.isFinite(env) && env > 0) return env;
|
||||
return DEFAULT_CHUNK_BYTE_BUDGET;
|
||||
// Auto: size each chunk so a dispatch can fan across the whole pool. A
|
||||
// single-worker (tiny-repo) run keeps the original 2 MB invalidation floor.
|
||||
return Math.max(DEFAULT_CHUNK_BYTE_BUDGET, effectivePoolSize * CHUNK_BYTES_PER_WORKER);
|
||||
}
|
||||
|
||||
// ── Main parse + resolve function ──────────────────────────────────────────
|
||||
|
|
@ -120,15 +147,13 @@ type ProgressFn = (progress: PipelineProgress) => void;
|
|||
* was detected, or the pool could not even be constructed. In every such case
|
||||
* the workers genuinely cannot start.
|
||||
*
|
||||
* Rather than silently degrade to the ~10× slower sequential parser — which
|
||||
* masked a worker-startup regression as a 2-hour "stuck" run in #1741 (rc99:
|
||||
* the failure was a dropped `logger.warn` and an unbounded sequential grind) —
|
||||
* GitNexus surfaces the real crash and aborts. An operator who genuinely wants
|
||||
* sequential parsing asks for it explicitly with `--workers 0`.
|
||||
*
|
||||
* The decision is automatic: NO `--allow-sequential-fallback` or pool-sizing
|
||||
* flag participates. The pool's own crash classification (`crashClass` on
|
||||
* WorkerPoolInitializationError) only sharpens the message.
|
||||
* There is no sequential parser to silently degrade to — that fallback was
|
||||
* removed (and it had masked a worker-startup regression as a 2-hour "stuck"
|
||||
* run in #1741, rc99: a dropped `logger.warn` plus an unbounded sequential
|
||||
* grind). GitNexus surfaces the real crash and aborts so the operator fixes the
|
||||
* worker startup (commonly a missing build). The pool's own crash
|
||||
* classification (`crashClass` on WorkerPoolInitializationError) sharpens the
|
||||
* message.
|
||||
*
|
||||
* @throws always — an actionable Error carrying the captured worker crash.
|
||||
* @internal Exported for unit tests; production callers are the parse loop's
|
||||
|
|
@ -174,11 +199,10 @@ export function handleWorkerStartupFailure(err: Error): never {
|
|||
|
||||
throw new Error(
|
||||
`Worker pool failed to start: ${cause}${failureDetail}\n\n` +
|
||||
`GitNexus will NOT silently fall back to the (much slower) sequential ` +
|
||||
`parser and hide this crash — that masked a worker-startup regression as ` +
|
||||
`a 2-hour "stuck" run in #1741. Options:\n` +
|
||||
` • ${fixHint}\n` +
|
||||
` • Re-run with --workers 0 to parse sequentially without the worker pool.`,
|
||||
`The worker pool is GitNexus's only parse path — there is no sequential ` +
|
||||
`fallback to hide this crash behind (silently degrading masked a ` +
|
||||
`worker-startup regression as a 2-hour "stuck" run in #1741). Fix:\n` +
|
||||
` • ${fixHint}`,
|
||||
);
|
||||
}
|
||||
|
||||
|
|
@ -186,7 +210,7 @@ export function handleWorkerStartupFailure(err: Error): never {
|
|||
* Chunked parse + resolve loop.
|
||||
*
|
||||
* Reads source in byte-budget chunks (~20MB each):
|
||||
* 1. Parse each chunk via worker pool (or sequential fallback)
|
||||
* 1. Parse each chunk via the worker pool (the sole parse path)
|
||||
* 2. After all chunks parse, emit route CALLS edges (deferred so resolution
|
||||
* sees the full repo graph) and collect the exported-type map
|
||||
* 3. Collect TypeEnv bindings for cross-file propagation
|
||||
|
|
@ -215,16 +239,12 @@ export async function runChunkedParseAndResolve(
|
|||
/** SemanticModel populated during parse — scope-resolution reads its
|
||||
* TypeRegistry / MethodRegistry / SymbolTable indexes. */
|
||||
model: MutableSemanticModel;
|
||||
/** Whether a worker pool was actually constructed for this run. False
|
||||
* means no pool was needed: a warm all-cache-hit run replays cached
|
||||
* worker output without spawning workers, or there were no parseable
|
||||
* files. There is no sequential parser — the pool is the sole parse path
|
||||
* whenever a chunk misses the cache. */
|
||||
usedWorkerPool: boolean;
|
||||
/** Cross-phase tree-sitter Tree cache populated by the sequential
|
||||
* parse path. Distinct from the chunk-local `astCache` used inside
|
||||
* the parse loop (that one is cleared between chunks). Empty when
|
||||
* every chunk ran via the worker pool (workers can't return native
|
||||
* tree-sitter Trees across the MessageChannel). Downstream phases
|
||||
* (scope-resolution) read from this to skip re-parsing the same
|
||||
* source. See plan
|
||||
* docs/plans/2026-04-20-002-perf-parse-heritage-mro-plan.md (Unit 4). */
|
||||
scopeTreeCache: ASTCache;
|
||||
/** Worker-produced ParsedFile artifacts aggregated across chunks.
|
||||
* Threaded into scope-resolution as a re-extract cache so the warm-
|
||||
* cache analyze run can skip the dominant `extractParsedFile` cost
|
||||
|
|
@ -263,6 +283,7 @@ export async function runChunkedParseAndResolve(
|
|||
parseableScanned.sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0));
|
||||
|
||||
const totalParseable = parseableScanned.length;
|
||||
const totalBytes = parseableScanned.reduce((sum, f) => sum + f.size, 0);
|
||||
|
||||
if (totalParseable === 0) {
|
||||
onProgress({
|
||||
|
|
@ -276,6 +297,30 @@ export async function runChunkedParseAndResolve(
|
|||
});
|
||||
}
|
||||
|
||||
// Sequential parsing has been removed: the worker pool (quarantine +
|
||||
// respawn/recycle + circuit breaker) is the sole parse path. The three
|
||||
// channels that used to select an in-process parser are now hard errors, so
|
||||
// an operator who set one gets an actionable message instead of a silently
|
||||
// slower (now nonexistent) fallback. Validated before any chunk work; a
|
||||
// zero-parseable-file repo is exempt (nothing to parse).
|
||||
if (totalParseable > 0) {
|
||||
const requestedPoolSize = options?.workerPoolSize;
|
||||
const disabledByEnv = requestedPoolSize === undefined && workerPoolDisabledByEnv();
|
||||
if (options?.skipWorkers || requestedPoolSize === 0 || disabledByEnv) {
|
||||
const reason = options?.skipWorkers
|
||||
? '`skipWorkers: true` was passed'
|
||||
: requestedPoolSize === 0
|
||||
? '`--workers 0` (workerPoolSize=0) was requested'
|
||||
: '`GITNEXUS_WORKER_POOL_SIZE=0` is set';
|
||||
throw new WorkerPoolDisabledError(
|
||||
`Worker-pool parsing cannot be disabled (${reason}). GitNexus no longer ` +
|
||||
`has a sequential parser — the worker pool self-heals via quarantine + ` +
|
||||
`respawn, so there is no slower path to fall back to. Pass ` +
|
||||
`\`--workers <N>\` with N>=1, or omit it for an auto-sized pool.`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Build byte-budget chunks. The budget is resolved per-call (U14): options
|
||||
// first, then env, then the built-in default. Pre-U14 this was a
|
||||
// module-load IIFE constant, which froze the env value at import time
|
||||
|
|
@ -283,7 +328,33 @@ export async function runChunkedParseAndResolve(
|
|||
// runs. Resolving in the function body restores per-call configurability
|
||||
// and matches the pattern used by resolveAutoPoolSize and the U1
|
||||
// parseChunkConcurrency resolver.
|
||||
const chunkByteBudget = resolveChunkByteBudget(options);
|
||||
// Effective worker count, computed up-front so the chunk budget can scale to
|
||||
// keep the whole pool busy (#worker-idle). The pool is ALWAYS used (sequential
|
||||
// parsing was removed; the disabled channels threw above). Size it to the
|
||||
// work: an explicit `--workers <N>` pins the size; otherwise the cores-based
|
||||
// auto size is capped by the repo's worth of work (~one worker per
|
||||
// CHUNK_BYTES_PER_WORKER of source) so a tiny repo spawns ~1 worker instead of
|
||||
// a full pool, replacing the job the deleted small-repo threshold used to do.
|
||||
// KTD-3 of the remove-sequential plan; the cap formula is intentionally coarse
|
||||
// (tuning deferred).
|
||||
const explicitPoolSize = options?.workerPoolSize;
|
||||
const workProportionalCap = Math.max(1, Math.ceil(totalBytes / CHUNK_BYTES_PER_WORKER));
|
||||
const effectivePoolSize =
|
||||
explicitPoolSize && explicitPoolSize > 0
|
||||
? explicitPoolSize
|
||||
: Math.min(resolveAutoPoolSize(), workProportionalCap);
|
||||
const chunkByteBudget = resolveChunkByteBudget(options, effectivePoolSize);
|
||||
// Sub-batch size so each chunk fans into ~`TARGET_JOBS_PER_WORKER` jobs per
|
||||
// worker, giving the pool's idle-slot assignment room to load-balance. An
|
||||
// explicit `GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES` operator override wins.
|
||||
const subBatchEnv = Number(process.env.GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES);
|
||||
const dispatchSubBatchMaxBytes =
|
||||
Number.isFinite(subBatchEnv) && subBatchEnv > 0
|
||||
? subBatchEnv
|
||||
: Math.max(
|
||||
MIN_SUB_BATCH_BYTES,
|
||||
Math.ceil(chunkByteBudget / (effectivePoolSize * TARGET_JOBS_PER_WORKER)),
|
||||
);
|
||||
const chunks: string[][] = [];
|
||||
let currentChunk: string[] = [];
|
||||
let currentBytes = 0;
|
||||
|
|
@ -320,53 +391,24 @@ export async function runChunkedParseAndResolve(
|
|||
});
|
||||
}
|
||||
|
||||
// Don't spawn workers for tiny repos — overhead exceeds benefit.
|
||||
// Test suites may lower the thresholds via `options.workerThresholdsForTest`
|
||||
// to exercise the worker-pool path with small fixtures; see PipelineOptions.
|
||||
const MIN_FILES_FOR_WORKERS = options?.workerThresholdsForTest?.minFiles ?? 15;
|
||||
const MIN_BYTES_FOR_WORKERS = options?.workerThresholdsForTest?.minBytes ?? 512 * 1024;
|
||||
const totalBytes = parseableScanned.reduce((s, f) => s + f.size, 0);
|
||||
|
||||
// Create worker pool lazily, reuse across cache-miss chunks.
|
||||
// Create the worker pool lazily, reusing it across cache-miss chunks.
|
||||
//
|
||||
// `workerPoolSize === 0` is a programmatic equivalent of `skipWorkers:
|
||||
// true` per the `PipelineOptions.workerPoolSize` contract. Short-
|
||||
// circuiting here avoids constructing a useless pool. The pool is
|
||||
// intentionally NOT created before parse-cache lookup: a warm-cache
|
||||
// all-hit run should replay cached worker output without loading
|
||||
// parse-worker.js or any tree-sitter/N-API native bindings.
|
||||
// `--workers 0` (workerPoolSize === 0) and `GITNEXUS_WORKER_POOL_SIZE=0` both
|
||||
// mean "no pool, parse sequentially". The env channel is consulted ONLY when
|
||||
// no explicit `--workers <N>` was given, so an explicit positive size always
|
||||
// wins over an ambient env=0 (#1741). Without this, env=0 built a size-0 pool
|
||||
// that failed fast with a fabricated "retry budget exhausted" crash.
|
||||
const envDisablesWorkers = options?.workerPoolSize === undefined && workerPoolDisabledByEnv();
|
||||
const meetsWorkerThreshold =
|
||||
totalParseable >= MIN_FILES_FOR_WORKERS || totalBytes >= MIN_BYTES_FOR_WORKERS;
|
||||
// Log only when env=0 actually skips a pool we'd otherwise have used, so the
|
||||
// undocumented (possibly accidental) env=0 case is observable instead of a
|
||||
// silent degrade — small repos go sequential anyway and need no notice.
|
||||
if (envDisablesWorkers && meetsWorkerThreshold) {
|
||||
logger.warn(
|
||||
'GITNEXUS_WORKER_POOL_SIZE=0 → parsing sequentially; unset it or pass --workers <N> to use the worker pool.',
|
||||
);
|
||||
}
|
||||
const shouldUseWorkers =
|
||||
!options?.skipWorkers &&
|
||||
options?.workerPoolSize !== 0 &&
|
||||
!envDisablesWorkers &&
|
||||
meetsWorkerThreshold;
|
||||
// KTD-8 — the pool is intentionally NOT created before the parse-cache
|
||||
// lookup: a warm-cache all-hit run must replay cached worker output without
|
||||
// loading parse-worker.js or any tree-sitter/N-API native bindings. So
|
||||
// `getOrCreateWorkerPool` is called only from inside the chunk loop, on the
|
||||
// first cache MISS. There is no longer a "should we use workers?" gate:
|
||||
// sequential parsing was removed and the disabled channels (`--workers 0` /
|
||||
// env=0 / `skipWorkers`) threw above, so for any repo with parseable files
|
||||
// the pool is always the parse path.
|
||||
let workerPool: WorkerPool | undefined;
|
||||
const getOrCreateWorkerPool = (): WorkerPool | undefined => {
|
||||
if (!shouldUseWorkers) return undefined;
|
||||
const getOrCreateWorkerPool = (): WorkerPool => {
|
||||
if (workerPool) return workerPool;
|
||||
try {
|
||||
// U20.U3 test-only injection: integration tests pass a custom
|
||||
// worker script URL via `workerUrlForTest` (mirrors the
|
||||
// `workerThresholdsForTest` precedent) so they can drive the
|
||||
// chunk-loop with deterministically-misbehaving workers without
|
||||
// mocking the module import graph. When unset, the normal src/
|
||||
// → dist/ resolution runs.
|
||||
// Test-only injection: integration tests pass a custom worker script URL
|
||||
// via `workerUrlForTest` so they can drive the chunk-loop with
|
||||
// deterministically-misbehaving workers without mocking the module import
|
||||
// graph. When unset, the normal src/ → dist/ resolution runs.
|
||||
let workerUrl =
|
||||
options?.workerUrlForTest ?? new URL('../workers/parse-worker.js', import.meta.url);
|
||||
// When running under vitest, import.meta.url points to src/ where no .js exists.
|
||||
|
|
@ -389,33 +431,38 @@ export async function runChunkedParseAndResolve(
|
|||
workerUrl = pathToFileURL(distWorker);
|
||||
}
|
||||
}
|
||||
workerPool = createWorkerPool(workerUrl, options?.workerPoolSize);
|
||||
// Thread the ParsedFile store path into the pool so workers write their
|
||||
// own shards (#1983 parallel serialization). `parsedFileStorePath` is
|
||||
// declared below but this closure only runs from inside the chunk loop,
|
||||
// after it is initialized; `undefined` drives the worker no-store
|
||||
// fallback (return ParsedFiles in the result).
|
||||
workerPool = createWorkerPool(workerUrl, effectivePoolSize, {
|
||||
parsedFileStoreStoragePath: parsedFileStorePath,
|
||||
// Fan each chunk across the whole pool (#worker-idle): without this a
|
||||
// chunk smaller than the 8 MB sub-batch cap became a single job on a
|
||||
// single worker. Honors an explicit `subBatchMaxBytes` / env override.
|
||||
subBatchMaxBytes: dispatchSubBatchMaxBytes,
|
||||
});
|
||||
return workerPool;
|
||||
} catch (err) {
|
||||
// Pool *construction* failed (e.g. the worker script is missing — a
|
||||
// broken install). Fail fast with the cause rather than silently
|
||||
// degrading to the slow sequential parser (#1741); `--workers 0` is the
|
||||
// explicit opt-out for anyone who genuinely wants sequential parsing.
|
||||
// broken install). Fail fast with the cause (#1741). There is no
|
||||
// sequential parser to fall back to; the operator must fix the worker
|
||||
// startup (commonly a missing build so dist/ has no parse-worker).
|
||||
handleWorkerStartupFailure(err as Error);
|
||||
}
|
||||
};
|
||||
|
||||
let filesParsedSoFar = 0;
|
||||
|
||||
// Two caches with different lifetimes:
|
||||
// - `astCache` (chunk-local, cleared between chunks) — call /
|
||||
// heritage / import processors read it during parse to avoid
|
||||
// re-parsing within the same chunk.
|
||||
// - `scopeTreeCache` (total-parseable-sized, never cleared by
|
||||
// parse-impl) — exposed via ParseOutput so scope-resolution can
|
||||
// skip a second tree-sitter parse. Worker-mode parses don't
|
||||
// populate either; consumers fall back to a fresh parse.
|
||||
// See plan docs/plans/2026-04-20-002-perf-parse-heritage-mro-plan.md (Unit 4).
|
||||
// Chunk-local tree-sitter cache, cleared between chunks — call / heritage /
|
||||
// import processors read it during parse to avoid re-parsing within the same
|
||||
// chunk. (The former cross-phase `scopeTreeCache` was only ever populated by
|
||||
// the sequential parser, which has been removed; workers can't return native
|
||||
// Trees across the MessageChannel, so scope-resolution re-parses as needed.)
|
||||
const maxChunkFiles = chunks.reduce((max, c) => Math.max(max, c.length), 0);
|
||||
let astCache = createASTCache(maxChunkFiles);
|
||||
const scopeTreeCache = createASTCache(Math.max(parseableScanned.length, 1));
|
||||
const astCache = createASTCache(maxChunkFiles);
|
||||
|
||||
const sequentialChunkPaths: string[][] = [];
|
||||
const exportedTypeMap: ExportedTypeMap = new Map();
|
||||
const bindingAccumulator = new BindingAccumulator();
|
||||
const allFetchCalls: ExtractedFetchCall[] = [];
|
||||
|
|
@ -439,6 +486,17 @@ export async function runChunkedParseAndResolve(
|
|||
// run's, replay the cached ParseWorkerResult[] instead of dispatching
|
||||
// to workers. See gitnexus/src/storage/parse-cache.ts.
|
||||
const parseCache = options?.parseCache;
|
||||
// Disk-backed ParsedFile store (#1983): when a storage path is available we
|
||||
// flush worker-produced ParsedFiles to disk per chunk (instead of retaining
|
||||
// them in `allParsedFiles`) and scope-resolution streams them back per
|
||||
// language — avoiding both the ~1× semantic-model RAM cost of holding them
|
||||
// and, critically, the unbounded native tree-sitter re-parse leak that
|
||||
// scope-resolution's main-thread re-extraction otherwise accumulates. When
|
||||
// there is no storage path (tests / direct pipeline calls), we fall back to
|
||||
// retaining them in `allParsedFiles` (small-repo path, preserves prior
|
||||
// behavior). Cleared up-front so a prior run's shards never leak in.
|
||||
const parsedFileStorePath = parseCache?.storagePath;
|
||||
if (parsedFileStorePath) await clearParsedFileStore(parsedFileStorePath);
|
||||
let chunkCacheHits = 0;
|
||||
let chunkCacheMisses = 0;
|
||||
|
||||
|
|
@ -478,192 +536,49 @@ export async function runChunkedParseAndResolve(
|
|||
const verboseThroughputLog = isDev || isVerboseIngestionEnabled();
|
||||
const heapProbeEveryN = isDebugHeapEnabled() ? 25 : 0;
|
||||
|
||||
for (let chunkIdx = 0; chunkIdx < numChunks; chunkIdx++) {
|
||||
if (heapProbeEveryN > 0 && chunkIdx > 0 && chunkIdx % heapProbeEveryN === 0) {
|
||||
logHeapProbe(
|
||||
`parse-chunk-${chunkIdx}`,
|
||||
`nodes=${graph.nodeCount} parsedFiles=${allParsedFiles.length}`,
|
||||
);
|
||||
}
|
||||
const chunkPaths = chunks[chunkIdx];
|
||||
// Start wall-clock for the per-chunk throughput log emitted at end
|
||||
// of this iteration. The gate is computed once above; here we just
|
||||
// sample the clock if the gate is on. Computed when either
|
||||
// NODE_ENV=development OR the operator passed `--verbose`
|
||||
// (GITNEXUS_VERBOSE) — the previous `isDev`-only gate meant
|
||||
// operators running `gitnexus analyze --verbose` in production
|
||||
// never saw the log (M3 from PR #1693 review).
|
||||
const chunkStartMs: number | null = verboseThroughputLog ? Date.now() : null;
|
||||
// ── Merge pipelining (#worker-idle) ──────────────────────────────────────
|
||||
// Merging a chunk's worker results into the graph is the only remaining
|
||||
// serial main-thread step (ParsedFile serialization now runs in workers).
|
||||
// To stop the whole pool idling during that merge, we OVERLAP it with the
|
||||
// NEXT chunk's worker parse: a freshly-dispatched worker chunk is parked in
|
||||
// `pendingWorkerChunk`, and we merge+finalize it only AFTER starting the
|
||||
// following chunk's dispatch — so the workers parse chunk N+1 while the
|
||||
// main thread merges chunk N. Chunk ORDER is preserved (N finalized before
|
||||
// N+1), which keeps the deferred aggregation deterministic. Cache-hit
|
||||
// chunks drain any pending chunk first, then finalize inline (no worker
|
||||
// dispatch to overlap).
|
||||
interface PendingWorkerChunk {
|
||||
readonly rawResults: ParseWorkerResult[];
|
||||
readonly chunkIdx: number;
|
||||
readonly chunkHash: string | null;
|
||||
readonly chunkFiles: Array<{ path: string; content: string }>;
|
||||
readonly chunkStartMs: number | null;
|
||||
}
|
||||
let pendingWorkerChunk: PendingWorkerChunk | null = null;
|
||||
|
||||
const chunkContentPromise = chunkContentPromises[chunkIdx];
|
||||
if (!chunkContentPromise) {
|
||||
throw new Error(`Missing prefetched parse chunk ${chunkIdx + 1}/${numChunks}`);
|
||||
}
|
||||
const chunkContents = await chunkContentPromise;
|
||||
chunkContentPromises[chunkIdx] = undefined; // release the in-memory copy
|
||||
startChunkPrefetch(chunkIdx + parseChunkConcurrency);
|
||||
const chunkFiles: Array<{ path: string; content: string }> = [];
|
||||
for (const p of chunkPaths) {
|
||||
const content = chunkContents.get(p);
|
||||
if (content !== undefined) chunkFiles.push({ path: p, content });
|
||||
}
|
||||
|
||||
// Compute the chunk's content-hash signature (if cache available).
|
||||
let chunkHash: string | null = null;
|
||||
if (parseCache) {
|
||||
const entries = chunkFiles.map((f) => ({
|
||||
filePath: f.path,
|
||||
contentHash: fileContentHash(f.content),
|
||||
}));
|
||||
chunkHash = computeChunkHash(entries);
|
||||
}
|
||||
|
||||
let chunkWorkerData: WorkerExtractedData | null;
|
||||
const cachedRaw =
|
||||
chunkHash && parseCache ? await loadParseCacheChunk(parseCache, chunkHash) : undefined;
|
||||
|
||||
// Track every chunk hash we touched so the orchestrator can
|
||||
// prune stale entries (chunks whose composition no longer
|
||||
// corresponds to a live chunk in the current scan) before saving.
|
||||
if (parseCache && chunkHash) parseCache.usedKeys.add(chunkHash);
|
||||
|
||||
if (cachedRaw && cachedRaw.length > 0) {
|
||||
// Cache hit: replay the cached worker output through the same
|
||||
// merge logic the live worker path uses.
|
||||
chunkCacheHits++;
|
||||
chunkWorkerData = mergeChunkResults(graph, symbolTable, cachedRaw, exportedTypeMap);
|
||||
if (isDev) {
|
||||
logger.info(
|
||||
`📦 parse-cache HIT: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash?.slice(0, 8) ?? 'unknown'})`,
|
||||
);
|
||||
}
|
||||
// Progress update so UI advances even on a cache hit.
|
||||
const cachedFiles = chunkFiles.length;
|
||||
onProgress({
|
||||
phase: 'parsing',
|
||||
// Parse phase covers 20-70 (50 points). Deferred extraction below
|
||||
// takes 70-95 so the UI advances through the (potentially long)
|
||||
// resolution stages instead of holding at 82 (M2 from PR #1693
|
||||
// review).
|
||||
percent: Math.round(20 + ((filesParsedSoFar + cachedFiles) / totalParseable) * 50),
|
||||
message: `Parsing chunk ${chunkIdx + 1}/${numChunks} (cache)...`,
|
||||
stats: {
|
||||
filesProcessed: filesParsedSoFar + cachedFiles,
|
||||
totalFiles: totalParseable,
|
||||
nodesCreated: graph.nodeCount,
|
||||
},
|
||||
});
|
||||
} else {
|
||||
// Cache miss: dispatch to workers, capture the raw results, store
|
||||
// them under the chunk hash for the next run.
|
||||
chunkCacheMisses++;
|
||||
const rawResults: ParseWorkerResult[] = [];
|
||||
const progressForChunk = (current: number, _total: number, filePath: string) => {
|
||||
const globalCurrent = filesParsedSoFar + current;
|
||||
// Parse phase covers 20-70 (M2). Deferred extraction handles 70-95.
|
||||
const parsingProgress = 20 + (globalCurrent / totalParseable) * 50;
|
||||
onProgress({
|
||||
phase: 'parsing',
|
||||
percent: Math.round(parsingProgress),
|
||||
message: `Parsing chunk ${chunkIdx + 1}/${numChunks}...`,
|
||||
detail: filePath,
|
||||
stats: {
|
||||
filesProcessed: globalCurrent,
|
||||
totalFiles: totalParseable,
|
||||
nodesCreated: graph.nodeCount,
|
||||
},
|
||||
});
|
||||
};
|
||||
const activeWorkerPool = getOrCreateWorkerPool();
|
||||
try {
|
||||
chunkWorkerData = await processParsing(
|
||||
graph,
|
||||
chunkFiles,
|
||||
symbolTable,
|
||||
astCache,
|
||||
scopeTreeCache,
|
||||
progressForChunk,
|
||||
activeWorkerPool,
|
||||
// Capture raw results only when we have a cache to write to —
|
||||
// otherwise we'd retain extra arrays for nothing.
|
||||
parseCache && chunkHash && activeWorkerPool ? rawResults : undefined,
|
||||
exportedTypeMap,
|
||||
);
|
||||
} catch (err) {
|
||||
if (!(err instanceof WorkerPoolInitializationError)) throw err;
|
||||
// Every worker crashed during startup and the pool's bounded
|
||||
// self-heal (jittered restart, deterministic crash-loop detection —
|
||||
// see worker-pool.ts) was exhausted. Fail fast with the captured
|
||||
// cause rather than silently degrading to the ~10× slower sequential
|
||||
// parser, which masked this exact regression as a 2-hour "stuck" run
|
||||
// in #1741. The failed (zero-worker) pool is torn down by the outer
|
||||
// finally. `--workers 0` is the explicit opt-in to sequential.
|
||||
rawResults.length = 0;
|
||||
handleWorkerStartupFailure(err); // always throws
|
||||
}
|
||||
// Persist the raw results for this chunk hash. Sequential path
|
||||
// doesn't populate rawResults (it writes directly to graph), so
|
||||
// small repos without worker pool simply don't cache. That's fine.
|
||||
//
|
||||
// U20.U2: refuse the write when any chunk file is in the
|
||||
// worker pool's cumulative quarantine snapshot. The chunkHash
|
||||
// is computed from EVERY file in the chunk, but the pool's
|
||||
// Layer 3 quarantine filters quarantined files out of dispatch
|
||||
// — so `rawResults` is narrower than the chunkHash key implies.
|
||||
// Caching it would silently replay incomplete results on the
|
||||
// next run with unchanged content (the corruption class Codex's
|
||||
// adversarial review of PR #1693 flagged).
|
||||
//
|
||||
// Skipping the write means the next analyze gets a cache miss
|
||||
// for this chunk and re-dispatches against a fresh worker pool
|
||||
// (quarantine is session-scoped — `createQuarantine` is called
|
||||
// per-pool at worker-pool.ts), giving the quarantined file
|
||||
// another chance. If quarantine fires again, U20.U1's
|
||||
// sequential gap-fill still produces a complete graph for this
|
||||
// run; the cache just stays empty for this chunk until a fully-
|
||||
// clean dispatch lands.
|
||||
if (parseCache && chunkHash && rawResults.length > 0) {
|
||||
const quarantineSnapshot = workerPool?.getQuarantinedPaths?.() ?? [];
|
||||
const quarantineSet = new Set(quarantineSnapshot);
|
||||
const chunkHadQuarantine = chunkFiles.some((f) => quarantineSet.has(f.path));
|
||||
if (chunkHadQuarantine) {
|
||||
if (isDev) {
|
||||
const quarantinedInChunk = chunkFiles.filter((f) => quarantineSet.has(f.path)).length;
|
||||
logger.info(
|
||||
`📦 parse-cache SKIP: chunk ${chunkIdx + 1}/${numChunks} ` +
|
||||
`had ${quarantinedInChunk} worker-quarantined file(s); ` +
|
||||
`next run will rediscover (${chunkHash.slice(0, 8)})`,
|
||||
);
|
||||
}
|
||||
// Apply one chunk's merged worker data: per-chunk aggregation into the
|
||||
// run-level accumulators + the throughput log. Shared by the cache-hit
|
||||
// (inline) and worker (deferred) paths. The `| null` guard is defensive —
|
||||
// every live caller passes real worker data now that sequential parsing
|
||||
// (which was the only path that passed null) is gone.
|
||||
const applyChunkResults = async (
|
||||
chunkWorkerData: WorkerExtractedData | null,
|
||||
chunkIdx: number,
|
||||
chunkFiles: Array<{ path: string; content: string }>,
|
||||
chunkStartMs: number | null,
|
||||
): Promise<void> => {
|
||||
if (chunkWorkerData) {
|
||||
if (chunkWorkerData.parsedFiles?.length) {
|
||||
if (parsedFileStorePath) {
|
||||
await persistParsedFileChunk(
|
||||
parsedFileStorePath,
|
||||
`chunk-${chunkIdx}`,
|
||||
chunkWorkerData.parsedFiles,
|
||||
);
|
||||
} else {
|
||||
await persistParseCacheChunk(parseCache, chunkHash, rawResults);
|
||||
if (isDev) {
|
||||
logger.info(
|
||||
`📦 parse-cache MISS+store: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash.slice(0, 8)})`,
|
||||
);
|
||||
}
|
||||
for (const item of chunkWorkerData.parsedFiles) allParsedFiles.push(item);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Route resolution is moved out of the chunk loop into a single
|
||||
// end-of-loop pass below. (Import resolution and wildcard synthesis
|
||||
// used to run here too; they were removed in RING4-2 #943 — IMPORTS
|
||||
// edges now come from the scope-resolution phase.)
|
||||
// Reason: per-chunk extraction blocked the chunk loop on
|
||||
// main-thread work between worker dispatches — workers sat idle
|
||||
// and total CPU utilization plateaued at 4-5% on multi-core boxes.
|
||||
// Deferring keeps workers busy chunk-after-chunk; route resolution
|
||||
// sees strictly-more-information (full repo graph) so cross-chunk
|
||||
// controller targets resolve at least as well as before.
|
||||
if (chunkWorkerData) {
|
||||
// Aggregate worker-produced ParsedFile artifacts so scope-
|
||||
// resolution can use them as a re-extraction cache (skips its
|
||||
// own tree-sitter re-parse on warm runs).
|
||||
if (chunkWorkerData.parsedFiles?.length) {
|
||||
for (const item of chunkWorkerData.parsedFiles) allParsedFiles.push(item);
|
||||
}
|
||||
|
||||
if (chunkWorkerData.fileScopeBindings?.length) {
|
||||
for (const { filePath, bindings } of chunkWorkerData.fileScopeBindings) {
|
||||
if (typeof filePath !== 'string' || filePath.length === 0) continue;
|
||||
|
|
@ -707,17 +622,11 @@ export async function runChunkedParseAndResolve(
|
|||
if (chunkWorkerData.ormQueries?.length) {
|
||||
for (const item of chunkWorkerData.ormQueries) allORMQueries.push(item);
|
||||
}
|
||||
} else {
|
||||
sequentialChunkPaths.push(chunkPaths);
|
||||
}
|
||||
|
||||
filesParsedSoFar += chunkFiles.length;
|
||||
astCache.clear();
|
||||
|
||||
// Throughput observability (U3): emit a per-chunk metrics line
|
||||
// under verbose ingestion mode so operators can verify CPU
|
||||
// utilization moved + tune `--workers` / batch sizes without
|
||||
// guessing. Cheap snapshot — just reads pool closure state.
|
||||
if (verboseThroughputLog && chunkStartMs !== null) {
|
||||
const elapsedMs = Date.now() - chunkStartMs;
|
||||
const filesPerSec = elapsedMs > 0 ? (chunkFiles.length * 1000) / elapsedMs : 0;
|
||||
|
|
@ -725,12 +634,192 @@ export async function runChunkedParseAndResolve(
|
|||
const poolFrag = stats
|
||||
? ` pool: ${stats.activeSlots}/${stats.size} active, ` +
|
||||
`${stats.quarantined} quarantined${stats.poolBroken ? ', BROKEN' : ''}`
|
||||
: ' (sequential)';
|
||||
: ' (cache replay)';
|
||||
logger.info(
|
||||
`📊 chunk ${chunkIdx + 1}/${numChunks}: ${chunkFiles.length} files in ${elapsedMs}ms ` +
|
||||
`(${filesPerSec.toFixed(1)} files/s)${poolFrag}`,
|
||||
);
|
||||
}
|
||||
};
|
||||
|
||||
// Merge + finalize a parked worker chunk: graph merge (the overlapped
|
||||
// main-thread step) → parse-cache write-guard → run-level aggregation.
|
||||
const finalizeWorkerChunk = async (p: PendingWorkerChunk): Promise<void> => {
|
||||
const chunkWorkerData = mergeChunkResults(graph, symbolTable, p.rawResults, exportedTypeMap);
|
||||
// Persist raw results for this chunk hash (skipping when any chunk file
|
||||
// was worker-quarantined, so the narrower rawResults isn't cached under
|
||||
// the full-chunk key — see the original inline note / U20.U2).
|
||||
if (parseCache && p.chunkHash && p.rawResults.length > 0) {
|
||||
const quarantineSet = new Set(workerPool?.getQuarantinedPaths?.() ?? []);
|
||||
const chunkHadQuarantine = p.chunkFiles.some((f) => quarantineSet.has(f.path));
|
||||
if (chunkHadQuarantine) {
|
||||
if (isDev) {
|
||||
const quarantinedInChunk = p.chunkFiles.filter((f) => quarantineSet.has(f.path)).length;
|
||||
logger.info(
|
||||
`📦 parse-cache SKIP: chunk ${p.chunkIdx + 1}/${numChunks} ` +
|
||||
`had ${quarantinedInChunk} worker-quarantined file(s); ` +
|
||||
`next run will rediscover (${p.chunkHash.slice(0, 8)})`,
|
||||
);
|
||||
}
|
||||
} else {
|
||||
await persistParseCacheChunk(parseCache, p.chunkHash, p.rawResults);
|
||||
if (isDev) {
|
||||
logger.info(
|
||||
`📦 parse-cache MISS+store: chunk ${p.chunkIdx + 1}/${numChunks} (${p.chunkFiles.length} files, ${p.chunkHash.slice(0, 8)})`,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
await applyChunkResults(chunkWorkerData, p.chunkIdx, p.chunkFiles, p.chunkStartMs);
|
||||
};
|
||||
|
||||
for (let chunkIdx = 0; chunkIdx < numChunks; chunkIdx++) {
|
||||
if (heapProbeEveryN > 0 && chunkIdx > 0 && chunkIdx % heapProbeEveryN === 0) {
|
||||
logHeapProbe(
|
||||
`parse-chunk-${chunkIdx}`,
|
||||
`nodes=${graph.nodeCount} parsedFiles=${allParsedFiles.length}`,
|
||||
);
|
||||
}
|
||||
const chunkPaths = chunks[chunkIdx];
|
||||
// Start wall-clock for the per-chunk throughput log emitted at end
|
||||
// of this iteration. The gate is computed once above; here we just
|
||||
// sample the clock if the gate is on. Computed when either
|
||||
// NODE_ENV=development OR the operator passed `--verbose`
|
||||
// (GITNEXUS_VERBOSE) — the previous `isDev`-only gate meant
|
||||
// operators running `gitnexus analyze --verbose` in production
|
||||
// never saw the log (M3 from PR #1693 review).
|
||||
const chunkStartMs: number | null = verboseThroughputLog ? Date.now() : null;
|
||||
|
||||
const chunkContentPromise = chunkContentPromises[chunkIdx];
|
||||
if (!chunkContentPromise) {
|
||||
throw new Error(`Missing prefetched parse chunk ${chunkIdx + 1}/${numChunks}`);
|
||||
}
|
||||
const chunkContents = await chunkContentPromise;
|
||||
chunkContentPromises[chunkIdx] = undefined; // release the in-memory copy
|
||||
startChunkPrefetch(chunkIdx + parseChunkConcurrency);
|
||||
const chunkFiles: Array<{ path: string; content: string }> = [];
|
||||
for (const p of chunkPaths) {
|
||||
const content = chunkContents.get(p);
|
||||
if (content !== undefined) chunkFiles.push({ path: p, content });
|
||||
}
|
||||
|
||||
// Compute the chunk's content-hash signature (if cache available).
|
||||
let chunkHash: string | null = null;
|
||||
if (parseCache) {
|
||||
const entries = chunkFiles.map((f) => ({
|
||||
filePath: f.path,
|
||||
contentHash: fileContentHash(f.content),
|
||||
}));
|
||||
chunkHash = computeChunkHash(entries);
|
||||
}
|
||||
|
||||
const cachedRaw =
|
||||
chunkHash && parseCache ? await loadParseCacheChunk(parseCache, chunkHash) : undefined;
|
||||
|
||||
// Track every chunk hash we touched so the orchestrator can
|
||||
// prune stale entries (chunks whose composition no longer
|
||||
// corresponds to a live chunk in the current scan) before saving.
|
||||
if (parseCache && chunkHash) parseCache.usedKeys.add(chunkHash);
|
||||
|
||||
if (cachedRaw && cachedRaw.length > 0) {
|
||||
// Cache hit: replay cached worker output. Finalize any parked worker
|
||||
// chunk FIRST so deferred aggregation stays in chunk order, then merge
|
||||
// + apply this hit inline (no worker dispatch to overlap).
|
||||
if (pendingWorkerChunk) {
|
||||
await finalizeWorkerChunk(pendingWorkerChunk);
|
||||
pendingWorkerChunk = null;
|
||||
}
|
||||
chunkCacheHits++;
|
||||
const chunkWorkerData = mergeChunkResults(graph, symbolTable, cachedRaw, exportedTypeMap);
|
||||
if (isDev) {
|
||||
logger.info(
|
||||
`📦 parse-cache HIT: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash?.slice(0, 8) ?? 'unknown'})`,
|
||||
);
|
||||
}
|
||||
// Progress update so UI advances even on a cache hit.
|
||||
const cachedFiles = chunkFiles.length;
|
||||
onProgress({
|
||||
phase: 'parsing',
|
||||
// Parse phase covers 20-70 (50 points). Deferred extraction below
|
||||
// takes 70-95 so the UI advances through the (potentially long)
|
||||
// resolution stages instead of holding at 82 (M2 from PR #1693
|
||||
// review).
|
||||
percent: Math.round(20 + ((filesParsedSoFar + cachedFiles) / totalParseable) * 50),
|
||||
message: `Parsing chunk ${chunkIdx + 1}/${numChunks} (cache)...`,
|
||||
stats: {
|
||||
filesProcessed: filesParsedSoFar + cachedFiles,
|
||||
totalFiles: totalParseable,
|
||||
nodesCreated: graph.nodeCount,
|
||||
},
|
||||
});
|
||||
await applyChunkResults(chunkWorkerData, chunkIdx, chunkFiles, chunkStartMs);
|
||||
} else {
|
||||
// Cache miss: dispatch to workers, capture the raw results, store
|
||||
// them under the chunk hash for the next run.
|
||||
chunkCacheMisses++;
|
||||
const progressForChunk = (current: number, _total: number, filePath: string) => {
|
||||
const globalCurrent = filesParsedSoFar + current;
|
||||
// Parse phase covers 20-70 (M2). Deferred extraction handles 70-95.
|
||||
const parsingProgress = 20 + (globalCurrent / totalParseable) * 50;
|
||||
onProgress({
|
||||
phase: 'parsing',
|
||||
percent: Math.round(parsingProgress),
|
||||
message: `Parsing chunk ${chunkIdx + 1}/${numChunks}...`,
|
||||
detail: filePath,
|
||||
stats: {
|
||||
filesProcessed: globalCurrent,
|
||||
totalFiles: totalParseable,
|
||||
nodesCreated: graph.nodeCount,
|
||||
},
|
||||
});
|
||||
};
|
||||
const activeWorkerPool = getOrCreateWorkerPool();
|
||||
// Worker path — PIPELINE: kick off this chunk's dispatch, merge the
|
||||
// PREVIOUS chunk while these workers parse, then park this chunk for
|
||||
// the next iteration to merge (overlapping its parse). The deferred
|
||||
// merge + parse-cache write-guard + aggregation all run in
|
||||
// `finalizeWorkerChunk`, in chunk order. The pool is the sole parse
|
||||
// path — `getOrCreateWorkerPool` returns a pool or throws.
|
||||
const dispatchPromise = dispatchChunkParse(chunkFiles, activeWorkerPool, progressForChunk);
|
||||
// Mark handled so a rejection during the overlap drain below isn't
|
||||
// flagged as unhandled; the `await` re-throws it for real handling.
|
||||
dispatchPromise.catch(() => {});
|
||||
if (pendingWorkerChunk) {
|
||||
await finalizeWorkerChunk(pendingWorkerChunk);
|
||||
pendingWorkerChunk = null;
|
||||
}
|
||||
let chunkResults: ParseWorkerResult[];
|
||||
try {
|
||||
chunkResults = await dispatchPromise;
|
||||
} catch (err) {
|
||||
if (!(err instanceof WorkerPoolInitializationError)) throw err;
|
||||
// Every worker crashed during startup and the pool's bounded
|
||||
// self-heal was exhausted. Fail fast (#1741) — there is no sequential
|
||||
// parser to degrade to. `handleWorkerStartupFailure` always throws, so
|
||||
// `chunkResults` stays definitely assigned for the parked chunk below.
|
||||
handleWorkerStartupFailure(err);
|
||||
}
|
||||
pendingWorkerChunk = {
|
||||
rawResults: chunkResults,
|
||||
chunkIdx,
|
||||
chunkHash,
|
||||
chunkFiles,
|
||||
chunkStartMs,
|
||||
};
|
||||
}
|
||||
|
||||
// (Per-chunk aggregation + parse-cache write + throughput log now run in
|
||||
// `applyChunkResults` / `finalizeWorkerChunk` — see the merge-pipelining
|
||||
// block above. Route/import/inheritance edges are emitted later: route
|
||||
// resolution in the single end-of-loop pass below, the rest by the
|
||||
// scope-resolution phase, RING4-2 #943.)
|
||||
}
|
||||
|
||||
// Drain the final parked worker chunk — the last pipelined chunk has no
|
||||
// successor to overlap its merge with, so merge + finalize it here.
|
||||
if (pendingWorkerChunk) {
|
||||
await finalizeWorkerChunk(pendingWorkerChunk);
|
||||
pendingWorkerChunk = null;
|
||||
}
|
||||
|
||||
if (isDev && parseCache && (chunkCacheHits > 0 || chunkCacheMisses > 0)) {
|
||||
|
|
@ -791,68 +880,28 @@ export async function runChunkedParseAndResolve(
|
|||
await workerPool?.terminate();
|
||||
}
|
||||
|
||||
// Sequential fallback chunks.
|
||||
//
|
||||
// U6: wrap the fallback loop and the finalize/enrich steps in a try/finally
|
||||
// so cleanup still runs on a mid-fallback throw. The `finally` guarantees:
|
||||
// 1. `astCache.clear()` releases any tree-sitter trees held by the most
|
||||
// recently allocated per-chunk cache, mirroring the per-chunk
|
||||
// `astCache.clear()` calls on the happy path.
|
||||
// 2. `bindingAccumulator.finalize()` runs before `crossFile` disposes the
|
||||
// accumulator downstream — callers that inspect partial TypeEnv state
|
||||
// (or consume it via `enrichExportedTypeMap` on a partial recovery)
|
||||
// still see a finalized accumulator.
|
||||
// 3. `enrichExportedTypeMap` runs so any bindings already accumulated
|
||||
// are propagated into `exportedTypeMap` even if the fallback aborted.
|
||||
//
|
||||
// Disposal of the accumulator remains with `crossFile` (owned by U2). We do
|
||||
// NOT call `bindingAccumulator.dispose()` here.
|
||||
// Fetch calls + ORM queries were already extracted inside each worker
|
||||
// (returned in ParseWorkerResult, aggregated per chunk in applyChunkResults).
|
||||
// With sequential parsing removed there is no post-loop drain to run — only
|
||||
// the TypeEnv finalize + enrichment that the drain's `finally` used to host.
|
||||
// Finalize the accumulator and propagate any fixpoint-inferred exports before
|
||||
// `crossFile` disposes it downstream. Wrapped in try/catch so a cleanup
|
||||
// failure never masks a real parse error; disposal stays with `crossFile`.
|
||||
astCache.clear();
|
||||
try {
|
||||
// Sequential fallback: calls, inheritance, and imports are emitted by the
|
||||
// scope-resolution phase, not here (RING4-1 #942 removed the legacy
|
||||
// call/heritage passes; RING4-2 #943 removed the legacy import resolution).
|
||||
// This loop still extracts fetch routes + ORM queries, which are
|
||||
// language-agnostic edge sources independent of call resolution.
|
||||
for (const chunkPaths of sequentialChunkPaths) {
|
||||
const chunkContents = await readFileContents(repoPath, chunkPaths);
|
||||
const chunkFiles: Array<{ path: string; content: string }> = [];
|
||||
for (const p of chunkPaths) {
|
||||
const content = chunkContents.get(p);
|
||||
if (content !== undefined) chunkFiles.push({ path: p, content });
|
||||
}
|
||||
astCache = createASTCache(chunkFiles.length);
|
||||
const chunkFetchCalls = await extractFetchCallsFromFiles(chunkFiles, astCache);
|
||||
if (chunkFetchCalls.length > 0) {
|
||||
for (const item of chunkFetchCalls) allFetchCalls.push(item);
|
||||
}
|
||||
for (const f of chunkFiles) {
|
||||
extractORMQueriesInline(f.path, f.content, allORMQueries);
|
||||
}
|
||||
astCache.clear();
|
||||
bindingAccumulator.finalize();
|
||||
const enriched = enrichExportedTypeMap(bindingAccumulator, graph, exportedTypeMap);
|
||||
if (isDev && enriched > 0) {
|
||||
logger.info(
|
||||
`🔗 Worker TypeEnv enrichment: ${enriched} fixpoint-inferred exports added to ExportedTypeMap`,
|
||||
);
|
||||
}
|
||||
} finally {
|
||||
// Clearing an already-empty cache is a no-op, so this is idempotent-safe
|
||||
// on the happy path where every per-chunk block already cleared astCache.
|
||||
astCache.clear();
|
||||
|
||||
// Run finalize + enrichment inside try/catch so a cleanup failure never
|
||||
// masks the original fallback error. finalize must precede crossFile's
|
||||
// dispose (U2) and enrichExportedTypeMap depends on finalized bindings.
|
||||
try {
|
||||
bindingAccumulator.finalize();
|
||||
const enriched = enrichExportedTypeMap(bindingAccumulator, graph, exportedTypeMap);
|
||||
if (isDev && enriched > 0) {
|
||||
logger.info(
|
||||
`🔗 Worker TypeEnv enrichment: ${enriched} fixpoint-inferred exports added to ExportedTypeMap`,
|
||||
);
|
||||
}
|
||||
} catch (enrichErr) {
|
||||
if (isDev) {
|
||||
logger.warn(
|
||||
{ err: (enrichErr as Error).message },
|
||||
'Post-fallback finalize/enrich failed during cleanup:',
|
||||
);
|
||||
}
|
||||
} catch (enrichErr) {
|
||||
if (isDev) {
|
||||
logger.warn(
|
||||
{ err: (enrichErr as Error).message },
|
||||
'Post-parse finalize/enrich failed during cleanup:',
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -1018,6 +1067,10 @@ export async function runChunkedParseAndResolve(
|
|||
}
|
||||
}
|
||||
|
||||
logHeapProbe(
|
||||
'parse-impl-return',
|
||||
`exportedTypeMap=${exportedTypeMap.size} parsedFiles=${allParsedFiles.length} nodes=${graph.nodeCount}`,
|
||||
);
|
||||
return {
|
||||
exportedTypeMap,
|
||||
allFetchCalls,
|
||||
|
|
@ -1028,22 +1081,14 @@ export async function runChunkedParseAndResolve(
|
|||
allORMQueries,
|
||||
bindingAccumulator,
|
||||
model,
|
||||
// Whether a worker pool was actually live for this run. False means the
|
||||
// sequential fallback handled every chunk (either due to `skipWorkers`,
|
||||
// the file-count/byte thresholds, or a pool-creation failure).
|
||||
// Whether a worker pool was actually constructed for this run. False means
|
||||
// no pool was needed: a warm all-cache-hit run replays cached worker output
|
||||
// without spawning workers, or there were no parseable files.
|
||||
usedWorkerPool: workerPool !== undefined,
|
||||
// Surface the persistent scope cache so downstream phases
|
||||
// (scope-resolution) can skip re-parsing files that the
|
||||
// sequential path already parsed. Survives chunk boundaries; the
|
||||
// chunk-local `astCache` above is intentionally NOT exposed
|
||||
// because parse-impl clears it between chunks.
|
||||
scopeTreeCache,
|
||||
// Per-file ParsedFile artifacts produced by workers' calls to
|
||||
// `extractParsedFile`. Empty when only the sequential path ran
|
||||
// (sequential doesn't go through the worker, and extracts ParsedFile
|
||||
// inline rather than emitting it). Consumed by scope-resolution as
|
||||
// a re-extraction cache: when the file's ParsedFile is here,
|
||||
// scope-resolution skips its own `extractParsedFile` call.
|
||||
// `extractParsedFile`. Consumed by scope-resolution as a re-extraction
|
||||
// cache: when the file's ParsedFile is here, scope-resolution skips its own
|
||||
// `extractParsedFile` call.
|
||||
parsedFiles: allParsedFiles,
|
||||
};
|
||||
}
|
||||
|
|
|
|||
|
|
@ -2,8 +2,8 @@
|
|||
* Phase: parse
|
||||
*
|
||||
* Chunked parse + resolve loop: reads source in byte-budget chunks,
|
||||
* parses via worker pool (or sequential fallback), resolves imports,
|
||||
* heritage, and calls, synthesizes wildcard bindings.
|
||||
* parses via the worker pool (the sole parse path — no sequential fallback),
|
||||
* resolves imports, heritage, and calls, synthesizes wildcard bindings.
|
||||
*
|
||||
* This phase encapsulates the entire `runChunkedParseAndResolve` function
|
||||
* from the original pipeline. The chunk loop is a memory optimization
|
||||
|
|
@ -31,14 +31,14 @@ import type {
|
|||
} from '../workers/parse-worker.js';
|
||||
import { runChunkedParseAndResolve } from './parse-impl.js';
|
||||
import type { MutableSemanticModel } from '../model/index.js';
|
||||
import type { ASTCache } from '../ast-cache.js';
|
||||
|
||||
export interface ParseOutput {
|
||||
/**
|
||||
* Read-only snapshot of exported type bindings keyed by file path.
|
||||
*
|
||||
* Fully populated by `parse` (sequential path via `enrichExportedTypeMap`
|
||||
* and worker path via `buildExportedTypeMapFromGraph` in the main thread).
|
||||
* Fully populated by `parse` on the main thread after the worker parse:
|
||||
* `enrichExportedTypeMap` propagates fixpoint-inferred TypeEnv bindings and
|
||||
* `buildExportedTypeMapFromGraph` reconstructs from graph nodes.
|
||||
* Downstream phases — including `crossFile` — receive it as a true
|
||||
* `ReadonlyMap`; `crossFile` builds its own mutable working copy locally
|
||||
* for per-file re-resolution writes, so this snapshot is never mutated
|
||||
|
|
@ -62,29 +62,12 @@ export interface ParseOutput {
|
|||
/** Pass-through: total file count for progress reporting. */
|
||||
totalFiles: number;
|
||||
/**
|
||||
* True if the parse phase spawned a live worker pool for this run.
|
||||
* False means every chunk ran through the sequential fallback (skipWorkers,
|
||||
* thresholds not met, or pool-creation failure). Primarily a test affordance:
|
||||
* see `PipelineOptions.workerThresholdsForTest`.
|
||||
* True if the parse phase constructed a worker pool for this run. False
|
||||
* means no pool was needed — a warm all-cache-hit run replays cached worker
|
||||
* output without spawning workers, or there were no parseable files. There
|
||||
* is no sequential parser; the pool is the sole parse path on a cache miss.
|
||||
*/
|
||||
readonly usedWorkerPool: boolean;
|
||||
/**
|
||||
* Cross-phase tree-sitter Tree cache populated by the sequential
|
||||
* parse path. Separate from the chunk-local `astCache` used *inside*
|
||||
* the parse phase (which is cleared between chunks) — this one
|
||||
* survives the whole phase and hands Trees to scope-resolution so
|
||||
* it can skip a second parse.
|
||||
*
|
||||
* Empty entries for files that ran through the worker pool
|
||||
* (workers can't return native tree-sitter Trees across the
|
||||
* MessageChannel). Cache miss is safe — consumers fall back to a
|
||||
* fresh parse. See plan
|
||||
* docs/plans/2026-04-20-002-perf-parse-heritage-mro-plan.md (Unit 4).
|
||||
*
|
||||
* Disposed by `scopeResolutionPhase` (the sole consumer) via
|
||||
* `scopeTreeCache.clear()` after its extract loop finishes.
|
||||
*/
|
||||
readonly scopeTreeCache: ASTCache;
|
||||
/**
|
||||
* Per-file `ParsedFile` artifacts produced by workers' calls to
|
||||
* `extractParsedFile`. Threaded through to `scopeResolutionPhase`
|
||||
|
|
@ -92,10 +75,6 @@ export interface ParseOutput {
|
|||
* scope-resolution can skip its own `extractParsedFile` (which would
|
||||
* otherwise re-parse the file with tree-sitter on the main thread,
|
||||
* costing ~58s on a 1000-file repo).
|
||||
*
|
||||
* Empty for files that went through the sequential parse fallback —
|
||||
* sequential doesn't emit ParsedFile artifacts; scope-resolution
|
||||
* falls back to a fresh extract for those.
|
||||
*/
|
||||
readonly parsedFiles: readonly ParsedFile[];
|
||||
}
|
||||
|
|
|
|||
|
|
@ -43,27 +43,22 @@ import {
|
|||
export interface PipelineOptions {
|
||||
/** Skip MRO, community detection, and process extraction for faster test runs. */
|
||||
skipGraphPhases?: boolean;
|
||||
/** Force sequential parsing (no worker pool). Useful for testing the sequential path. */
|
||||
skipWorkers?: boolean;
|
||||
/**
|
||||
* @internal Test-only override for worker-pool gating thresholds.
|
||||
* When unset, production defaults apply (15 files OR 512 KB total bytes).
|
||||
* Setting either field lowers the corresponding threshold so small test
|
||||
* fixtures can still exercise the worker-pool path. Do not use from
|
||||
* production call sites.
|
||||
* Request parsing with the worker pool disabled. The sequential parser was
|
||||
* removed — the worker pool is the sole parse path — so setting this now
|
||||
* makes the parse phase throw a `WorkerPoolDisabledError` (equivalent to
|
||||
* `--workers 0`). Retained so callers get an actionable error rather than a
|
||||
* silently-different result.
|
||||
*/
|
||||
workerThresholdsForTest?: {
|
||||
minFiles?: number;
|
||||
minBytes?: number;
|
||||
};
|
||||
skipWorkers?: boolean;
|
||||
/**
|
||||
* @internal Test-only override for the worker script URL the pool
|
||||
* spawns. When unset, parse-impl resolves `parse-worker.js` from the
|
||||
* adjacent `workers/` directory (or the compiled `dist/` fallback
|
||||
* under vitest). Integration tests use this to inject a custom
|
||||
* worker script that deterministically triggers worker-pool
|
||||
* resilience paths (e.g., crash-on-poison-file) — same precedent as
|
||||
* `workerThresholdsForTest`. Do not use from production call sites.
|
||||
* resilience paths (e.g., crash-on-poison-file). Do not use from production
|
||||
* call sites.
|
||||
*/
|
||||
workerUrlForTest?: URL;
|
||||
/**
|
||||
|
|
@ -85,11 +80,11 @@ export interface PipelineOptions {
|
|||
* `createWorkerPool` so the pool sizing bypasses the env-var fallback
|
||||
* in `resolveAutoPoolSize`. The env-var channel
|
||||
* (`GITNEXUS_WORKER_POOL_SIZE`) remains as a back-compat fallback when
|
||||
* this field is undefined. Setting `workerPoolSize: 0` disables the
|
||||
* pool entirely (sequential fallback) — equivalent to `skipWorkers`
|
||||
* but expressed in the same units as `--workers <N>` so long-running
|
||||
* hosts (eval-server, MCP daemon) can size per-call without leaking
|
||||
* `process.env` state across analyze invocations.
|
||||
* this field is undefined. Must be a positive integer — `0` hard-errors
|
||||
* (sequential parsing was removed; equivalent to `skipWorkers`), expressed
|
||||
* in the same units as `--workers <N>` so long-running hosts (eval-server,
|
||||
* MCP daemon) can size per-call without leaking `process.env` state across
|
||||
* analyze invocations.
|
||||
*/
|
||||
workerPoolSize?: number;
|
||||
/**
|
||||
|
|
|
|||
|
|
@ -514,6 +514,45 @@ export interface ScopeResolver {
|
|||
resolutionConfig?: unknown,
|
||||
) => void;
|
||||
|
||||
/**
|
||||
* Restore capture-time per-file side-channel state that `emitScopeCaptures`
|
||||
* produces as a side effect into module-level maps, NOT onto the returned
|
||||
* `ParsedFile`. Such state never crosses the worker boundary: in worker-mode
|
||||
* parses `emitScopeCaptures` runs inside the worker, so its module-level
|
||||
* marks are populated in the WORKER process. The main thread then reuses the
|
||||
* serialized `ParsedFile` (see `RunScopeResolutionInput.preExtractedParsedFiles`)
|
||||
* and skips `extractParsedFile`, so those marks would otherwise be missing in
|
||||
* the main process where resolution consumes them.
|
||||
*
|
||||
* This hook reads the worker-serialized snapshot from
|
||||
* `parsed.captureSideChannel` (produced by the matching
|
||||
* `LanguageProvider.collectCaptureSideChannel` hook in the worker) and writes
|
||||
* it back into the module maps. It does NO tree-sitter parse and needs no
|
||||
* source `content` — that is the whole point of #1983 (the prior re-parse
|
||||
* replay re-introduced the main-thread tree-sitter OOM on huge `.h`/`.cpp`
|
||||
* repos and was replaced by this data-only restore).
|
||||
*
|
||||
* C++ is the only language with this pattern today: `emitCppScopeCaptures`
|
||||
* records ADL call-site arg shapes, inline-/anonymous-namespace ranges,
|
||||
* dependent-base names, and file-local linkage into module maps that
|
||||
* `populateOwners` and the ADL / two-phase-lookup passes read on the main
|
||||
* thread. Without this restore, all of that is empty on the worker path and
|
||||
* advanced C++ resolution (ADL / SFINAE-adjacent / inline-namespace) silently
|
||||
* produces zero edges.
|
||||
*
|
||||
* Called by `runScopeResolution` ONLY for pre-extracted files (the worker
|
||||
* already populated the marks in-process for freshly extracted files, so the
|
||||
* fresh-extract leg never calls this). Runs BEFORE `populateOwners(parsed)`
|
||||
* so the resolved-range Sets it repopulates are visible to that hook.
|
||||
*
|
||||
* Languages whose `emitScopeCaptures` is pure (the contract default — see
|
||||
* `scope-extractor.ts`) leave this undefined; the restore is a no-op for them.
|
||||
*
|
||||
* @param parsed The pre-extracted ParsedFile being reused. Its
|
||||
* `captureSideChannel` carries the worker-computed data.
|
||||
*/
|
||||
readonly applyCaptureSideChannel?: (parsed: ParsedFile) => void;
|
||||
|
||||
/**
|
||||
* Mutate `parsed.localDefs[i].ownerId` to point at the structural
|
||||
* owner. Python's rule: methods (Function defs whose parent scope
|
||||
|
|
|
|||
|
|
@ -31,8 +31,14 @@ import type { ParseOutput } from '../../pipeline-phases/parse.js';
|
|||
import { SupportedLanguages, getLanguageFromFilename } from 'gitnexus-shared';
|
||||
import { readFileContents } from '../../filesystem-walker.js';
|
||||
import { runScopeResolution, type ScopeResolutionSubPhase } from './run.js';
|
||||
import { buildGraphNodeLookup } from '../graph-bridge/node-lookup.js';
|
||||
import { SCOPE_RESOLVERS } from './registry.js';
|
||||
import { isDev, isSemanticModelValidatorEnabled } from '../../utils/env.js';
|
||||
import { logHeapProbe } from '../../utils/heap-probe.js';
|
||||
import {
|
||||
clearParsedFileStore,
|
||||
loadParsedFilesForPaths,
|
||||
} from '../../../../storage/parsedfile-store.js';
|
||||
import type { ResolutionOutcome } from '../resolution-outcome.js';
|
||||
|
||||
import { logger } from '../../../logger.js';
|
||||
|
|
@ -85,13 +91,10 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
ctx: PipelineContext,
|
||||
deps: ReadonlyMap<string, PhaseResult<unknown>>,
|
||||
): Promise<ScopeResolutionOutput> {
|
||||
logHeapProbe('scopeResolution-enter');
|
||||
const { scannedFiles } = getPhaseOutput<StructureOutput>(deps, 'structure');
|
||||
// Reach into the parse phase's AST cache so per-file extract can
|
||||
// skip a second tree-sitter parse. Cache miss is safe (re-parses).
|
||||
// Worker-mode parses leave the cache empty for those files; they
|
||||
// also fall back to a fresh parse — no correctness impact.
|
||||
const parseOutput = getPhaseOutput<ParseOutput>(deps, 'parse');
|
||||
const { scopeTreeCache, model, parsedFiles: workerParsedFiles } = parseOutput;
|
||||
const { model, parsedFiles: workerParsedFiles } = parseOutput;
|
||||
// SemanticModel populated during `parse`: scope-resolution consumes
|
||||
// TypeRegistry / MethodRegistry / SymbolTable lookups instead of
|
||||
// rebuilding parallel indexes. See ARCHITECTURE.md § "Semantic-model
|
||||
|
|
@ -109,6 +112,16 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
preExtractedByPath.set(pf.filePath, pf);
|
||||
}
|
||||
|
||||
// Disk-backed ParsedFile store (#1983): on huge repos the parse phase
|
||||
// flushes worker-produced ParsedFiles to disk (instead of `workerParsedFiles`
|
||||
// heap) and we stream them back here per language. Reusing them means
|
||||
// scope-resolution does ZERO tree-sitter parsing on the main thread, which
|
||||
// is what eliminates the unbounded native re-parse leak that OOM'd large
|
||||
// analyses. `undefined` when no storage path (tests / direct calls) — those
|
||||
// runs use the in-heap `workerParsedFiles` above or fall back to a fresh
|
||||
// extract per file inside runScopeResolution.
|
||||
const parsedFileStorePath = ctx.options?.parseCache?.storagePath;
|
||||
|
||||
// Drop pre-extracted entries for standalone providers — these
|
||||
// languages are skipped by the canonical guard below (line 164)
|
||||
// and never consume preExtractedByPath, so holding onto their
|
||||
|
|
@ -164,6 +177,13 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
});
|
||||
}
|
||||
|
||||
// Build the graph-node lookup ONCE and share it across every language pass.
|
||||
// It scans the whole graph (~2 GB on the kernel) and is language-agnostic,
|
||||
// so the previous per-language rebuild burned that CPU+heap N times and, on
|
||||
// a huge repo, a tiny language's full-graph copy overlapped the next
|
||||
// language's — a real contributor to the scope-resolution memory peak.
|
||||
const sharedNodeLookup = totalScopeFiles > 0 ? buildGraphNodeLookup(ctx.graph) : undefined;
|
||||
|
||||
for (const [lang, provider] of SCOPE_RESOLVERS) {
|
||||
// Standalone providers (COBOL, JCL) don't emit graph edges yet
|
||||
// through the scope-resolution path. This is the canonical guard:
|
||||
|
|
@ -194,9 +214,37 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
// To avoid reading primary files twice (once for the hook, once for
|
||||
// the resolution pass), we read them upfront and merge with the
|
||||
// extra context paths the hook may add.
|
||||
// Stream this language's pre-built ParsedFiles in from the disk store
|
||||
// FIRST (huge-repo path). Doing it before reading source lets us skip
|
||||
// loading content for files the store already covers — for a provider
|
||||
// with no content-consuming hook that source is pure dead weight once
|
||||
// extraction is served from the store (~1.5 GB on the kernel's C pass).
|
||||
// Merged into `preExtractedByPath`; the per-language release block below
|
||||
// evicts these again before the next language, so only one language's
|
||||
// ParsedFiles are resident at a time.
|
||||
const loadStoreFor = async (paths: ReadonlySet<string>): Promise<void> => {
|
||||
if (!parsedFileStorePath) return;
|
||||
const fromDisk = await loadParsedFilesForPaths(parsedFileStorePath, paths);
|
||||
for (const [fp, pf] of fromDisk) preExtractedByPath.set(fp, pf);
|
||||
};
|
||||
|
||||
// A provider that feeds source text into a post-extract hook
|
||||
// (populateWorkspaceOwners / populateNamespaceSiblings /
|
||||
// populateRangeBindings / emitPostResolutionEdges) needs content for ALL
|
||||
// its files; one without those hooks only needs content for files the
|
||||
// store does NOT cover (fresh-extract fallback). Keep this in sync with
|
||||
// the getFileContents() call-sites in run.ts.
|
||||
const providerNeedsAllContent =
|
||||
provider.populateWorkspaceOwners !== undefined ||
|
||||
provider.populateNamespaceSiblings !== undefined ||
|
||||
provider.populateRangeBindings !== undefined ||
|
||||
provider.emitPostResolutionEdges !== undefined;
|
||||
|
||||
let scopeFilePaths: Set<string>;
|
||||
let contents: Map<string, string>;
|
||||
if (provider.collectScopeContextPaths !== undefined) {
|
||||
// Context-expanding providers (e.g. Vue) need every primary file's
|
||||
// source up front for the closure hook, so load it all.
|
||||
const entryFileContents = await readFileContents(ctx.repoPath, primaryFilePaths);
|
||||
scopeFilePaths = provider.collectScopeContextPaths({
|
||||
primaryFilePaths,
|
||||
|
|
@ -209,18 +257,36 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
const extraPaths = [...scopeFilePaths].filter((p) => !entryFileContents.has(p));
|
||||
const extraContents = await readFileContents(ctx.repoPath, extraPaths);
|
||||
contents = new Map([...entryFileContents, ...extraContents]);
|
||||
await loadStoreFor(scopeFilePaths);
|
||||
} else {
|
||||
scopeFilePaths = new Set(primaryFilePaths);
|
||||
contents = await readFileContents(ctx.repoPath, primaryFilePaths);
|
||||
await loadStoreFor(scopeFilePaths);
|
||||
const pathsToRead = providerNeedsAllContent
|
||||
? primaryFilePaths
|
||||
: primaryFilePaths.filter((p) => !preExtractedByPath.has(p));
|
||||
contents = await readFileContents(ctx.repoPath, pathsToRead);
|
||||
}
|
||||
const filePaths = [...scopeFilePaths];
|
||||
const files: { path: string; content: string }[] = [];
|
||||
for (const fp of filePaths) {
|
||||
const content = contents.get(fp);
|
||||
if (content !== undefined) files.push({ path: fp, content });
|
||||
if (content !== undefined) {
|
||||
files.push({ path: fp, content });
|
||||
} else if (preExtractedByPath.has(fp)) {
|
||||
// Store covers extraction for this file and we deliberately skipped
|
||||
// reading its source; the empty string is never consumed (the
|
||||
// extract loop uses the pre-extracted ParsedFile and this provider
|
||||
// has no content hook).
|
||||
files.push({ path: fp, content: '' });
|
||||
}
|
||||
// else: uncovered AND unreadable → skip (unchanged from prior behavior).
|
||||
}
|
||||
|
||||
const langFileCount = files.length;
|
||||
logHeapProbe(
|
||||
'scope-lang-start',
|
||||
`lang=${lang} files=${langFileCount} contentsLoaded=${contents.size}`,
|
||||
);
|
||||
const langLabel = lang.charAt(0).toUpperCase() + lang.slice(1);
|
||||
currentLangIdx++;
|
||||
const langTag =
|
||||
|
|
@ -242,8 +308,8 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
graph: ctx.graph,
|
||||
model,
|
||||
files,
|
||||
treeCache: scopeTreeCache,
|
||||
resolutionConfig,
|
||||
prebuiltNodeLookup: sharedNodeLookup,
|
||||
preExtractedParsedFiles: preExtractedByPath,
|
||||
recordResolutionOutcome: (outcome) => {
|
||||
resolutionOutcomes.push(outcome);
|
||||
|
|
@ -308,6 +374,7 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
for (const fp of filePaths) {
|
||||
preExtractedByPath.delete(fp);
|
||||
}
|
||||
logHeapProbe('scope-lang-end', `lang=${lang} filesProcessed=${stats.filesProcessed}`);
|
||||
|
||||
processedScopeFiles += langFileCount;
|
||||
anyRan = true;
|
||||
|
|
@ -336,13 +403,17 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
});
|
||||
}
|
||||
|
||||
// Dispose the cross-phase Tree cache — scope-resolution is the
|
||||
// only consumer. Holding Trees past this point is pure memory
|
||||
// pressure: downstream phases (mro, community, csv-generator)
|
||||
// never read them, and tree-sitter Trees hold native-heap memory
|
||||
// under WASM runtimes. ASTCache.clear() fires the LRU dispose
|
||||
// handler which calls tree.delete?.() on each retained Tree.
|
||||
scopeTreeCache.clear();
|
||||
// Scope-resolution is the sole consumer of the disk-backed ParsedFile
|
||||
// store; remove its shards now (can be many GB on a huge repo) so they
|
||||
// don't linger in `.gitnexus`. Best-effort — never fail the phase on a
|
||||
// cleanup error.
|
||||
if (parsedFileStorePath) {
|
||||
try {
|
||||
await clearParsedFileStore(parsedFileStorePath);
|
||||
} catch {
|
||||
/* best-effort cleanup */
|
||||
}
|
||||
}
|
||||
|
||||
if (!anyRan) return NOOP_OUTPUT;
|
||||
|
||||
|
|
|
|||
|
|
@ -45,6 +45,7 @@ import type { ScopeResolver } from '../contract/scope-resolver.js';
|
|||
import { findEnclosingClassDef, resolveInheritanceBaseInScope } from '../scope/walkers.js';
|
||||
import { buildWorkspaceResolutionIndex } from '../workspace-index.js';
|
||||
import type { ResolutionOutcome, ResolutionOutcomeRecorder } from '../resolution-outcome.js';
|
||||
import { logHeapProbe } from '../../utils/heap-probe.js';
|
||||
|
||||
import { logger } from '../../../logger.js';
|
||||
|
||||
|
|
@ -240,13 +241,26 @@ interface RunScopeResolutionInput {
|
|||
readonly files: readonly { readonly path: string; readonly content: string }[];
|
||||
readonly onWarn?: (message: string) => void;
|
||||
/**
|
||||
* Optional pre-parsed-Tree lookup keyed by file path. When the
|
||||
* pipeline's parse phase ran sequentially, it populated an
|
||||
* `ASTCache`; passing that here lets the per-file extract step
|
||||
* skip a second `tree-sitter parser.parse(...)` call. Cache miss
|
||||
* is safe — falls back to a fresh parse inside the provider.
|
||||
* Optional pre-parsed-Tree lookup keyed by file path: a cache hit lets the
|
||||
* per-file extract step skip a second `tree-sitter parser.parse(...)` call.
|
||||
* Currently always empty — the only producer was the (removed) sequential
|
||||
* parser, and workers can't return native Trees across the MessageChannel,
|
||||
* so the parse phase no longer threads one. Kept as an extension point;
|
||||
* cache miss is safe (the provider re-parses).
|
||||
*/
|
||||
readonly treeCache?: { get(filePath: string): unknown };
|
||||
/**
|
||||
* Optional graph-node lookup built ONCE by the caller and shared across
|
||||
* every language pass. `buildGraphNodeLookup` scans the whole graph and is
|
||||
* language-agnostic, so rebuilding it per language wastes both CPU and ~GBs
|
||||
* of heap (on the kernel it is ~2 GB; a 5-file language would otherwise build
|
||||
* its own full copy that then overlaps the next language's). When omitted
|
||||
* (tests / isolated calls) the lookup is built locally as before. Providers
|
||||
* that add graph nodes mid-pass (e.g. Ruby heritage Property nodes) still
|
||||
* rebuild a fresh post-heritage lookup internally, so sharing the pre-loop
|
||||
* base is safe.
|
||||
*/
|
||||
readonly prebuiltNodeLookup?: ReturnType<typeof buildGraphNodeLookup>;
|
||||
/**
|
||||
* Opaque per-language import-resolution config (e.g. tsconfig path
|
||||
* aliases for TypeScript). Loaded once by the caller via
|
||||
|
|
@ -328,15 +342,20 @@ export function runScopeResolution(
|
|||
let preExtractedHits = 0;
|
||||
const progressInterval = files.length > 0 ? Math.max(1, Math.floor(files.length / 50)) : 1;
|
||||
input.onProgress?.('extracting', 0, files.length);
|
||||
logHeapProbe('sr-extract-start', `lang=${provider.language} files=${files.length}`);
|
||||
for (let fileIdx = 0; fileIdx < files.length; fileIdx++) {
|
||||
const file = files[fileIdx];
|
||||
let parsed: ParsedFile | undefined;
|
||||
// Fast path: a worker (during the parse phase) already produced a
|
||||
// ParsedFile for this file via `extractParsedFile`. Reuse it
|
||||
// directly — skips a tree-sitter re-parse on the main thread.
|
||||
let reusedPreExtracted = false;
|
||||
if (preExtracted !== undefined) {
|
||||
parsed = preExtracted.get(file.path);
|
||||
if (parsed !== undefined) preExtractedHits++;
|
||||
if (parsed !== undefined) {
|
||||
preExtractedHits++;
|
||||
reusedPreExtracted = true;
|
||||
}
|
||||
}
|
||||
if (parsed === undefined) {
|
||||
const cachedTree = treeCache?.get(file.path);
|
||||
|
|
@ -352,18 +371,37 @@ export function runScopeResolution(
|
|||
continue;
|
||||
}
|
||||
}
|
||||
// Worker-boundary restore: a pre-extracted ParsedFile was produced by
|
||||
// `extractParsedFile` running INSIDE a worker, so any capture-time
|
||||
// module-level side-channel state (`emitScopeCaptures` side effects that
|
||||
// are NOT serialized onto the ParsedFile's scopes/defs — C++ ADL/namespace
|
||||
// marks) was populated in the worker process and is missing here. The
|
||||
// worker stashed a plain-data snapshot on `parsed.captureSideChannel` (via
|
||||
// `collectCaptureSideChannel`); write it back into the module maps now,
|
||||
// BEFORE populateOwners consumes the resolved ranges. NO re-parse — that is
|
||||
// the #1983 fix. The fresh-extract leg above already populated those marks
|
||||
// in this process, so it skips the restore. See
|
||||
// `ScopeResolver.applyCaptureSideChannel`.
|
||||
if (reusedPreExtracted && provider.applyCaptureSideChannel !== undefined) {
|
||||
provider.applyCaptureSideChannel(parsed);
|
||||
}
|
||||
provider.populateOwners(parsed);
|
||||
parsedFiles.push(parsed);
|
||||
if (
|
||||
input.onProgress &&
|
||||
((fileIdx + 1) % progressInterval === 0 || fileIdx === files.length - 1)
|
||||
) {
|
||||
input.onProgress('extracting', fileIdx + 1, files.length);
|
||||
if ((fileIdx + 1) % progressInterval === 0 || fileIdx === files.length - 1) {
|
||||
input.onProgress?.('extracting', fileIdx + 1, files.length);
|
||||
logHeapProbe(
|
||||
'sr-extract-progress',
|
||||
`lang=${provider.language} idx=${fileIdx + 1}/${files.length} parsedFiles=${parsedFiles.length} preExtractedHits=${preExtractedHits}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
if (PROF && preExtracted !== undefined) {
|
||||
logger.warn(`[scope-resolution prof] pre-extracted hits: ${preExtractedHits}/${files.length}`);
|
||||
}
|
||||
logHeapProbe(
|
||||
'sr-extract-end',
|
||||
`lang=${provider.language} parsedFiles=${parsedFiles.length} preExtractedHits=${preExtractedHits} skipped=${filesSkipped}`,
|
||||
);
|
||||
provider.populateWorkspaceOwners?.(parsedFiles, { fileContents: getFileContents() });
|
||||
|
||||
// Reconcile scope-resolution's ownership view into the SemanticModel.
|
||||
|
|
@ -397,7 +435,9 @@ export function runScopeResolution(
|
|||
// ── Phase 2: finalize → ScopeResolutionIndexes ─────────────────────────
|
||||
input.onProgress?.('analyzing types', files.length, files.length);
|
||||
const allFilePaths = new Set(parsedFiles.map((f) => f.filePath));
|
||||
const nodeLookup = buildGraphNodeLookup(graph);
|
||||
logHeapProbe('sr-pre-nodeLookup', `lang=${provider.language}`);
|
||||
const nodeLookup = input.prebuiltNodeLookup ?? buildGraphNodeLookup(graph);
|
||||
logHeapProbe('sr-post-nodeLookup', `lang=${provider.language}`);
|
||||
|
||||
const resolutionConfig = input.resolutionConfig;
|
||||
const finalized = finalizeScopeModel(parsedFiles, {
|
||||
|
|
@ -410,6 +450,7 @@ export function runScopeResolution(
|
|||
provider.mergeBindings(existing, incoming, scopeId),
|
||||
},
|
||||
});
|
||||
logHeapProbe('sr-post-finalize', `lang=${provider.language}`);
|
||||
const preEmittedInheritanceSites = preEmitInheritanceEdges(graph, finalized, nodeLookup);
|
||||
// Call-based heritage hook (e.g., Ruby include/extend/prepend) — emits
|
||||
// IMPLEMENTS edges that `preEmitInheritanceEdges` cannot produce because
|
||||
|
|
@ -462,6 +503,7 @@ export function runScopeResolution(
|
|||
// attributed correctly) and AFTER finalize (so module-scope
|
||||
// bindings are available).
|
||||
const workspaceIndex = buildWorkspaceResolutionIndex(parsedFiles);
|
||||
logHeapProbe('sr-post-workspaceIndex', `lang=${provider.language}`);
|
||||
|
||||
// Cross-file implicit-namespace visibility (C#). Must run before
|
||||
// propagateImportedReturnTypes so the latter pass sees siblings'
|
||||
|
|
@ -522,6 +564,7 @@ export function runScopeResolution(
|
|||
lookupOwnedMembersByOwner(readonlyModel, ownerDefId, memberName),
|
||||
});
|
||||
const tResolve = PROF ? process.hrtime.bigint() : 0n;
|
||||
logHeapProbe('sr-post-resolve', `lang=${provider.language}`);
|
||||
|
||||
// ── Phase 4: emit graph edges (LOAD-BEARING ORDER — see I1) ────────────
|
||||
input.onProgress?.('linking symbols', files.length, files.length);
|
||||
|
|
@ -608,6 +651,8 @@ export function runScopeResolution(
|
|||
);
|
||||
}
|
||||
|
||||
logHeapProbe('sr-end', `lang=${provider.language} parsedFiles=${parsedFiles.length}`);
|
||||
|
||||
return {
|
||||
filesProcessed: parsedFiles.length,
|
||||
filesSkipped,
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@
|
|||
* Enabled with `GITNEXUS_DEBUG_HEAP=1` or `GITNEXUS_PROFILE_DEFERRED=1`.
|
||||
*/
|
||||
|
||||
import { appendFileSync } from 'node:fs';
|
||||
import { parseTruthyEnv } from './env.js';
|
||||
import { isDeferredResolutionProfileEnabled } from './deferred-resolution-profile.js';
|
||||
|
||||
|
|
@ -13,9 +14,30 @@ export const isDebugHeapEnabled = (): boolean =>
|
|||
|
||||
export const heapUsedMb = (): number => Math.round(process.memoryUsage().heapUsed / 1024 / 1024);
|
||||
|
||||
/** Flush a one-line heap snapshot to stderr. */
|
||||
const rssMb = (): number => Math.round(process.memoryUsage().rss / 1024 / 1024);
|
||||
|
||||
/**
|
||||
* Flush a one-line heap snapshot to stderr.
|
||||
*
|
||||
* stderr is async on a pipe (Linux), so a probe written just before an
|
||||
* OOM-kill is lost in the pipe buffer. For OOM investigation set
|
||||
* `GITNEXUS_HEAP_PROBE_FILE=/path` — each probe is then ALSO appended to
|
||||
* that file with a synchronous `appendFileSync`, whose `write(2)` syscall
|
||||
* completes (data handed to the kernel) before the call returns, so the
|
||||
* last line survives SIGKILL. Includes rss alongside heapUsed so the
|
||||
* native/off-heap gap is visible.
|
||||
*/
|
||||
export const logHeapProbe = (label: string, detail?: string): void => {
|
||||
if (!isDebugHeapEnabled()) return;
|
||||
const suffix = detail ? ` ${detail}` : '';
|
||||
process.stderr.write(`[gitnexus-heap] ${label} used_mb=${heapUsedMb()}${suffix}\n`);
|
||||
const line = `[gitnexus-heap] ${label} used_mb=${heapUsedMb()} rss_mb=${rssMb()}${suffix}\n`;
|
||||
process.stderr.write(line);
|
||||
const file = process.env.GITNEXUS_HEAP_PROBE_FILE;
|
||||
if (file) {
|
||||
try {
|
||||
appendFileSync(file, line);
|
||||
} catch {
|
||||
// Best-effort diagnostics sink; never let a probe failure abort analyze.
|
||||
}
|
||||
}
|
||||
};
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
import { parentPort, threadId } from 'node:worker_threads';
|
||||
import { parentPort, threadId, workerData } from 'node:worker_threads';
|
||||
import Parser from 'tree-sitter';
|
||||
import JavaScript from 'tree-sitter-javascript';
|
||||
import TypeScript from 'tree-sitter-typescript';
|
||||
|
|
@ -93,14 +93,34 @@ import {
|
|||
buildCollisionGroups,
|
||||
parameterShapeIdTag,
|
||||
} from '../utils/method-props.js';
|
||||
import { extractTemplateArguments, templateArgumentsIdTag } from '../utils/template-arguments.js';
|
||||
import {
|
||||
extractTemplateArguments,
|
||||
templateArgumentsIdTag,
|
||||
templateConstraintsIdTag,
|
||||
} from '../utils/template-arguments.js';
|
||||
import type { LanguageProvider } from '../language-provider.js';
|
||||
import type { ParsedFile } from 'gitnexus-shared';
|
||||
import { extractParsedFile, type ScopeCaptureSourceKind } from '../scope-extractor-bridge.js';
|
||||
import { persistParsedFileShardSync } from '../../../storage/parsedfile-store.js';
|
||||
import { extractLaravelRoutes, type ExtractedRoute } from '../route-extractors/laravel.js';
|
||||
|
||||
import { logger } from '../../logger.js';
|
||||
export type { ExtractedRoute } from '../route-extractors/laravel.js';
|
||||
|
||||
// ── ParsedFile store (#1983 parallel serialization) ─────────────────────────
|
||||
// Read ONCE at worker init from `workerData` (immutable for the run, inherited
|
||||
// by respawned workers via the pool's factory closure). When set, this worker
|
||||
// writes its own ParsedFile shards to disk at each job flush instead of
|
||||
// returning them over the MessageChannel — parallelizing serialization off the
|
||||
// main thread. `undefined` ⇒ return ParsedFiles in the result (no-store
|
||||
// fallback). `shardSeq` makes each shard name unique within this worker; global
|
||||
// uniqueness for the run rests on the process-monotonic `threadId` (never reused
|
||||
// across respawns) plus the per-run store clear on the main thread.
|
||||
const PARSED_FILE_STORE_STORAGE_PATH: string | undefined = (
|
||||
workerData as { parsedFileStoreStoragePath?: string } | undefined
|
||||
)?.parsedFileStoreStoragePath;
|
||||
let shardSeq = 0;
|
||||
|
||||
// ── Bootstrap-stage diagnostics (#1741) ────────────────────────────────────
|
||||
// When GITNEXUS_WORKER_BOOTSTRAP=1 (or --verbose sets GITNEXUS_VERBOSE), each
|
||||
// worker reports its startup stage timings to stderr — which the pool tees
|
||||
|
|
@ -1081,12 +1101,14 @@ const processFileGroup = (
|
|||
|
||||
// Vue SFC preprocessing: extract <script> block content
|
||||
let parseContent = file.content;
|
||||
let scopeSourceKind: ScopeCaptureSourceKind = 'full-file';
|
||||
let lineOffset = 0;
|
||||
let isVueSetup = false;
|
||||
if (language === SupportedLanguages.Vue) {
|
||||
const extracted = extractVueScript(file.content);
|
||||
if (!extracted) continue; // skip .vue files with no script block
|
||||
parseContent = extracted.scriptContent;
|
||||
scopeSourceKind = 'pre-extracted-script';
|
||||
lineOffset = extracted.lineOffset;
|
||||
isVueSetup = extracted.isSetup;
|
||||
}
|
||||
|
|
@ -1126,11 +1148,46 @@ const processFileGroup = (
|
|||
|
||||
const provider = getProvider(language);
|
||||
|
||||
// The worker no longer builds a `ParsedFile` here. Scope-resolution
|
||||
// re-extracts each file from source on the main thread (scope-resolution/
|
||||
// pipeline/run.ts); the worker copy retained ~2× the semantic model in RAM
|
||||
// on huge repos (#1983) and, with every language on the scope-resolution
|
||||
// path, was consumed by no one. `result.parsedFiles` stays empty.
|
||||
// Produce the `ParsedFile` for the scope-resolution pipeline HERE, reusing
|
||||
// the tree we just parsed (no second tree-sitter parse). Scope-resolution
|
||||
// consumes these via the disk-backed parsedfile-store instead of
|
||||
// re-extracting each file from source on the main thread — which
|
||||
// accumulated an unbounded native tree-sitter leak on huge repos (#1983;
|
||||
// see parsedfile-store.ts). parse-impl flushes `result.parsedFiles` to disk
|
||||
// per chunk and does NOT retain them in main-thread heap, so this no longer
|
||||
// costs ~1× the semantic model in RAM during parse.
|
||||
const parsedFile = extractParsedFile(
|
||||
provider,
|
||||
parseContent,
|
||||
file.path,
|
||||
(message) => {
|
||||
if (parentPort) {
|
||||
parentPort.postMessage({ type: 'warning', message });
|
||||
} else {
|
||||
logger.warn(message);
|
||||
}
|
||||
},
|
||||
tree,
|
||||
scopeSourceKind,
|
||||
);
|
||||
if (parsedFile !== undefined) {
|
||||
// Capture-time side-channel (#1983): `extractParsedFile` just ran the
|
||||
// provider's `emitScopeCaptures`, which (for C++) populated module-level
|
||||
// maps as a SIDE EFFECT that is NOT on `parsedFile`'s scopes/defs. Snapshot
|
||||
// that per-file state as plain data onto `ParsedFile.captureSideChannel`
|
||||
// so the main thread can restore it (via `ScopeResolver.applyCaptureSideChannel`)
|
||||
// WITHOUT a re-parse, after this ParsedFile crosses the worker boundary /
|
||||
// disk store. Providers without capture-time side effects leave the hook
|
||||
// undefined and this is a no-op. `undefined` return ⇒ no field added.
|
||||
//
|
||||
// `extractParsedFile` returns a frozen ParsedFile, so re-wrap (shallow
|
||||
// copy — scopes/defs are carried by reference) to attach the field rather
|
||||
// than mutate the frozen object.
|
||||
const sideChannel = provider.collectCaptureSideChannel?.(file.path);
|
||||
result.parsedFiles.push(
|
||||
sideChannel !== undefined ? { ...parsedFile, captureSideChannel: sideChannel } : parsedFile,
|
||||
);
|
||||
}
|
||||
|
||||
// Build per-file type environment + constructor bindings in a single AST walk.
|
||||
// The legacy heritage pre-pass that seeded a file-local parentMap for
|
||||
|
|
@ -1858,9 +1915,38 @@ const processFileGroup = (
|
|||
classTemplateArguments.length > 0
|
||||
? templateArgumentsIdTag(classTemplateArguments)
|
||||
: '';
|
||||
// SFINAE / `requires`-clause aware ID disambiguation (issue #1579).
|
||||
// Function-template overloads with identical parameterTypes but
|
||||
// mutually-exclusive constraints (e.g. `enable_if_t<is_integral_v<T>>`
|
||||
// vs `enable_if_t<is_floating_point_v<T>>`) need distinct graph nodes
|
||||
// so the constraint-filter step in `narrowOverloadCandidates` has two
|
||||
// candidates to narrow between. Without this tag they collapse to a
|
||||
// single Function node and the SFINAE call resolves to only one edge
|
||||
// regardless of which overload's constraint holds. This mirrors the
|
||||
// sequential `parsing-processor` path removed in #1983 — the worker is
|
||||
// now the sole parse path, so it must stamp the constraint tag and the
|
||||
// `templateConstraints` node property the resolver looks up by re-
|
||||
// hashing the def's constraints (see graph-bridge ids.ts / node-lookup.ts).
|
||||
let parsedTemplateConstraints: unknown = undefined;
|
||||
let constraintsTag = '';
|
||||
if (
|
||||
(nodeLabel === 'Function' || nodeLabel === 'Method') &&
|
||||
provider.extractTemplateConstraints !== undefined &&
|
||||
definitionNode
|
||||
) {
|
||||
try {
|
||||
parsedTemplateConstraints = provider.extractTemplateConstraints(definitionNode);
|
||||
if (parsedTemplateConstraints !== undefined) {
|
||||
constraintsTag = templateConstraintsIdTag(parsedTemplateConstraints);
|
||||
}
|
||||
} catch {
|
||||
parsedTemplateConstraints = undefined;
|
||||
constraintsTag = '';
|
||||
}
|
||||
}
|
||||
const nodeId = generateId(
|
||||
nodeLabel,
|
||||
`${file.path}:${qualifiedName}${classTemplateTag}${arityTag}${parameterShapeTag}`,
|
||||
`${file.path}:${qualifiedName}${classTemplateTag}${arityTag}${parameterShapeTag}${constraintsTag}`,
|
||||
);
|
||||
|
||||
const description = provider.descriptionExtractor?.(nodeLabel, nodeName, captureMap);
|
||||
|
|
@ -1981,6 +2067,9 @@ const processFileGroup = (
|
|||
...(classTemplateArguments !== undefined && classTemplateArguments.length > 0
|
||||
? { templateArguments: classTemplateArguments }
|
||||
: {}),
|
||||
...(parsedTemplateConstraints !== undefined
|
||||
? { templateConstraints: parsedTemplateConstraints }
|
||||
: {}),
|
||||
...(frameworkHint
|
||||
? {
|
||||
astFrameworkMultiplier: frameworkHint.entryPointMultiplier,
|
||||
|
|
@ -2250,6 +2339,25 @@ parentPort!.on('message', (msg: WorkerIncomingMessage) => {
|
|||
|
||||
// Flush: send accumulated results
|
||||
if (msg.type === 'flush') {
|
||||
// #1983 parallel serialization: when a store path is configured, write
|
||||
// this job's ParsedFiles to our own disk shard HERE (at the flush
|
||||
// boundary, where `accumulated.parsedFiles` is complete) and drop them
|
||||
// from the result so the main thread never deserializes/re-serializes
|
||||
// them. Writing at flush — not per sub-batch — encodes the invariant
|
||||
// "a shard is written iff its result is delivered": a worker that dies
|
||||
// before flush wrote no shard, so the pool's job retry yields exactly
|
||||
// one. `undefined` store path keeps ParsedFiles in the result (no-store
|
||||
// fallback). The write is synchronous: blocking this dedicated worker
|
||||
// thread protects the main thread and avoids threading async through the
|
||||
// accumulate path; per-job write time is small vs the parse it follows.
|
||||
if (PARSED_FILE_STORE_STORAGE_PATH && accumulated.parsedFiles.length > 0) {
|
||||
persistParsedFileShardSync(
|
||||
PARSED_FILE_STORE_STORAGE_PATH,
|
||||
`w${threadId}-${shardSeq++}`,
|
||||
accumulated.parsedFiles,
|
||||
);
|
||||
accumulated.parsedFiles = [];
|
||||
}
|
||||
parentPort!.postMessage({ type: 'result', data: accumulated });
|
||||
// Reset for potential reuse
|
||||
accumulated = {
|
||||
|
|
|
|||
|
|
@ -209,6 +209,17 @@ export interface WorkerPoolOptions {
|
|||
* code should leave this unset.
|
||||
*/
|
||||
workerFactory?: (workerUrl: URL) => Worker;
|
||||
/**
|
||||
* Storage path for the disk-backed ParsedFile store (#1983 parallel
|
||||
* serialization). When set, it is baked into every spawned worker's
|
||||
* `workerData` so the worker writes its own ParsedFile shards to disk
|
||||
* instead of returning them over the MessageChannel for the main thread to
|
||||
* serialize. Immutable for the run; captured in the default factory closure
|
||||
* so RESPAWNED workers inherit it automatically (all spawn sites reuse the
|
||||
* same factory). `undefined` ⇒ workers fall back to returning ParsedFiles in
|
||||
* the result (small-repo / no-storage path).
|
||||
*/
|
||||
parsedFileStoreStoragePath?: string;
|
||||
}
|
||||
|
||||
export class WorkerPoolDispatchError extends Error {
|
||||
|
|
@ -265,6 +276,23 @@ export class WorkerPoolInitializationError extends WorkerPoolDispatchError {
|
|||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Thrown when a caller asks GitNexus to parse without the worker pool —
|
||||
* `--workers 0`, `GITNEXUS_WORKER_POOL_SIZE=0`, or `skipWorkers: true`.
|
||||
*
|
||||
* GitNexus no longer has a sequential parser: the worker pool (with its
|
||||
* quarantine + respawn/recycle + circuit-breaker resilience) is the SOLE
|
||||
* parse path. These channels used to select an in-process fallback; they are
|
||||
* now hard configuration errors so the operator gets an actionable message
|
||||
* instead of silently parsing through a (deleted) slower path.
|
||||
*/
|
||||
export class WorkerPoolDisabledError extends Error {
|
||||
constructor(message: string) {
|
||||
super(message);
|
||||
this.name = 'WorkerPoolDisabledError';
|
||||
}
|
||||
}
|
||||
|
||||
/** Message shapes sent back by worker threads. */
|
||||
type WorkerOutgoingMessage =
|
||||
| { type: 'progress'; filesProcessed: number }
|
||||
|
|
@ -502,7 +530,7 @@ export function resolveWorkerPoolOptions(
|
|||
* The pool size requested via the `GITNEXUS_WORKER_POOL_SIZE` env var, or
|
||||
* `undefined` when unset, empty/whitespace, or invalid. Module-internal sizing
|
||||
* reader consumed by {@link resolveAutoPoolSize} (the env override) and
|
||||
* {@link workerPoolDisabledByEnv} (the sequential-routing gate). Reads only —
|
||||
* {@link workerPoolDisabledByEnv} (the disabled-channel check). Reads only —
|
||||
* never mutates `process.env`. Empty/whitespace is treated as *unset* (falls
|
||||
* through to the auto formula), not as 0 — an empty assignment (`export
|
||||
* GITNEXUS_WORKER_POOL_SIZE=`) is an accident, not a request for zero workers;
|
||||
|
|
@ -515,12 +543,11 @@ function envWorkerPoolSize(): number | undefined {
|
|||
}
|
||||
|
||||
/**
|
||||
* True when the operator explicitly disabled the worker pool via
|
||||
* `GITNEXUS_WORKER_POOL_SIZE=0` — the env-channel equivalent of `--workers 0`.
|
||||
* The parse phase's `shouldUseWorkers` gate consults this (only when no
|
||||
* explicit `--workers <N>` was passed) to route to sequential parsing instead
|
||||
* of constructing a useless size-0 pool that would fail fast on a phantom
|
||||
* crash (#1741). An explicit positive `--workers N` always wins.
|
||||
* True when the operator set `GITNEXUS_WORKER_POOL_SIZE=0` — the env-channel
|
||||
* equivalent of `--workers 0`. The parse phase consults this (only when no
|
||||
* explicit `--workers <N>` was passed) and HARD-ERRORS: sequential parsing was
|
||||
* removed, so a disabled pool is an actionable configuration error, not a
|
||||
* silent fallback. An explicit positive `--workers N` always wins.
|
||||
*/
|
||||
export function workerPoolDisabledByEnv(): boolean {
|
||||
return envWorkerPoolSize() === 0;
|
||||
|
|
@ -800,7 +827,18 @@ export const createWorkerPool = (
|
|||
// (see captureWorkerStderr) and attach to readiness-failure messages —
|
||||
// instead of the generic "did not report ready" that hid the real cause
|
||||
// in #1741. Test factories (workerFactory) are used verbatim.
|
||||
const spawnWorker = options?.workerFactory ?? ((url: URL) => new Worker(url, { stderr: true }));
|
||||
// Bake the (immutable) ParsedFile store path into the factory closure so it
|
||||
// reaches EVERY spawned worker — including respawns, which reuse this same
|
||||
// factory — via `workerData`, read once at worker init. The `(url) => Worker`
|
||||
// signature is unchanged so the zero-arg test factories keep working.
|
||||
const parsedFileStoreStoragePath = options?.parsedFileStoreStoragePath;
|
||||
const spawnWorker =
|
||||
options?.workerFactory ??
|
||||
((url: URL) =>
|
||||
new Worker(url, {
|
||||
stderr: true,
|
||||
workerData: parsedFileStoreStoragePath ? { parsedFileStoreStoragePath } : undefined,
|
||||
}));
|
||||
/** Spawn + wire stderr capture in one step (used by all spawn sites). */
|
||||
const spawnAndCapture = (url: URL): Worker => {
|
||||
const worker = spawnWorker(url);
|
||||
|
|
|
|||
|
|
@ -132,8 +132,8 @@ export interface AnalyzeOptions {
|
|||
* Worker pool size override, threaded from the CLI `--workers` flag.
|
||||
* Forwarded to `PipelineOptions.workerPoolSize` so the parse phase
|
||||
* sizes the pool without `analyzeCommand` mutating `process.env`.
|
||||
* `0` disables the pool (sequential fallback); positive integer sets
|
||||
* the count; `undefined` defers to the env / auto-formula fallback.
|
||||
* Must be a positive integer — `0` hard-errors (sequential parsing was
|
||||
* removed); `undefined` defers to the env / auto-formula fallback.
|
||||
*/
|
||||
workerPoolSize?: number;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -154,13 +154,13 @@ export const computeChunkHash = (
|
|||
const MAP_TAG = '__$mapEntries$__';
|
||||
const SET_TAG = '__$setValues$__';
|
||||
|
||||
const mapReplacer = (_key: string, value: unknown): unknown => {
|
||||
export const mapReplacer = (_key: string, value: unknown): unknown => {
|
||||
if (value instanceof Map) return { [MAP_TAG]: Array.from(value.entries()) };
|
||||
if (value instanceof Set) return { [SET_TAG]: Array.from(value.values()) };
|
||||
return value;
|
||||
};
|
||||
|
||||
const mapReviver = (_key: string, value: unknown): unknown => {
|
||||
export const mapReviver = (_key: string, value: unknown): unknown => {
|
||||
if (value && typeof value === 'object') {
|
||||
const v = value as Record<string, unknown>;
|
||||
if (Array.isArray(v[MAP_TAG])) return new Map(v[MAP_TAG] as [unknown, unknown][]);
|
||||
|
|
|
|||
216
gitnexus/src/storage/parsedfile-store.ts
Normal file
216
gitnexus/src/storage/parsedfile-store.ts
Normal file
|
|
@ -0,0 +1,216 @@
|
|||
/**
|
||||
* Disk-backed `ParsedFile` store (#1983 scope-resolution OOM).
|
||||
*
|
||||
* ## Why this exists
|
||||
*
|
||||
* The scope-resolution phase needs a `ParsedFile` (scopes / defs / reference
|
||||
* sites) for every file. Historically it re-extracted each file from source on
|
||||
* the **main thread** via `extractParsedFile` → `parseSourceSafe`. On a huge
|
||||
* repo (Linux kernel, ~64k C files) that re-parse accumulates an unbounded
|
||||
* **native** memory leak in `tree-sitter` 0.21.1 (`CallbackInput` retains the
|
||||
* input string with no destructor; node-tree-sitter PR #201) — the leaked
|
||||
* `TSTree` memory is invisible to V8, never reclaimed by GC, and not freed by
|
||||
* worker_thread teardown. The parse phase escapes it only because each parse is
|
||||
* relatively cheap there; a second full re-parse of every file on the immortal
|
||||
* main thread pushes RSS past the heap cap and the OOM-killer fires.
|
||||
*
|
||||
* The fix: the parse workers already build a tree-sitter `Tree` per file, so
|
||||
* they emit the `ParsedFile` directly (reusing that tree — no second parse).
|
||||
* Holding all of them in main-thread heap is what the original #1983 work
|
||||
* removed (it cost ~1× the semantic model in RAM during parse), so instead we
|
||||
* flush them to this disk store per chunk and stream them back per language in
|
||||
* scope-resolution. Net effect: the file is parsed exactly once (in a worker),
|
||||
* scope-resolution does ZERO parsing, and peak heap stays bounded.
|
||||
*
|
||||
* ## Shape
|
||||
*
|
||||
* `<storagePath>/parsedfile-store/<shardId>.json` — one shard per parse chunk,
|
||||
* a JSON array of `ParsedFile` serialized with the same `mapReplacer` the parse
|
||||
* cache uses (Scope.bindings / Scope.typeBindings are `Map`s). The store is
|
||||
* cleared at the start of each parse and after scope-resolution consumes it, so
|
||||
* it never lingers and never goes stale across runs.
|
||||
*/
|
||||
|
||||
import { promises as fs, mkdirSync, writeFileSync } from 'node:fs';
|
||||
import path from 'node:path';
|
||||
import v8 from 'node:v8';
|
||||
import vm from 'node:vm';
|
||||
import type { ParsedFile } from 'gitnexus-shared';
|
||||
import { mapReplacer, mapReviver } from './parse-cache.js';
|
||||
|
||||
const STORE_DIRNAME = 'parsedfile-store';
|
||||
|
||||
/**
|
||||
* Build a JSON.parse reviver that (a) interns every string against a shared
|
||||
* pool and (b) applies the parse-cache `mapReviver` (Map/Set reconstruction).
|
||||
*
|
||||
* `JSON.parse` allocates a DISTINCT string object for every textual token, so a
|
||||
* `ParsedFile` graph round-tripped through disk holds millions of duplicate
|
||||
* strings — every def repeats its `filePath`, and common type/qualified names
|
||||
* (`int`, `void`, `struct …`) recur across the whole repo. On the Linux kernel
|
||||
* that roughly DOUBLES the deserialized heap (~15 GB vs ~7.6 GB interned).
|
||||
* Interning IN the reviver collapses duplicates as the tree is revived (one
|
||||
* pass, no second walk). The pool is per-load; the interned strings stay shared
|
||||
* through the retained `ParsedFile` references after the pool is dropped.
|
||||
*/
|
||||
const makeInterningReviver = (pool: Map<string, string>) => {
|
||||
return (key: string, value: unknown): unknown => {
|
||||
if (typeof value === 'string') {
|
||||
const hit = pool.get(value);
|
||||
if (hit !== undefined) return hit;
|
||||
pool.set(value, value);
|
||||
return value;
|
||||
}
|
||||
return mapReviver(key, value);
|
||||
};
|
||||
};
|
||||
|
||||
/**
|
||||
* Best-effort forced garbage collection. `JSON.parse` of each shard builds a
|
||||
* transient bloated (pre-intern) tree; across hundreds of shards that churn
|
||||
* outpaces V8's incremental GC and piles up against the heap limit (measured
|
||||
* ~5 GB of avoidable transient on the kernel). A periodic full GC during the
|
||||
* load keeps the peak at the retained set rather than retained + churn. Uses
|
||||
* the global `gc` when exposed, else the v8/vm trick — and degrades to a no-op
|
||||
* if neither is available, so it never throws.
|
||||
*/
|
||||
let cachedGc: (() => void) | null | undefined;
|
||||
const forceGc = (): void => {
|
||||
const g = (globalThis as { gc?: () => void }).gc;
|
||||
if (typeof g === 'function') {
|
||||
g();
|
||||
return;
|
||||
}
|
||||
if (cachedGc === undefined) {
|
||||
cachedGc = null;
|
||||
try {
|
||||
v8.setFlagsFromString('--expose-gc');
|
||||
cachedGc = vm.runInNewContext('gc') as () => void;
|
||||
v8.setFlagsFromString('--no-expose-gc');
|
||||
} catch {
|
||||
cachedGc = null;
|
||||
}
|
||||
}
|
||||
cachedGc?.();
|
||||
};
|
||||
|
||||
export const getParsedFileStoreDir = (storagePath: string): string =>
|
||||
path.join(storagePath, STORE_DIRNAME);
|
||||
|
||||
/** Remove any prior run's shards so a fresh parse starts clean. Idempotent. */
|
||||
export const clearParsedFileStore = async (storagePath: string): Promise<void> => {
|
||||
await fs.rm(getParsedFileStoreDir(storagePath), { recursive: true, force: true });
|
||||
};
|
||||
|
||||
/**
|
||||
* Single source of truth for a shard's bytes. Returns `null` for an empty
|
||||
* chunk (caller writes nothing). Both the async (`persistParsedFileChunk`) and
|
||||
* sync (`persistParsedFileShardSync`) writers go through this so the two paths
|
||||
* are guaranteed byte-identical — the shards must round-trip through the same
|
||||
* `mapReviver`, and matching bytes by having both authors type the same
|
||||
* `mapReplacer` call would be a coincidence, not a guarantee.
|
||||
*/
|
||||
const serializeParsedFileShard = (parsedFiles: readonly ParsedFile[]): string | null => {
|
||||
if (parsedFiles.length === 0) return null;
|
||||
return JSON.stringify(parsedFiles, mapReplacer);
|
||||
};
|
||||
|
||||
const shardPath = (storagePath: string, shardId: string): string =>
|
||||
path.join(getParsedFileStoreDir(storagePath), `${shardId}.json`);
|
||||
|
||||
/**
|
||||
* Write one parse chunk's `ParsedFile[]` to the store as a single shard (async).
|
||||
* No-op for an empty chunk. `shardId` must be unique within a run. Used by the
|
||||
* main-thread no-store-disabled fallback and any non-worker writer; the worker
|
||||
* store path uses {@link persistParsedFileShardSync}.
|
||||
*/
|
||||
export const persistParsedFileChunk = async (
|
||||
storagePath: string,
|
||||
shardId: string,
|
||||
parsedFiles: readonly ParsedFile[],
|
||||
): Promise<void> => {
|
||||
const payload = serializeParsedFileShard(parsedFiles);
|
||||
if (payload === null) return;
|
||||
await fs.mkdir(getParsedFileStoreDir(storagePath), { recursive: true });
|
||||
await fs.writeFile(shardPath(storagePath, shardId), payload, 'utf-8');
|
||||
};
|
||||
|
||||
// Per-process set of store dirs we've already `mkdir`ed, so the sync worker
|
||||
// writer (called once per job, many times into the same dir) doesn't issue a
|
||||
// `mkdirSync` syscall on every shard. Mirrors parse-cache.ts's `createdCacheDirs`.
|
||||
const createdStoreDirs = new Set<string>();
|
||||
|
||||
/**
|
||||
* Synchronous shard writer for use INSIDE a parse worker (#1983 parallel
|
||||
* serialization). The worker is a dedicated thread, so a blocking write there
|
||||
* protects the main thread, and a sync write avoids threading `async`/`await`
|
||||
* through the synchronous per-file extract loop. Produces byte-identical shards
|
||||
* to {@link persistParsedFileChunk} via the shared {@link serializeParsedFileShard}.
|
||||
* No-op for an empty chunk. `shardId` must be globally unique for the run (the
|
||||
* worker uses `w<threadId>-<seq>`); a duplicate would silently overwrite.
|
||||
*/
|
||||
export const persistParsedFileShardSync = (
|
||||
storagePath: string,
|
||||
shardId: string,
|
||||
parsedFiles: readonly ParsedFile[],
|
||||
): void => {
|
||||
const payload = serializeParsedFileShard(parsedFiles);
|
||||
if (payload === null) return;
|
||||
const dir = getParsedFileStoreDir(storagePath);
|
||||
if (!createdStoreDirs.has(dir)) {
|
||||
mkdirSync(dir, { recursive: true });
|
||||
createdStoreDirs.add(dir);
|
||||
}
|
||||
writeFileSync(shardPath(storagePath, shardId), payload, 'utf-8');
|
||||
};
|
||||
|
||||
/**
|
||||
* Stream the store and return the `ParsedFile`s whose `filePath` is in
|
||||
* `wantPaths`, keyed by path. Loads one shard at a time and retains only the
|
||||
* matching entries, so peak heap is bounded by (matched set) + (one shard)
|
||||
* rather than the whole store. Returns an empty map when the store is absent
|
||||
* (e.g. tests, or a run with no worker pool) — callers fall back to a fresh
|
||||
* extract for the missing files.
|
||||
*/
|
||||
export const loadParsedFilesForPaths = async (
|
||||
storagePath: string,
|
||||
wantPaths: ReadonlySet<string>,
|
||||
): Promise<Map<string, ParsedFile>> => {
|
||||
const out = new Map<string, ParsedFile>();
|
||||
if (wantPaths.size === 0) return out;
|
||||
const dir = getParsedFileStoreDir(storagePath);
|
||||
let shards: string[];
|
||||
try {
|
||||
shards = (await fs.readdir(dir)).filter((f) => f.endsWith('.json'));
|
||||
} catch {
|
||||
return out; // store absent
|
||||
}
|
||||
// Shared interning pool for this load — deduplicates strings ACROSS shards
|
||||
// (one `int` / one repeated filePath for the whole language), which is where
|
||||
// most of the saving comes from. Dropped when this function returns.
|
||||
const pool = new Map<string, string>();
|
||||
const reviver = makeInterningReviver(pool);
|
||||
for (let i = 0; i < shards.length; i++) {
|
||||
let parsed: ParsedFile[];
|
||||
try {
|
||||
const raw = await fs.readFile(path.join(dir, shards[i]), 'utf-8');
|
||||
parsed = JSON.parse(raw, reviver) as ParsedFile[];
|
||||
} catch {
|
||||
continue; // skip a corrupt shard; missing files fall back to fresh extract
|
||||
}
|
||||
if (!Array.isArray(parsed)) continue;
|
||||
for (const pf of parsed) {
|
||||
if (pf && typeof pf.filePath === 'string' && wantPaths.has(pf.filePath)) {
|
||||
out.set(pf.filePath, pf);
|
||||
}
|
||||
}
|
||||
// Every few shards, reclaim the transient pre-intern parse churn before it
|
||||
// piles up against the heap limit (~5 GB avoidable on the kernel), and
|
||||
// yield so the GC + any pending I/O can run.
|
||||
if ((i & 7) === 7) {
|
||||
forceGc();
|
||||
await new Promise<void>((resolve) => setImmediate(resolve));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
};
|
||||
79
gitnexus/test/helpers/worker-parse.ts
Normal file
79
gitnexus/test/helpers/worker-parse.ts
Normal file
|
|
@ -0,0 +1,79 @@
|
|||
/**
|
||||
* Worker-backed parse helper for tests.
|
||||
*
|
||||
* Since the sequential (in-process) parser was removed, the worker pool is
|
||||
* GitNexus's only parse path. Tests that used to call `processParsing` with no
|
||||
* pool (driving the in-process parser) now route in-memory fixture files
|
||||
* through a REAL worker pool here and assert on the resulting graph exactly as
|
||||
* before — the assertions are about graph content, not which path produced it.
|
||||
*
|
||||
* Requires the compiled `dist/.../parse-worker.js`. The integration test tier
|
||||
* builds it via `pretest:integration`; unit-tier runs do not, so suites using
|
||||
* this helper must live under `test/integration/` (or otherwise ensure a build).
|
||||
* Use {@link distWorkerExists} to guard/skip when the build is absent.
|
||||
*/
|
||||
import path from 'node:path';
|
||||
import fs from 'node:fs';
|
||||
import { fileURLToPath, pathToFileURL } from 'node:url';
|
||||
import { createWorkerPool } from '../../src/core/ingestion/workers/worker-pool.js';
|
||||
import {
|
||||
processParsing,
|
||||
type WorkerExtractedData,
|
||||
} from '../../src/core/ingestion/parsing-processor.js';
|
||||
import { createASTCache } from '../../src/core/ingestion/ast-cache.js';
|
||||
import { createSemanticModel } from '../../src/core/ingestion/model/semantic-model.js';
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import type { KnowledgeGraph } from '../../src/core/graph/types.js';
|
||||
import type { MutableSemanticModel } from '../../src/core/ingestion/model/index.js';
|
||||
import type { ParseWorkerResult } from '../../src/core/ingestion/workers/parse-worker.js';
|
||||
import type { ExportedTypeMap } from '../../src/core/ingestion/call-processor.js';
|
||||
|
||||
const HELPER_DIR = path.dirname(fileURLToPath(import.meta.url));
|
||||
|
||||
/** The compiled worker the integration tier builds via `pretest:integration`. */
|
||||
export const DIST_WORKER_URL = pathToFileURL(
|
||||
path.resolve(HELPER_DIR, '..', '..', 'dist', 'core', 'ingestion', 'workers', 'parse-worker.js'),
|
||||
);
|
||||
|
||||
/** True when `dist/.../parse-worker.js` exists (i.e. the build has run). */
|
||||
export const distWorkerExists = (): boolean => fs.existsSync(fileURLToPath(DIST_WORKER_URL));
|
||||
|
||||
export interface WorkerParseResult {
|
||||
graph: KnowledgeGraph;
|
||||
model: MutableSemanticModel;
|
||||
data: WorkerExtractedData;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse in-memory fixture files through a real worker pool and return the
|
||||
* populated graph + semantic model. One worker by default — fixtures are tiny
|
||||
* and a single worker keeps startup cost down while exercising the real
|
||||
* dispatch + merge path.
|
||||
*/
|
||||
export const parseFilesWithWorkers = async (
|
||||
files: { path: string; content: string }[],
|
||||
opts: {
|
||||
poolSize?: number;
|
||||
exportedTypeMap?: ExportedTypeMap;
|
||||
outRawResults?: ParseWorkerResult[];
|
||||
} = {},
|
||||
): Promise<WorkerParseResult> => {
|
||||
const graph = createKnowledgeGraph();
|
||||
const model = createSemanticModel();
|
||||
const pool = createWorkerPool(DIST_WORKER_URL, opts.poolSize ?? 1);
|
||||
try {
|
||||
const data = await processParsing(
|
||||
graph,
|
||||
files,
|
||||
model.symbols,
|
||||
createASTCache(Math.max(files.length, 1)),
|
||||
pool,
|
||||
undefined,
|
||||
opts.outRawResults,
|
||||
opts.exportedTypeMap,
|
||||
);
|
||||
return { graph, model, data };
|
||||
} finally {
|
||||
await pool.terminate();
|
||||
}
|
||||
};
|
||||
|
|
@ -1,19 +1,8 @@
|
|||
import { describe, expect, it } from 'vitest';
|
||||
import { createASTCache } from '../../src/core/ingestion/ast-cache.js';
|
||||
import { processParsing } from '../../src/core/ingestion/parsing-processor.js';
|
||||
import { createSemanticModel } from '../../src/core/ingestion/model/semantic-model.js';
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import { parseFilesWithWorkers } from '../helpers/worker-parse.js';
|
||||
|
||||
const parseNodes = async (path: string, content: string) => {
|
||||
const graph = createKnowledgeGraph();
|
||||
const model = createSemanticModel();
|
||||
await processParsing(
|
||||
graph,
|
||||
[{ path, content }],
|
||||
model.symbols,
|
||||
createASTCache(),
|
||||
createASTCache(),
|
||||
);
|
||||
const { graph } = await parseFilesWithWorkers([{ path, content }]);
|
||||
return graph.nodes;
|
||||
};
|
||||
|
||||
|
|
|
|||
|
|
@ -9,10 +9,13 @@
|
|||
*
|
||||
* Run: GITNEXUS_BENCH=1 npx vitest run test/integration/cpp-adl-benchmark.test.ts
|
||||
*
|
||||
* WHY EMIT MS, NOT WALL TIME: the fixture is parsed single-threaded
|
||||
* (workerPoolSize: 0, so no dist build is needed), and parse dominates total
|
||||
* wall time — masking the ADL cost. We isolate the scope-resolution `emit` ms
|
||||
* from the profiler log (captured in-process via the logger test destination).
|
||||
* WHY EMIT MS, NOT WALL TIME: parse dominates total wall time — masking the ADL
|
||||
* cost — so we isolate the scope-resolution `emit` ms from the profiler log
|
||||
* (captured in-process via the logger test destination). Because the metric is
|
||||
* the emit-ms RATIO (downstream of and independent from parsing), it is robust
|
||||
* to the parse path: the fixture is parsed with a single-worker pool
|
||||
* (`workerPoolSize: 1`) since the sequential parser was removed. Requires the
|
||||
* built `dist/parse-worker.js` (run `npm run build` first).
|
||||
*
|
||||
* WHY CO-SCALE FILES AND SITES: the regression is O(sites × files). At fixed
|
||||
* files, both the old and new code are linear in sites and indistinguishable.
|
||||
|
|
@ -94,7 +97,7 @@ async function runBenchmark(fileCount: number, siteCount: number): Promise<Bench
|
|||
const cap = _captureLogger();
|
||||
try {
|
||||
const start = Date.now();
|
||||
const result = await runPipelineFromRepo(dir, () => {}, { workerPoolSize: 0 });
|
||||
const result = await runPipelineFromRepo(dir, () => {}, { workerPoolSize: 1 });
|
||||
const elapsedMs = Date.now() - start;
|
||||
const emitMs = extractEmitMs(cap.records());
|
||||
|
||||
|
|
|
|||
|
|
@ -9,8 +9,12 @@
|
|||
*
|
||||
* Run: GITNEXUS_BENCH=1 npx vitest run test/integration/cpp-pipeline-benchmark.test.ts
|
||||
*
|
||||
* Runs build-free (workerPoolSize: 0 → no dist/parse-worker.js needed), so it
|
||||
* parses single-threaded; scales are kept modest accordingly.
|
||||
* Parses with a single-worker pool (`workerPoolSize: 1`) — the sequential
|
||||
* parser was removed, so the worker pool is the only parse path. NOTE: this
|
||||
* needs the built `dist/parse-worker.js`; run `npm run build` first. The
|
||||
* wall-clock numbers now include 1-worker IPC overhead, so re-baseline before
|
||||
* trusting the timeRatio margin and confirm the node-ratio guard below still
|
||||
* trips on an injected O(n²) regression. Scales are kept modest accordingly.
|
||||
*
|
||||
* IMPORTANT — this benchmark measures scaling in FILE COUNT, so per-file work
|
||||
* must stay constant as fileCount grows. Each translation unit therefore
|
||||
|
|
@ -123,7 +127,7 @@ async function runBenchmark(fileCount: number, budgetMs: number): Promise<BenchR
|
|||
try {
|
||||
const start = Date.now();
|
||||
const result = await Promise.race([
|
||||
runPipelineFromRepo(dir, () => {}, { workerPoolSize: 0 }),
|
||||
runPipelineFromRepo(dir, () => {}, { workerPoolSize: 1 }),
|
||||
new Promise<never>((_, reject) =>
|
||||
setTimeout(
|
||||
() => reject(new Error(`Pipeline exceeded ${budgetMs}ms at ${fileCount} files`)),
|
||||
|
|
|
|||
|
|
@ -52,9 +52,7 @@ describe('FastAPI include_router(prefix=…) — ingestion pipeline', () => {
|
|||
// fallback, which historically does NOT run the FastAPI router
|
||||
// bindings extractor — the very behaviour we want to pin lives
|
||||
// exclusively inside the worker entry point.
|
||||
result = await runPipelineFromRepo(FIXTURE, () => {}, {
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
});
|
||||
result = await runPipelineFromRepo(FIXTURE, () => {}, {});
|
||||
}, 60_000);
|
||||
|
||||
function routeNames(): string[] {
|
||||
|
|
|
|||
|
|
@ -6,10 +6,7 @@ import {
|
|||
walkRepositoryPaths,
|
||||
readFileContents,
|
||||
} from '../../src/core/ingestion/filesystem-walker.js';
|
||||
import { processParsing } from '../../src/core/ingestion/parsing-processor.js';
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import { createSymbolTable } from '../../src/core/ingestion/model/symbol-table.js';
|
||||
import { createASTCache } from '../../src/core/ingestion/ast-cache.js';
|
||||
import { parseFilesWithWorkers } from '../helpers/worker-parse.js';
|
||||
import { isLanguageAvailable } from '../../src/core/tree-sitter/parser-loader.js';
|
||||
import { SupportedLanguages } from '../../src/config/supported-languages.js';
|
||||
|
||||
|
|
@ -120,13 +117,10 @@ describe('ignore + language-skip E2E', () => {
|
|||
content,
|
||||
}));
|
||||
|
||||
// Phase 3: parse (sequential — no worker pool)
|
||||
const graph = createKnowledgeGraph();
|
||||
const symbolTable = createSymbolTable();
|
||||
const astCache = createASTCache();
|
||||
|
||||
// Should NOT throw even if Swift grammar is unavailable
|
||||
await processParsing(graph, files, symbolTable, astCache);
|
||||
// Phase 3: parse through the worker pool (the sole parse path).
|
||||
// Should NOT throw even if the Swift grammar is unavailable — the
|
||||
// worker skips files whose native parser can't load.
|
||||
const { graph } = await parseFilesWithWorkers(files);
|
||||
|
||||
// TypeScript files should produce Function nodes
|
||||
const nodes = graph.nodes;
|
||||
|
|
|
|||
|
|
@ -0,0 +1,47 @@
|
|||
/**
|
||||
* TypeScript object-literal method exports — real-parse coverage.
|
||||
*
|
||||
* Relocated from `test/unit/parsing-worker-fallback.test.ts` when the
|
||||
* sequential parser was removed: this asserts on real parse output (a `Const`
|
||||
* for the exported object plus its shorthand `Method`s and `HAS_METHOD` edges),
|
||||
* so it must run through a real worker pool rather than the deleted in-process
|
||||
* parser. Lives in the integration tier where the dist worker is built.
|
||||
*/
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { parseFilesWithWorkers } from '../helpers/worker-parse.js';
|
||||
|
||||
describe('TypeScript object literal method exports', () => {
|
||||
it('links exported object literal shorthand methods back to the exported object', async () => {
|
||||
const { graph } = await parseFilesWithWorkers([
|
||||
{
|
||||
path: 'src/foo.ts',
|
||||
content: `export const fooService = {
|
||||
async getUser(id: string) {
|
||||
return findUser(id);
|
||||
},
|
||||
saveUser(user: User) {
|
||||
return persist(user);
|
||||
},
|
||||
};
|
||||
`,
|
||||
},
|
||||
]);
|
||||
|
||||
const service = graph.nodes.find(
|
||||
(node) => node.label === 'Const' && node.properties.name === 'fooService',
|
||||
);
|
||||
expect(service, 'exported object literal should be captured as a Const').toBeDefined();
|
||||
|
||||
const methodNames = new Set(
|
||||
graph.nodes.filter((node) => node.label === 'Method').map((node) => node.properties.name),
|
||||
);
|
||||
expect(methodNames).toEqual(new Set(['getUser', 'saveUser']));
|
||||
|
||||
const linkedMethodNames = graph.relationships
|
||||
.filter((rel) => rel.type === 'HAS_METHOD' && rel.sourceId === service!.id)
|
||||
.map((rel) => graph.getNode(rel.targetId)?.properties.name)
|
||||
.sort();
|
||||
|
||||
expect(linkedMethodNames).toEqual(['getUser', 'saveUser']);
|
||||
});
|
||||
});
|
||||
|
|
@ -17,12 +17,11 @@
|
|||
* caller. Asserting the edge directly avoids wiring an entire `withTestLbugDB`
|
||||
* fixture for what is effectively a graph-shape assertion.
|
||||
*
|
||||
* Test set:
|
||||
* - Test A: sequential pipeline produces both edges with the right `ownerId`
|
||||
* - Test B: worker-mode pipeline produces identical edge sets (skipped when
|
||||
* `dist/parse-worker.js` is missing; CI builds it before running tests)
|
||||
* Test set (all through the worker pool — the sole parse path; skipped locally
|
||||
* when `dist/parse-worker.js` is missing, with a CI tripwire so CI never skips):
|
||||
* - Test A: pipeline produces both edges with the right `ownerId`
|
||||
* - Test C: local-scoped object literal inside a function emits no false-
|
||||
* positive HAS_METHOD (proves U1 boundary guard is load-bearing)
|
||||
* positive HAS_METHOD (proves the boundary guard is load-bearing)
|
||||
* - Test D: nested object literal binds neither method to outer (safe
|
||||
* under-approximation proof)
|
||||
*/
|
||||
|
|
@ -50,12 +49,11 @@ const DIST_WORKER = path.resolve(
|
|||
);
|
||||
const hasDistWorker = fs.existsSync(DIST_WORKER);
|
||||
|
||||
// CI tripwire: worker-parity test (Test B below) silently skips when
|
||||
// `dist/parse-worker.js` is missing. That's fine locally — devs may not
|
||||
// have run `npm run build` — but on CI a missing dist would mean U3
|
||||
// (worker-path ownerId emission) is unverified. Fail hard so a missing
|
||||
// dist surfaces as a red build, not a green test with a silent skip.
|
||||
// Locally, run `npm run build` before this suite to exercise worker mode.
|
||||
// CI tripwire: these suites silently skip when `dist/parse-worker.js` is
|
||||
// missing. That's fine locally — devs may not have run `npm run build` — but on
|
||||
// CI a missing dist would leave worker-path ownerId emission unverified. Fail
|
||||
// hard so a missing dist surfaces as a red build, not a silent skip.
|
||||
// Locally, run `npm run build` before this suite.
|
||||
if (!hasDistWorker && process.env.CI) {
|
||||
throw new Error(
|
||||
'dist/parse-worker.js missing on CI — worker-parity test would silently skip. ' +
|
||||
|
|
@ -91,9 +89,9 @@ export function caller(id: string) {
|
|||
}
|
||||
`;
|
||||
|
||||
// ── Test A: sequential pipeline ──────────────────────────────────────────────
|
||||
// ── Test A: worker pipeline ──────────────────────────────────────────────────
|
||||
|
||||
describe('object-literal owner resolution — sequential pipeline (PR #1718)', () => {
|
||||
describe.skipIf(!hasDistWorker)('object-literal owner resolution — worker pipeline', () => {
|
||||
let repoRoot: string;
|
||||
let result: PipelineResult;
|
||||
|
||||
|
|
@ -104,7 +102,6 @@ describe('object-literal owner resolution — sequential pipeline (PR #1718)', (
|
|||
});
|
||||
result = await runPipelineFromRepo(repoRoot, () => undefined, {
|
||||
skipGraphPhases: true,
|
||||
skipWorkers: true,
|
||||
});
|
||||
}, 60000);
|
||||
|
||||
|
|
@ -160,120 +157,82 @@ describe('object-literal owner resolution — sequential pipeline (PR #1718)', (
|
|||
});
|
||||
});
|
||||
|
||||
// ── Test B: worker-mode parity ───────────────────────────────────────────────
|
||||
|
||||
describe.skipIf(!hasDistWorker)('object-literal owner resolution — worker parity', () => {
|
||||
let repoRoot: string;
|
||||
let sequentialResult: PipelineResult;
|
||||
let workerResult: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
repoRoot = writeFixture({
|
||||
'src/service.ts': SERVICE_TS,
|
||||
'src/consumer.ts': CONSUMER_TS,
|
||||
});
|
||||
sequentialResult = await runPipelineFromRepo(repoRoot, () => undefined, {
|
||||
skipGraphPhases: true,
|
||||
skipWorkers: true,
|
||||
});
|
||||
workerResult = await runPipelineFromRepo(repoRoot, () => undefined, {
|
||||
skipGraphPhases: true,
|
||||
skipWorkers: false,
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
});
|
||||
}, 90000);
|
||||
|
||||
afterAll(() => removeFixture(repoRoot));
|
||||
|
||||
it('produces the same HAS_METHOD edge set as sequential', () => {
|
||||
const seqEdges = getRelationships(sequentialResult, 'HAS_METHOD')
|
||||
.map((e) => `${e.source}->${e.target}`)
|
||||
.sort();
|
||||
const workerEdges = getRelationships(workerResult, 'HAS_METHOD')
|
||||
.map((e) => `${e.source}->${e.target}`)
|
||||
.sort();
|
||||
expect(workerEdges).toEqual(seqEdges);
|
||||
});
|
||||
|
||||
it('produces the same CALLS edge set as sequential', () => {
|
||||
const seqEdges = getRelationships(sequentialResult, 'CALLS')
|
||||
.map((e) => `${e.source}->${e.target}`)
|
||||
.sort();
|
||||
const workerEdges = getRelationships(workerResult, 'CALLS')
|
||||
.map((e) => `${e.source}->${e.target}`)
|
||||
.sort();
|
||||
expect(workerEdges).toEqual(seqEdges);
|
||||
});
|
||||
});
|
||||
// (Former Test B — worker-vs-sequential parity — removed: the sequential parser
|
||||
// was deleted, so there is no second mode to diff. Test A above already proves
|
||||
// the worker path emits the HAS_METHOD / CALLS edges with the right ownerId.)
|
||||
|
||||
// ── Test C: negative — local object literal inside a function body ──────────
|
||||
|
||||
describe('object-literal owner resolution — negative (local literal)', () => {
|
||||
let repoRoot: string;
|
||||
let result: PipelineResult;
|
||||
describe.skipIf(!hasDistWorker)(
|
||||
'object-literal owner resolution — negative (local literal)',
|
||||
() => {
|
||||
let repoRoot: string;
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
repoRoot = writeFixture({
|
||||
'src/p.ts': `export function processAll() {
|
||||
beforeAll(async () => {
|
||||
repoRoot = writeFixture({
|
||||
'src/p.ts': `export function processAll() {
|
||||
const handler = { run(id: string) { return id; } };
|
||||
return handler;
|
||||
}
|
||||
`,
|
||||
});
|
||||
result = await runPipelineFromRepo(repoRoot, () => undefined, {
|
||||
skipGraphPhases: true,
|
||||
skipWorkers: true,
|
||||
});
|
||||
}, 60000);
|
||||
});
|
||||
result = await runPipelineFromRepo(repoRoot, () => undefined, {
|
||||
skipGraphPhases: true,
|
||||
});
|
||||
}, 60000);
|
||||
|
||||
afterAll(() => removeFixture(repoRoot));
|
||||
afterAll(() => removeFixture(repoRoot));
|
||||
|
||||
it('emits no HAS_METHOD edge targeting `run` (no false-positive owner attribution)', () => {
|
||||
const hasMethod = getRelationships(result, 'HAS_METHOD');
|
||||
const targetingRun = hasMethod.filter((e) => e.target === 'run');
|
||||
expect(targetingRun.length).toBe(0);
|
||||
});
|
||||
|
||||
it('the run method node carries no ownerId property', () => {
|
||||
let runNode: { properties: { name: string; ownerId?: string }; label: string } | undefined;
|
||||
result.graph.forEachNode((n) => {
|
||||
if (n.label === 'Method' && n.properties.name === 'run') {
|
||||
runNode = n as typeof runNode;
|
||||
}
|
||||
it('emits no HAS_METHOD edge targeting `run` (no false-positive owner attribution)', () => {
|
||||
const hasMethod = getRelationships(result, 'HAS_METHOD');
|
||||
const targetingRun = hasMethod.filter((e) => e.target === 'run');
|
||||
expect(targetingRun.length).toBe(0);
|
||||
});
|
||||
expect(runNode).toBeDefined();
|
||||
expect(runNode!.properties.ownerId).toBe(undefined);
|
||||
});
|
||||
});
|
||||
|
||||
it('the run method node carries no ownerId property', () => {
|
||||
let runNode: { properties: { name: string; ownerId?: string }; label: string } | undefined;
|
||||
result.graph.forEachNode((n) => {
|
||||
if (n.label === 'Method' && n.properties.name === 'run') {
|
||||
runNode = n as typeof runNode;
|
||||
}
|
||||
});
|
||||
expect(runNode).toBeDefined();
|
||||
expect(runNode!.properties.ownerId).toBe(undefined);
|
||||
});
|
||||
},
|
||||
);
|
||||
|
||||
// ── Test D: negative — nested object literal ─────────────────────────────────
|
||||
|
||||
describe('object-literal owner resolution — negative (nested literal)', () => {
|
||||
let repoRoot: string;
|
||||
let result: PipelineResult;
|
||||
describe.skipIf(!hasDistWorker)(
|
||||
'object-literal owner resolution — negative (nested literal)',
|
||||
() => {
|
||||
let repoRoot: string;
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
repoRoot = writeFixture({
|
||||
'src/n.ts': `export const s = {
|
||||
beforeAll(async () => {
|
||||
repoRoot = writeFixture({
|
||||
'src/n.ts': `export const s = {
|
||||
nested: { method(id: string) { return id; } },
|
||||
outer(id: string) { return id; },
|
||||
};
|
||||
`,
|
||||
});
|
||||
result = await runPipelineFromRepo(repoRoot, () => undefined, {
|
||||
skipGraphPhases: true,
|
||||
skipWorkers: true,
|
||||
});
|
||||
}, 60000);
|
||||
});
|
||||
result = await runPipelineFromRepo(repoRoot, () => undefined, {
|
||||
skipGraphPhases: true,
|
||||
});
|
||||
}, 60000);
|
||||
|
||||
afterAll(() => removeFixture(repoRoot));
|
||||
afterAll(() => removeFixture(repoRoot));
|
||||
|
||||
it('binds the top-level outer method to s but does NOT bind the nested method', () => {
|
||||
const hasMethod = getRelationships(result, 'HAS_METHOD');
|
||||
const fromS = hasMethod
|
||||
.filter((e) => e.source === 's')
|
||||
.map((e) => e.target)
|
||||
.sort();
|
||||
expect(fromS).toEqual(['outer']);
|
||||
});
|
||||
});
|
||||
it('binds the top-level outer method to s but does NOT bind the nested method', () => {
|
||||
const hasMethod = getRelationships(result, 'HAS_METHOD');
|
||||
const fromS = hasMethod
|
||||
.filter((e) => e.source === 's')
|
||||
.map((e) => e.target)
|
||||
.sort();
|
||||
expect(fromS).toEqual(['outer']);
|
||||
});
|
||||
},
|
||||
);
|
||||
|
|
|
|||
|
|
@ -53,13 +53,11 @@ describe('parse-impl chunk concurrency (U1)', () => {
|
|||
|
||||
const g1 = createKnowledgeGraph();
|
||||
await runChunkedParseAndResolve(g1, scan, files, files.length, repoPath, Date.now(), () => {}, {
|
||||
skipWorkers: true,
|
||||
parseChunkConcurrency: 1,
|
||||
});
|
||||
|
||||
const g2 = createKnowledgeGraph();
|
||||
await runChunkedParseAndResolve(g2, scan, files, files.length, repoPath, Date.now(), () => {}, {
|
||||
skipWorkers: true,
|
||||
parseChunkConcurrency: 2,
|
||||
});
|
||||
|
||||
|
|
@ -82,7 +80,7 @@ describe('parse-impl chunk concurrency (U1)', () => {
|
|||
repoPath,
|
||||
Date.now(),
|
||||
() => {},
|
||||
{ skipWorkers: true, parseChunkConcurrency: 1 },
|
||||
{ parseChunkConcurrency: 1 },
|
||||
);
|
||||
// Exact assertions per DoD §2.7: pin specific symbols from the fixture
|
||||
// so a regression in either the chunk loop or the resolver surfaces
|
||||
|
|
@ -109,7 +107,7 @@ describe('parse-impl chunk concurrency (U1)', () => {
|
|||
repoPath,
|
||||
Date.now(),
|
||||
() => {},
|
||||
{ skipWorkers: true },
|
||||
{},
|
||||
);
|
||||
// Resolver reads the env when options.parseChunkConcurrency is
|
||||
// undefined. The env value (3) must produce the same fixture
|
||||
|
|
@ -75,7 +75,11 @@ async function countChunksFromProgress(
|
|||
const m = /Parsing chunk (\d+)\/(\d+)/.exec(p.message);
|
||||
if (m !== null) chunkIndices.add(`${m[1]}/${m[2]}`);
|
||||
},
|
||||
{ skipWorkers: true, ...options },
|
||||
// Chunk count is byte-budget-driven and emitted before the pool runs, so it
|
||||
// is independent of worker vs sequential. Sequential parsing was removed, so
|
||||
// this runs through the (auto-sized, dist-backed) worker pool — hence the
|
||||
// integration tier.
|
||||
{ ...options },
|
||||
);
|
||||
return chunkIndices.size;
|
||||
}
|
||||
|
|
@ -12,15 +12,13 @@
|
|||
* wall-clock so a regression that re-introduces the hang fails this
|
||||
* test loudly instead of slipping past via inequality assertions.
|
||||
*
|
||||
* Scope note: this test runs the sequential-fallback path (skipWorkers).
|
||||
* The full "real workers + actually-pathological file" scenario from
|
||||
* the plan requires a built `dist/parse-worker.js` and ~60s wall-clock
|
||||
* per run, which is more appropriate for a CI-integration job than a
|
||||
* vitest. Once the dist worker is wired into the test harness (a Phase 2
|
||||
* follow-up), this file can be extended to swap skipWorkers off. The
|
||||
* load-bearing invariants verified here — multi-chunk parsing
|
||||
* completes within a bounded budget and produces all expected symbols
|
||||
* — catch the bulk of the regressions B3 was concerned about.
|
||||
* This runs the real worker pool (the sole parse path since the sequential
|
||||
* parser was removed) on a multi-chunk fixture, BOUNDED by a wall-clock so a
|
||||
* regression that re-introduces the hang fails loudly instead of slipping past
|
||||
* via inequality assertions. The load-bearing invariants — multi-chunk parsing
|
||||
* completes within a bounded budget and produces all expected symbols — catch
|
||||
* the bulk of the regressions B3 was concerned about. Needs the dist worker
|
||||
* (integration tier).
|
||||
*/
|
||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
|
||||
import fs from 'node:fs';
|
||||
|
|
@ -94,9 +92,7 @@ async function runFixture(): Promise<{
|
|||
// surfaces the actual regression class.
|
||||
const start = Date.now();
|
||||
await Promise.race([
|
||||
runChunkedParseAndResolve(graph, scanned, files, files.length, repoPath, start, () => {}, {
|
||||
skipWorkers: true,
|
||||
}),
|
||||
runChunkedParseAndResolve(graph, scanned, files, files.length, repoPath, start, () => {}, {}),
|
||||
new Promise<never>((_, reject) =>
|
||||
setTimeout(
|
||||
() =>
|
||||
|
|
|
|||
|
|
@ -9,9 +9,12 @@
|
|||
*
|
||||
* After M2, parse phase covers 20-70 and deferred extraction covers 70-95
|
||||
* across four labelled sub-bands. This test runs `runChunkedParseAndResolve`
|
||||
* on a small temp repo via the deterministic sequential-fallback path
|
||||
* (`skipWorkers: true`) and asserts the recorded percent stream is strictly
|
||||
* non-decreasing AND reaches the deferred band (>=70) before returning.
|
||||
* on a small temp repo through the worker pool (the sole parse path) and
|
||||
* asserts the recorded percent stream is strictly non-decreasing AND reaches
|
||||
* the deferred band (>=70) before returning. The progress percents come from
|
||||
* the chunk loop / deferred bands, which are identical regardless of how files
|
||||
* are parsed — so this exercises the same progress contract the sequential
|
||||
* path used to. Integration tier because the pool needs the dist worker.
|
||||
*/
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import fs from 'node:fs';
|
||||
|
|
@ -70,7 +73,7 @@ describe('parse-impl progress monotonicity (U4 M2)', () => {
|
|||
(p) => {
|
||||
if (typeof p.percent === 'number') percents.push(p.percent);
|
||||
},
|
||||
{ skipWorkers: true },
|
||||
{},
|
||||
);
|
||||
|
||||
// The stream MUST be non-empty (a regression that stops emitting
|
||||
|
|
@ -96,8 +99,8 @@ describe('parse-impl progress monotonicity (U4 M2)', () => {
|
|||
const reachedDeferredBand = percents.some((p) => p >= 70 && p <= 95);
|
||||
expect(reachedDeferredBand).toBe(true);
|
||||
|
||||
// On this 3-file fixture in skipWorkers mode the deferred band
|
||||
// advances exactly to 70 (the start of the band). The orchestrator
|
||||
// On this 3-file fixture the deferred band advances exactly to 70 (the
|
||||
// start of the band). The orchestrator
|
||||
// (run-analyze) drives 70-100 itself once cross-chunk extraction
|
||||
// finishes. Pinning the exact observed value catches both an
|
||||
// upper-bound regression (anything >70 would unexpectedly land in
|
||||
|
|
@ -121,7 +124,7 @@ describe('parse-impl progress monotonicity (U4 M2)', () => {
|
|||
(p) => {
|
||||
if (typeof p.percent === 'number') percents.push(p.percent);
|
||||
},
|
||||
{ skipWorkers: true },
|
||||
{},
|
||||
);
|
||||
|
||||
// The early-return path must emit 95 (the new post-deferred ceiling),
|
||||
|
|
@ -245,7 +245,6 @@ describe('U20: parse-impl quarantine + chunk-cache integration (PR #1693 Codex f
|
|||
{
|
||||
skipWorkers: false,
|
||||
// Force the worker-pool gate to open on the 3-file fixture.
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
// Inject the custom worker script — the pool will spawn it
|
||||
// instead of the production parse-worker.js.
|
||||
workerUrlForTest: pathToFileURL(workerPath) as URL,
|
||||
|
|
@ -331,7 +330,6 @@ describe('U20: parse-impl quarantine + chunk-cache integration (PR #1693 Codex f
|
|||
() => {},
|
||||
{
|
||||
skipWorkers: false,
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
workerUrlForTest: pathToFileURL(workerPath) as URL,
|
||||
workerPoolSize: 1,
|
||||
parseCache,
|
||||
|
|
@ -361,7 +359,6 @@ describe('U20: parse-impl quarantine + chunk-cache integration (PR #1693 Codex f
|
|||
() => {},
|
||||
{
|
||||
skipWorkers: false,
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
workerUrlForTest: pathToFileURL(workerPath) as URL,
|
||||
workerPoolSize: 1,
|
||||
parseCache,
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@
|
|||
* Run: GITNEXUS_BENCH=1 npx vitest run test/integration/php-pipeline-benchmark.test.ts
|
||||
*
|
||||
* The benchmark uses workers (production path) by default. Set
|
||||
* skipWorkers to test the sequential fallback path.
|
||||
* a single-worker pool (the sequential parser was removed).
|
||||
*/
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import fs from 'node:fs';
|
||||
|
|
|
|||
|
|
@ -1,43 +1,26 @@
|
|||
import { describe, expect, it } from 'vitest';
|
||||
import { createASTCache } from '../../src/core/ingestion/ast-cache.js';
|
||||
import { processParsing } from '../../src/core/ingestion/parsing-processor.js';
|
||||
import { createSemanticModel } from '../../src/core/ingestion/model/semantic-model.js';
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import { parseFilesWithWorkers } from '../helpers/worker-parse.js';
|
||||
|
||||
describe('qualified class lookups', () => {
|
||||
it('derives canonical dot-separated names from namespaces, packages, and modules', async () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
const model = createSemanticModel();
|
||||
// model.symbols is the SymbolTable leaf that processParsing writes into.
|
||||
// Fan-out writes still reach model.types / model.methods / model.fields
|
||||
// via SemanticModel's wrappedAdd — this alias is purely for convenience
|
||||
// at call sites that want the SymbolTable-shaped interface.
|
||||
const symbolTable = model.symbols;
|
||||
const astCache = createASTCache();
|
||||
|
||||
await processParsing(
|
||||
graph,
|
||||
[
|
||||
{
|
||||
path: 'src/Services/User.cs',
|
||||
content: 'namespace Services.Auth;\npublic class User {}\n',
|
||||
},
|
||||
{
|
||||
path: 'src/Data/User.cs',
|
||||
content: 'namespace Data.Auth;\npublic class User {}\n',
|
||||
},
|
||||
{
|
||||
path: 'src/models/Config.java',
|
||||
content: 'package com.example.models;\nclass Config {}\n',
|
||||
},
|
||||
{
|
||||
path: 'lib/admin/user.rb',
|
||||
content: 'module Admin\n class User\n end\nend\n',
|
||||
},
|
||||
],
|
||||
symbolTable,
|
||||
astCache,
|
||||
);
|
||||
const { model } = await parseFilesWithWorkers([
|
||||
{
|
||||
path: 'src/Services/User.cs',
|
||||
content: 'namespace Services.Auth;\npublic class User {}\n',
|
||||
},
|
||||
{
|
||||
path: 'src/Data/User.cs',
|
||||
content: 'namespace Data.Auth;\npublic class User {}\n',
|
||||
},
|
||||
{
|
||||
path: 'src/models/Config.java',
|
||||
content: 'package com.example.models;\nclass Config {}\n',
|
||||
},
|
||||
{
|
||||
path: 'lib/admin/user.rb',
|
||||
content: 'module Admin\n class User\n end\nend\n',
|
||||
},
|
||||
]);
|
||||
|
||||
const userMatches = model.types.lookupClassByName('User');
|
||||
expect(userMatches).toHaveLength(3);
|
||||
|
|
@ -64,21 +47,9 @@ describe('qualified class lookups', () => {
|
|||
});
|
||||
|
||||
it('falls back to the simple name for top-level class-like symbols', async () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
const model = createSemanticModel();
|
||||
// model.symbols is the SymbolTable leaf that processParsing writes into.
|
||||
// Fan-out writes still reach model.types / model.methods / model.fields
|
||||
// via SemanticModel's wrappedAdd — this alias is purely for convenience
|
||||
// at call sites that want the SymbolTable-shaped interface.
|
||||
const symbolTable = model.symbols;
|
||||
const astCache = createASTCache();
|
||||
|
||||
await processParsing(
|
||||
graph,
|
||||
[{ path: 'src/plain-user.ts', content: 'export class User {}\n' }],
|
||||
symbolTable,
|
||||
astCache,
|
||||
);
|
||||
const { model } = await parseFilesWithWorkers([
|
||||
{ path: 'src/plain-user.ts', content: 'export class User {}\n' },
|
||||
]);
|
||||
|
||||
const simpleMatches = model.types.lookupClassByName('User');
|
||||
expect(simpleMatches).toHaveLength(1);
|
||||
|
|
|
|||
|
|
@ -3869,7 +3869,6 @@ describe('C++ inline nested same-tail collision — worker path parity (issue #1
|
|||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'cpp-nested-tail-collision'), () => {}, {
|
||||
// Force the worker-pool gate low so the 1-file fixture engages the pool.
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
workerPoolSize: 2,
|
||||
});
|
||||
}, 120000);
|
||||
|
|
@ -3962,7 +3961,7 @@ describe('C++ named-union nested same-tail collision — worker path parity (iss
|
|||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'cpp-union-nested-tail-collision'),
|
||||
() => {},
|
||||
{ workerThresholdsForTest: { minFiles: 1, minBytes: 1 }, workerPoolSize: 2 },
|
||||
{ workerPoolSize: 2 },
|
||||
);
|
||||
}, 120000);
|
||||
|
||||
|
|
@ -4033,7 +4032,7 @@ describe('C++ anonymous-namespace nested same-tail collision — worker path par
|
|||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'cpp-anon-ns-tail-collision'),
|
||||
() => {},
|
||||
{ workerThresholdsForTest: { minFiles: 1, minBytes: 1 }, workerPoolSize: 2 },
|
||||
{ workerPoolSize: 2 },
|
||||
);
|
||||
}, 120000);
|
||||
|
||||
|
|
@ -4146,7 +4145,6 @@ describe('C++ namespaced same-tail nested heritage — worker path parity (issue
|
|||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'cpp-namespaced-collision'), () => {}, {
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
workerPoolSize: 2,
|
||||
});
|
||||
}, 120000);
|
||||
|
|
@ -4229,7 +4227,7 @@ describe('C++ cross-namespace same-tail nested heritage — worker path parity (
|
|||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'cpp-cross-namespace-same-tail'),
|
||||
() => {},
|
||||
{ workerThresholdsForTest: { minFiles: 1, minBytes: 1 }, workerPoolSize: 2 },
|
||||
{ workerPoolSize: 2 },
|
||||
);
|
||||
}, 120000);
|
||||
|
||||
|
|
|
|||
|
|
@ -2509,7 +2509,7 @@ describe('C# large-file + frozen-bucket regression (issue #1066)', () => {
|
|||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'csharp-large-cache-miss-resolution'),
|
||||
() => {},
|
||||
{ workerThresholdsForTest: { minFiles: 1, minBytes: 0 } },
|
||||
{},
|
||||
);
|
||||
}, 120000);
|
||||
|
||||
|
|
@ -2619,7 +2619,7 @@ describe('C# namespace-as-root with no trailing newline (issue #1086)', () => {
|
|||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'csharp-namespace-as-root-no-trailing-newline'),
|
||||
() => {},
|
||||
{ workerThresholdsForTest: { minFiles: 1, minBytes: 0 } },
|
||||
{},
|
||||
);
|
||||
}, 60000);
|
||||
|
||||
|
|
|
|||
|
|
@ -210,7 +210,7 @@ describe('Go receiver method free-call resolution', () => {
|
|||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'go-receiver-method-free-call'),
|
||||
() => {},
|
||||
{ workerThresholdsForTest: { minFiles: 1, minBytes: 0 } },
|
||||
{},
|
||||
);
|
||||
}, 60000);
|
||||
|
||||
|
|
|
|||
|
|
@ -2178,18 +2178,14 @@ describe('Java overloaded method disambiguation (METHOD_IMPLEMENTS)', () => {
|
|||
|
||||
// ── Phase P: Sequential path parity — same-arity overloads ────────────────
|
||||
|
||||
describe('Java same-arity overloads via sequential path (skipWorkers)', () => {
|
||||
describe('Java same-arity overloads (worker path)', () => {
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'java-same-arity-cross-file'),
|
||||
() => {},
|
||||
{ skipWorkers: true },
|
||||
);
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'java-same-arity-cross-file'), () => {});
|
||||
}, 60000);
|
||||
|
||||
it('produces distinct graph nodes for find(int) and find(String) — sequential path', () => {
|
||||
it('produces distinct graph nodes for find(int) and find(String)', () => {
|
||||
const methods = getNodesByLabelFull(result, 'Method');
|
||||
const findNodes = methods.filter(
|
||||
(m) => m.name === 'find' && m.properties.filePath?.includes('DbLookup'),
|
||||
|
|
|
|||
|
|
@ -22,9 +22,11 @@ describe('Laravel route → controller qualified resolution', () => {
|
|||
// Force the worker path — Laravel route extraction runs in the worker
|
||||
// (the sequential fallback does not extract routes), so a small fixture
|
||||
// must lower the worker thresholds to exercise route resolution.
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'laravel-route-resolution'), () => {}, {
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
});
|
||||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'laravel-route-resolution'),
|
||||
() => {},
|
||||
{},
|
||||
);
|
||||
}, 60000);
|
||||
|
||||
const routeCalls = () =>
|
||||
|
|
|
|||
|
|
@ -2998,9 +2998,7 @@ def create_utf8_user():
|
|||
user.save()
|
||||
`,
|
||||
});
|
||||
result = await runPipelineFromRepo(repoDir, () => {}, {
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 0 },
|
||||
});
|
||||
result = await runPipelineFromRepo(repoDir, () => {}, {});
|
||||
}, 120000);
|
||||
|
||||
afterAll(() => {
|
||||
|
|
|
|||
|
|
@ -1300,13 +1300,11 @@ describe('Ruby method enrichment (visibility, isStatic, parameters)', () => {
|
|||
});
|
||||
});
|
||||
|
||||
describe('Ruby singleton_class handling via sequential path (skipWorkers)', () => {
|
||||
describe('Ruby singleton_class handling (worker path)', () => {
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'ruby-method-enrichment'), () => {}, {
|
||||
skipWorkers: true,
|
||||
});
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'ruby-method-enrichment'), () => {});
|
||||
}, 60000);
|
||||
|
||||
it('keeps Animal as the owner for class << self methods', () => {
|
||||
|
|
@ -1316,7 +1314,7 @@ describe('Ruby singleton_class handling via sequential path (skipWorkers)', () =
|
|||
).toBeDefined();
|
||||
});
|
||||
|
||||
it('marks from_habitat as static in the sequential path', () => {
|
||||
it('marks from_habitat as static', () => {
|
||||
const methods = getNodesByLabelFull(result, 'Method');
|
||||
const fromHabitat = methods.find(
|
||||
(m) => m.name === 'from_habitat' && m.properties.filePath?.includes('animal'),
|
||||
|
|
@ -1597,7 +1595,6 @@ describe('Ruby inline module-nested same-tail collision — worker path parity (
|
|||
path.join(FIXTURES, 'ruby-nested-tail-collision'),
|
||||
() => {},
|
||||
{
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
workerPoolSize: 2,
|
||||
},
|
||||
);
|
||||
|
|
@ -1690,7 +1687,7 @@ describe('Ruby same-tail nested mixin-module collision — worker path parity (i
|
|||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'ruby-nested-mixin-tail-collision'),
|
||||
() => {},
|
||||
{ workerThresholdsForTest: { minFiles: 1, minBytes: 1 }, workerPoolSize: 2 },
|
||||
{ workerPoolSize: 2 },
|
||||
);
|
||||
}, 120000);
|
||||
|
||||
|
|
|
|||
|
|
@ -2156,7 +2156,7 @@ describe('Rust generic inherent-impl ownership — worker path parity (issue #19
|
|||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'rust-nested-tail-collision-generic'),
|
||||
() => {},
|
||||
{ workerThresholdsForTest: { minFiles: 1, minBytes: 1 }, workerPoolSize: 2 },
|
||||
{ workerPoolSize: 2 },
|
||||
);
|
||||
}, 120000);
|
||||
|
||||
|
|
@ -2235,7 +2235,7 @@ describe('Rust same-tail generic impls with shared method name — worker path p
|
|||
result = await runPipelineFromRepo(
|
||||
path.join(FIXTURES, 'rust-generic-impl-same-method-name'),
|
||||
() => {},
|
||||
{ workerThresholdsForTest: { minFiles: 1, minBytes: 1 }, workerPoolSize: 2 },
|
||||
{ workerPoolSize: 2 },
|
||||
);
|
||||
}, 120000);
|
||||
|
||||
|
|
|
|||
|
|
@ -2942,9 +2942,7 @@ export function createUtf8User(): void {
|
|||
}
|
||||
`,
|
||||
});
|
||||
result = await runPipelineFromRepo(repoDir, () => {}, {
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 0 },
|
||||
});
|
||||
result = await runPipelineFromRepo(repoDir, () => {}, {});
|
||||
}, 120000);
|
||||
|
||||
afterAll(() => {
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@
|
|||
* Run: GITNEXUS_BENCH=1 npx vitest run test/integration/ruby-pipeline-benchmark.test.ts
|
||||
*
|
||||
* The benchmark uses workers (production path) by default. Set
|
||||
* skipWorkers to test the sequential fallback path.
|
||||
* a single-worker pool (the sequential parser was removed).
|
||||
*/
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import fs from 'node:fs';
|
||||
|
|
|
|||
|
|
@ -1,79 +0,0 @@
|
|||
/**
|
||||
* Worker-mode vs sequential-mode parity (#1741 / Problem D).
|
||||
*
|
||||
* rc99 produced almost no bindings/edges (13 bindings vs rc91's 106,305)
|
||||
* because a worker-path failure left extracted results unmerged while the run
|
||||
* still reported success. This test pins the invariant that regression broke:
|
||||
* for the same repo, the worker pool and the sequential path must produce the
|
||||
* SAME graph — identical CALLS / IMPORTS / DEFINES / HAS_METHOD edges and the
|
||||
* same defs. A silent divergence (either mode dropping results) fails here
|
||||
* instead of shipping a hollow index.
|
||||
*
|
||||
* Requires the compiled worker (`dist/.../parse-worker.js`); the integration
|
||||
* runner builds it via `pretest:integration`.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll } from 'vitest';
|
||||
import path from 'node:path';
|
||||
import {
|
||||
runPipelineFromRepo,
|
||||
getRelationships,
|
||||
getNodesByLabel,
|
||||
edgeSet,
|
||||
type PipelineResult,
|
||||
} from './resolvers/helpers.js';
|
||||
|
||||
const FIXTURE = path.resolve(__dirname, '..', 'fixtures', 'cross-file-binding', 'ts-simple');
|
||||
|
||||
const runMode = (mode: 'worker' | 'sequential'): Promise<PipelineResult> =>
|
||||
runPipelineFromRepo(FIXTURE, () => {}, {
|
||||
skipGraphPhases: true,
|
||||
// Force the worker-pool gate low so even a 3-file fixture engages the pool
|
||||
// in worker mode (production threshold is 15 files / 512 KB).
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
...(mode === 'worker'
|
||||
? { workerPoolSize: 2 }
|
||||
: // skipWorkers is the explicit "parse sequentially" path (the supported
|
||||
// way to opt out of workers, equivalent to --workers 0). It never
|
||||
// creates a pool, so the #1741 startup fail-fast does not apply.
|
||||
{ skipWorkers: true }),
|
||||
});
|
||||
|
||||
describe('worker vs sequential parity (#1741 Problem D)', () => {
|
||||
let worker: PipelineResult;
|
||||
let sequential: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
worker = await runMode('worker');
|
||||
sequential = await runMode('sequential');
|
||||
}, 120_000);
|
||||
|
||||
it('worker mode genuinely used the pool; sequential did not (guards against silent fallback)', () => {
|
||||
expect(worker.usedWorkerPool).toBe(true);
|
||||
expect(sequential.usedWorkerPool).toBe(false);
|
||||
});
|
||||
|
||||
it.each(['CALLS', 'IMPORTS', 'DEFINES', 'HAS_METHOD'])(
|
||||
'produces identical %s edges in both modes',
|
||||
(relType) => {
|
||||
const w = edgeSet(getRelationships(worker, relType));
|
||||
const s = edgeSet(getRelationships(sequential, relType));
|
||||
expect(w).toEqual(s);
|
||||
},
|
||||
);
|
||||
|
||||
it('does not silently collapse to ~zero edges (the rc99 regression signature)', () => {
|
||||
// A healthy run of this cross-file fixture has real CALLS and IMPORTS in
|
||||
// BOTH modes. The rc99 collapse showed up as near-empty output.
|
||||
expect(edgeSet(getRelationships(worker, 'CALLS')).length).toBeGreaterThan(0);
|
||||
expect(edgeSet(getRelationships(worker, 'IMPORTS')).length).toBeGreaterThan(0);
|
||||
expect(edgeSet(getRelationships(sequential, 'CALLS')).length).toBeGreaterThan(0);
|
||||
expect(edgeSet(getRelationships(sequential, 'IMPORTS')).length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it.each(['Class', 'Function', 'Method'])(
|
||||
'produces identical %s definitions in both modes',
|
||||
(label) => {
|
||||
expect(getNodesByLabel(worker, label)).toEqual(getNodesByLabel(sequential, label));
|
||||
},
|
||||
);
|
||||
});
|
||||
|
|
@ -62,9 +62,7 @@ describe('analyzeCommand --workers validation', () => {
|
|||
expect(
|
||||
cap
|
||||
.records()
|
||||
.some((r) =>
|
||||
String(r.msg ?? '').startsWith(' --workers must be a non-negative integer'),
|
||||
),
|
||||
.some((r) => String(r.msg ?? '').startsWith(' --workers must be a positive integer')),
|
||||
).toBe(true);
|
||||
expect(runFullAnalysisMock).not.toHaveBeenCalled();
|
||||
cap.restore();
|
||||
|
|
@ -89,7 +87,7 @@ describe('analyzeCommand --workers validation', () => {
|
|||
);
|
||||
});
|
||||
|
||||
it('threads --workers 0 as workerPoolSize: 0 (sequential-fallback signal)', async () => {
|
||||
it('rejects --workers 0 with a CLI error (sequential parsing was removed)', async () => {
|
||||
const { analyzeCommand } = await import('../../src/cli/analyze.js');
|
||||
runFullAnalysisMock.mockResolvedValue({
|
||||
repoName: 'repo',
|
||||
|
|
@ -100,12 +98,11 @@ describe('analyzeCommand --workers validation', () => {
|
|||
|
||||
await analyzeCommand(undefined, { workers: '0' });
|
||||
|
||||
expect(runFullAnalysisMock).toHaveBeenCalledWith(
|
||||
expect.any(String),
|
||||
expect.objectContaining({ workerPoolSize: 0 }),
|
||||
expect.any(Object),
|
||||
);
|
||||
expect(process.exitCode).toBeUndefined();
|
||||
// 0 is no longer a "disable the pool" signal — it must error out before the
|
||||
// pipeline runs, not thread workerPoolSize: 0.
|
||||
expect(runFullAnalysisMock).not.toHaveBeenCalled();
|
||||
expect(process.exitCode).toBe(1);
|
||||
process.exitCode = undefined; // reset the global so sibling tests aren't affected
|
||||
});
|
||||
|
||||
it('does not mutate GITNEXUS_WORKER_POOL_SIZE in process.env', async () => {
|
||||
|
|
|
|||
89
gitnexus/test/unit/language-availability-skip.test.ts
Normal file
89
gitnexus/test/unit/language-availability-skip.test.ts
Normal file
|
|
@ -0,0 +1,89 @@
|
|||
/**
|
||||
* Native-parser-unavailable handling.
|
||||
*
|
||||
* When a file's language has no loadable native parser (e.g. a Swift grammar
|
||||
* that didn't build), the parse phase must SKIP the file with a warning — never
|
||||
* crash. The sequential parser used to enforce this in-process; with it removed,
|
||||
* the guarantee lives in the parse phase's pre-dispatch availability filter
|
||||
* (`runChunkedParseAndResolve` → `isLanguageAvailable`), which runs on the MAIN
|
||||
* thread before any worker is spawned. Mocking `parser-loader` exercises that
|
||||
* filter; because the only file is filtered out, no worker pool is created — so
|
||||
* this stays a fast unit test with no dist dependency.
|
||||
*
|
||||
* (Replaces `sequential-language-availability.test.ts`, which drove the deleted
|
||||
* in-process parser and asserted its now-removed log message. `vi.mock` cannot
|
||||
* cross the worker_threads isolate boundary, so the worker's own skip path can't
|
||||
* be exercised this way — the main-thread filter is the testable seam.)
|
||||
*/
|
||||
import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest';
|
||||
|
||||
vi.mock('../../src/core/tree-sitter/parser-loader.js', async (importOriginal) => {
|
||||
const actual =
|
||||
await importOriginal<typeof import('../../src/core/tree-sitter/parser-loader.js')>();
|
||||
return { ...actual, isLanguageAvailable: vi.fn(() => true) };
|
||||
});
|
||||
|
||||
import fs from 'node:fs';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import { runChunkedParseAndResolve } from '../../src/core/ingestion/pipeline-phases/parse-impl.js';
|
||||
import * as parserLoader from '../../src/core/tree-sitter/parser-loader.js';
|
||||
import { _captureLogger } from '../../src/core/logger.js';
|
||||
import type { LoggerCapture } from '../../src/core/logger.js';
|
||||
|
||||
describe('native parser availability — unavailable language is skipped, not crashed', () => {
|
||||
let cap: LoggerCapture | undefined;
|
||||
let repoDir = '';
|
||||
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
repoDir = fs.mkdtempSync(path.join(os.tmpdir(), 'lang-availability-'));
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
cap?.restore();
|
||||
cap = undefined;
|
||||
if (repoDir) fs.rmSync(repoDir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
/** Run the parse phase over a single Swift file in the temp repo. */
|
||||
const runWithSwift = () => {
|
||||
const rel = 'App.swift';
|
||||
fs.writeFileSync(path.join(repoDir, rel), 'class AppViewController: UIViewController {}\n');
|
||||
const scanned = [{ path: rel, size: fs.statSync(path.join(repoDir, rel)).size }];
|
||||
return runChunkedParseAndResolve(
|
||||
createKnowledgeGraph(),
|
||||
scanned,
|
||||
[rel],
|
||||
1,
|
||||
repoDir,
|
||||
Date.now(),
|
||||
() => {},
|
||||
);
|
||||
};
|
||||
|
||||
it('skips the Swift file without crashing (and without spawning a pool) when its parser is unavailable', async () => {
|
||||
vi.mocked(parserLoader.isLanguageAvailable).mockReturnValue(false);
|
||||
// The only file is filtered out before dispatch, so the parse phase
|
||||
// completes (returns its result) instead of throwing, and never needs a
|
||||
// worker pool — `usedWorkerPool` stays false.
|
||||
const result = await runWithSwift();
|
||||
expect(result.usedWorkerPool).toBe(false);
|
||||
});
|
||||
|
||||
it('warns that the unavailable-parser file was skipped', async () => {
|
||||
cap = _captureLogger();
|
||||
vi.mocked(parserLoader.isLanguageAvailable).mockReturnValue(false);
|
||||
await runWithSwift();
|
||||
const warned = cap
|
||||
.records()
|
||||
.some(
|
||||
(r) =>
|
||||
typeof r.msg === 'string' &&
|
||||
r.msg.includes('Skipping 1 swift file(s)') &&
|
||||
r.msg.includes('swift parser not available'),
|
||||
);
|
||||
expect(warned).toBe(true);
|
||||
});
|
||||
});
|
||||
|
|
@ -14,7 +14,6 @@ import { pathToFileURL } from 'node:url';
|
|||
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import { runChunkedParseAndResolve } from '../../src/core/ingestion/pipeline-phases/parse-impl.js';
|
||||
import { buildExportedTypeMapFromGraph } from '../../src/core/ingestion/call-processor.js';
|
||||
import { computeChunkHash, fileContentHash } from '../../src/storage/parse-cache.js';
|
||||
import type { ParseWorkerResult } from '../../src/core/ingestion/workers/parse-worker.js';
|
||||
|
||||
|
|
@ -53,37 +52,6 @@ const emptyWorkerResult = (filePath: string, name: string): ParseWorkerResult =>
|
|||
fileCount: 1,
|
||||
});
|
||||
|
||||
// A cached chunk result carrying an exported, typed symbol — enough to make a
|
||||
// cache-hit replay push exportedTypeMap.size > 0, the precondition that
|
||||
// suppresses the full-graph rebuild and exposes the sequential-miss gap (#2038).
|
||||
const exportedTypedResult = (
|
||||
filePath: string,
|
||||
name: string,
|
||||
returnType: string,
|
||||
): ParseWorkerResult => {
|
||||
const id = `Function:${filePath}:${name}`;
|
||||
return {
|
||||
...emptyWorkerResult(filePath, name),
|
||||
nodes: [
|
||||
{
|
||||
id,
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name,
|
||||
filePath,
|
||||
startLine: 1,
|
||||
endLine: 1,
|
||||
language: 'typescript',
|
||||
isExported: true,
|
||||
},
|
||||
},
|
||||
],
|
||||
symbols: [
|
||||
{ filePath, name, nodeId: id, type: 'Function', returnType },
|
||||
] as ParseWorkerResult['symbols'],
|
||||
};
|
||||
};
|
||||
|
||||
const writeReadyWorker = (workerPath: string, markerPath: string): void => {
|
||||
fs.writeFileSync(
|
||||
workerPath,
|
||||
|
|
@ -183,7 +151,6 @@ describe('parse-impl worker pool lazy startup', () => {
|
|||
Date.now(),
|
||||
() => {},
|
||||
{
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
workerUrlForTest: pathToFileURL(workerPath),
|
||||
workerPoolSize: 1,
|
||||
parseCache,
|
||||
|
|
@ -223,7 +190,6 @@ describe('parse-impl worker pool lazy startup', () => {
|
|||
Date.now(),
|
||||
() => {},
|
||||
{
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
workerUrlForTest: pathToFileURL(workerPath),
|
||||
workerPoolSize: 1,
|
||||
parseCache,
|
||||
|
|
@ -262,7 +228,6 @@ describe('parse-impl worker pool lazy startup', () => {
|
|||
Date.now(),
|
||||
() => {},
|
||||
{
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
workerUrlForTest: pathToFileURL(workerPath),
|
||||
workerPoolSize: 1,
|
||||
// No flag: a total worker-startup failure always fails fast now.
|
||||
|
|
@ -275,7 +240,7 @@ describe('parse-impl worker pool lazy startup', () => {
|
|||
expect(Array.from(graph.nodes.values()).some((n) => n.properties.name === 'fatal')).toBe(false);
|
||||
});
|
||||
|
||||
it('parses sequentially when GITNEXUS_WORKER_POOL_SIZE=0 and no --workers flag (#1741)', async () => {
|
||||
it('throws when GITNEXUS_WORKER_POOL_SIZE=0 and no --workers flag (sequential parsing removed)', async () => {
|
||||
const saved = process.env.GITNEXUS_WORKER_POOL_SIZE;
|
||||
process.env.GITNEXUS_WORKER_POOL_SIZE = '0';
|
||||
try {
|
||||
|
|
@ -286,31 +251,31 @@ describe('parse-impl worker pool lazy startup', () => {
|
|||
fs.writeFileSync(full, content);
|
||||
|
||||
// A ready-worker double with a spawn marker — it must NOT be spawned,
|
||||
// because env=0 routes to the sequential path before any pool is built.
|
||||
// because the disabled-channel validation throws before any pool is built.
|
||||
const markerPath = path.join(tempDir, 'env0-worker.marker');
|
||||
const workerPath = path.join(tempDir, 'env0-ready-worker.js');
|
||||
writeReadyWorker(workerPath, markerPath);
|
||||
|
||||
const graph = createKnowledgeGraph();
|
||||
const result = await runChunkedParseAndResolve(
|
||||
graph,
|
||||
[{ path: rel, size: fs.statSync(full).size }],
|
||||
[rel],
|
||||
1,
|
||||
repoDir,
|
||||
Date.now(),
|
||||
() => {},
|
||||
{
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
workerUrlForTest: pathToFileURL(workerPath),
|
||||
// No workerPoolSize option — the env var is the only sizing signal.
|
||||
},
|
||||
);
|
||||
await expect(
|
||||
runChunkedParseAndResolve(
|
||||
graph,
|
||||
[{ path: rel, size: fs.statSync(full).size }],
|
||||
[rel],
|
||||
1,
|
||||
repoDir,
|
||||
Date.now(),
|
||||
() => {},
|
||||
{
|
||||
workerUrlForTest: pathToFileURL(workerPath),
|
||||
// No workerPoolSize option — the ambient env=0 is the only signal.
|
||||
},
|
||||
),
|
||||
).rejects.toThrow(/GITNEXUS_WORKER_POOL_SIZE=0/);
|
||||
|
||||
expect(result.usedWorkerPool).toBe(false); // env=0 → sequential, not a size-0 pool fail-fast
|
||||
expect(fs.existsSync(markerPath)).toBe(false); // no worker ever spawned
|
||||
// Sequential parsing still produced a complete graph for the file.
|
||||
expect(Array.from(graph.nodes.values()).some((n) => n.properties.name === 'env0')).toBe(true);
|
||||
// Sequential parsing was removed: env=0 is a hard error, not a silent
|
||||
// sequential run. The validation throws before any pool is constructed.
|
||||
expect(fs.existsSync(markerPath)).toBe(false);
|
||||
} finally {
|
||||
if (saved === undefined) delete process.env.GITNEXUS_WORKER_POOL_SIZE;
|
||||
else process.env.GITNEXUS_WORKER_POOL_SIZE = saved;
|
||||
|
|
@ -341,7 +306,6 @@ describe('parse-impl worker pool lazy startup', () => {
|
|||
Date.now(),
|
||||
() => {},
|
||||
{
|
||||
workerThresholdsForTest: { minFiles: 1, minBytes: 1 },
|
||||
workerUrlForTest: pathToFileURL(workerPath),
|
||||
workerPoolSize: 1, // explicit --workers 1 must win over ambient env=0
|
||||
},
|
||||
|
|
@ -354,76 +318,4 @@ describe('parse-impl worker pool lazy startup', () => {
|
|||
else process.env.GITNEXUS_WORKER_POOL_SIZE = saved;
|
||||
}
|
||||
});
|
||||
|
||||
it('threads exportedTypeMap through the sequential path: a no-worker run over a partially-warm cache keeps the sequential-miss chunk exported types (#2038)', async () => {
|
||||
const saved = process.env.GITNEXUS_WORKER_POOL_SIZE;
|
||||
process.env.GITNEXUS_WORKER_POOL_SIZE = '0'; // force the no-worker (sequential) path
|
||||
try {
|
||||
// Chunk A — cache HIT, pre-seeded with an exported typed symbol so the
|
||||
// replay makes exportedTypeMap.size > 0 and the size===0 rebuild is skipped.
|
||||
const relA = 'src/a_hit.ts';
|
||||
const contentA = 'export function aWidget(): number { return 2; }\n';
|
||||
const fullA = path.join(repoDir, relA);
|
||||
fs.mkdirSync(path.dirname(fullA), { recursive: true });
|
||||
fs.writeFileSync(fullA, contentA);
|
||||
|
||||
// Chunk B — cache MISS, parsed sequentially for real; exported + typed.
|
||||
const relB = 'src/b_miss.ts';
|
||||
const contentB = 'export function bWidget(): number { return 1; }\n';
|
||||
const fullB = path.join(repoDir, relB);
|
||||
fs.writeFileSync(fullB, contentB);
|
||||
|
||||
const chunkHashA = computeChunkHash([
|
||||
{ filePath: relA, contentHash: fileContentHash(contentA) },
|
||||
]);
|
||||
const parseCache = {
|
||||
version: 'test',
|
||||
entries: new Map<string, ParseWorkerResult[]>([
|
||||
[chunkHashA, [exportedTypedResult(relA, 'aWidget', 'number')]],
|
||||
]),
|
||||
usedKeys: new Set<string>(),
|
||||
};
|
||||
|
||||
const graph = createKnowledgeGraph();
|
||||
const result = await runChunkedParseAndResolve(
|
||||
graph,
|
||||
[
|
||||
{ path: relA, size: fs.statSync(fullA).size },
|
||||
{ path: relB, size: fs.statSync(fullB).size },
|
||||
],
|
||||
[relA, relB],
|
||||
2,
|
||||
repoDir,
|
||||
Date.now(),
|
||||
() => {},
|
||||
{
|
||||
// 1-byte budget → each file is its own chunk, so A hits while B misses.
|
||||
chunkByteBudget: 1,
|
||||
parseCache,
|
||||
},
|
||||
);
|
||||
|
||||
expect(result.usedWorkerPool).toBe(false);
|
||||
// Sanity: the cache-hit chunk populated the map — this is what makes
|
||||
// size > 0 and suppresses the full-graph rebuild on the size===0 guard.
|
||||
expect(result.exportedTypeMap.get(relA)?.get('aWidget')).toBe('number');
|
||||
// Regression (#2038): the sequential-miss chunk's exported type must
|
||||
// survive. Fails on pre-fix HEAD — the sequential path never populated
|
||||
// exportedTypeMap, and the rebuild was skipped because the hit made size > 0.
|
||||
expect(result.exportedTypeMap.get(relB)?.get('bWidget')).toBe('number');
|
||||
|
||||
// Differential oracle: the threaded map must match a fresh full-graph build
|
||||
// (both directions for the entries under test) on the actual mixed path.
|
||||
const oracle = buildExportedTypeMapFromGraph(graph, result.model.symbols);
|
||||
expect(result.exportedTypeMap.get(relB)?.get('bWidget')).toBe(
|
||||
oracle.get(relB)?.get('bWidget'),
|
||||
);
|
||||
expect(result.exportedTypeMap.get(relA)?.get('aWidget')).toBe(
|
||||
oracle.get(relA)?.get('aWidget'),
|
||||
);
|
||||
} finally {
|
||||
if (saved === undefined) delete process.env.GITNEXUS_WORKER_POOL_SIZE;
|
||||
else process.env.GITNEXUS_WORKER_POOL_SIZE = saved;
|
||||
}
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -5,13 +5,12 @@
|
|||
* bounded, jittered restart loop (see worker-pool.ts). `handleWorkerStartupFailure`
|
||||
* is reached only when that self-heal is EXHAUSTED, a deterministic crash-loop
|
||||
* was detected, or the pool could not be constructed — i.e. the workers truly
|
||||
* cannot start. In every such case it FAILS FAST with the captured cause,
|
||||
* rather than silently degrading to the ~10× slower sequential parser (which
|
||||
* masked a worker-startup regression as a 123-minute "stuck" run in #1741).
|
||||
*
|
||||
* The decision is automatic and flag-free: there is no `--allow-sequential-fallback`
|
||||
* and no dependence on how the pool was sized. An operator who genuinely wants
|
||||
* sequential parsing passes `--workers 0`.
|
||||
* cannot start. In every such case it FAILS FAST with the captured cause.
|
||||
* There is no sequential parser to silently degrade to — that fallback was
|
||||
* removed (it had masked a worker-startup regression as a 123-minute "stuck"
|
||||
* run in #1741). The message names the fix (the worker startup / construction
|
||||
* error) rather than offering a `--workers 0` escape hatch, which no longer
|
||||
* exists.
|
||||
*/
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { handleWorkerStartupFailure } from '../../src/core/ingestion/pipeline-phases/parse-impl.js';
|
||||
|
|
@ -49,7 +48,10 @@ describe('handleWorkerStartupFailure — always fail fast with the cause (#1741)
|
|||
);
|
||||
expect(message).toMatch(/deterministic/i);
|
||||
expect(message).toContain('tree-sitter-c-sharp'); // captured stderr propagated
|
||||
expect(message).toContain('--workers 0'); // the explicit sequential escape hatch
|
||||
// Sequential parsing was removed: the message no longer offers a `--workers 0`
|
||||
// escape hatch — it names the fix instead (fail clearly, no silent fallback).
|
||||
expect(message).not.toContain('--workers 0');
|
||||
expect(message).toContain('Fix the worker startup failure');
|
||||
// The native-binding hint is apt for an init crash and is kept (R7).
|
||||
expect(message).toMatch(/native binding/i);
|
||||
});
|
||||
|
|
@ -58,7 +60,8 @@ describe('handleWorkerStartupFailure — always fail fast with the cause (#1741)
|
|||
const message = messageFrom(() => handleWorkerStartupFailure(initError('transient-exhausted')));
|
||||
expect(message).toMatch(/exhausted the bounded startup retry budget/i);
|
||||
expect(message).toContain('tree-sitter-c-sharp');
|
||||
expect(message).toContain('--workers 0');
|
||||
expect(message).not.toContain('--workers 0');
|
||||
expect(message).toContain('Fix the worker startup failure');
|
||||
});
|
||||
|
||||
it('throws on a pool *construction* failure (plain Error, not an init error)', () => {
|
||||
|
|
@ -66,7 +69,8 @@ describe('handleWorkerStartupFailure — always fail fast with the cause (#1741)
|
|||
handleWorkerStartupFailure(new Error('Worker script not found: /tmp/parse-worker.js')),
|
||||
);
|
||||
expect(message).toMatch(/could not be constructed/i);
|
||||
expect(message).toContain('--workers 0');
|
||||
expect(message).not.toContain('--workers 0');
|
||||
expect(message).toContain('Fix the worker pool construction error');
|
||||
// The real construction error is surfaced verbatim (R7) …
|
||||
expect(message).toContain('Worker script not found: /tmp/parse-worker.js');
|
||||
// … and the native-binding guess is NOT applied to a construction failure.
|
||||
|
|
|
|||
236
gitnexus/test/unit/parsedfile-store.test.ts
Normal file
236
gitnexus/test/unit/parsedfile-store.test.ts
Normal file
|
|
@ -0,0 +1,236 @@
|
|||
import { describe, it, expect } from 'vitest';
|
||||
import { mkdtemp, rm, readdir, readFile } from 'fs/promises';
|
||||
import { tmpdir } from 'os';
|
||||
import path from 'path';
|
||||
import type { ParsedFile } from 'gitnexus-shared';
|
||||
import {
|
||||
clearParsedFileStore,
|
||||
persistParsedFileChunk,
|
||||
persistParsedFileShardSync,
|
||||
loadParsedFilesForPaths,
|
||||
getParsedFileStoreDir,
|
||||
} from '../../src/storage/parsedfile-store.js';
|
||||
|
||||
/**
|
||||
* Build a minimal ParsedFile whose Scope carries `bindings` / `typeBindings`
|
||||
* Maps — the round-trip's fidelity hinges on those Maps surviving JSON
|
||||
* serialization (they would otherwise collapse to `{}`).
|
||||
*/
|
||||
const makeParsedFile = (filePath: string): ParsedFile =>
|
||||
({
|
||||
filePath,
|
||||
moduleScope: `${filePath}:module`,
|
||||
parsedImports: [],
|
||||
localDefs: [
|
||||
{ nodeId: `Function:${filePath}:fn`, filePath, type: 'Function', qualifiedName: 'fn' },
|
||||
],
|
||||
referenceSites: [],
|
||||
scopes: [
|
||||
{
|
||||
id: `${filePath}:module`,
|
||||
parent: null,
|
||||
kind: 'Module',
|
||||
range: { startLine: 1, startCol: 0, endLine: 9, endCol: 0 },
|
||||
filePath,
|
||||
bindings: new Map([['fn', [{ defId: `Function:${filePath}:fn`, origin: 'local' }]]]),
|
||||
ownedDefs: [],
|
||||
imports: [],
|
||||
typeBindings: new Map([['x', { name: 'int' }]]),
|
||||
},
|
||||
],
|
||||
}) as unknown as ParsedFile;
|
||||
|
||||
describe('parsedfile-store', () => {
|
||||
it('round-trips ParsedFiles (incl. Scope Maps) and filters by requested paths', async () => {
|
||||
const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-'));
|
||||
try {
|
||||
await persistParsedFileChunk(dir, 'chunk-0', [makeParsedFile('a.c'), makeParsedFile('b.c')]);
|
||||
await persistParsedFileChunk(dir, 'chunk-1', [makeParsedFile('c.c')]);
|
||||
|
||||
// Filtering: only requested paths come back.
|
||||
const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c', 'c.c']));
|
||||
expect([...loaded.keys()].sort()).toEqual(['a.c', 'c.c']);
|
||||
expect(loaded.has('b.c')).toBe(false);
|
||||
|
||||
// Map fidelity: bindings / typeBindings survive as real Maps.
|
||||
const a = loaded.get('a.c')!;
|
||||
const scope = a.scopes[0];
|
||||
expect(scope.bindings).toBeInstanceOf(Map);
|
||||
expect(scope.bindings.get('fn')?.[0]?.defId).toBe('Function:a.c:fn');
|
||||
expect(scope.typeBindings).toBeInstanceOf(Map);
|
||||
expect((scope.typeBindings.get('x') as { name: string }).name).toBe('int');
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it('writes no shard for an empty chunk', async () => {
|
||||
const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-'));
|
||||
try {
|
||||
await persistParsedFileChunk(dir, 'chunk-empty', []);
|
||||
let shardCount = 0;
|
||||
try {
|
||||
shardCount = (await readdir(getParsedFileStoreDir(dir))).length;
|
||||
} catch {
|
||||
shardCount = 0; // dir not created — also fine
|
||||
}
|
||||
expect(shardCount).toBe(0);
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it('clearParsedFileStore removes all shards (subsequent load is empty)', async () => {
|
||||
const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-'));
|
||||
try {
|
||||
await persistParsedFileChunk(dir, 'chunk-0', [makeParsedFile('a.c')]);
|
||||
await clearParsedFileStore(dir);
|
||||
const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c']));
|
||||
expect(loaded.size).toBe(0);
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it('returns empty map when the store is absent', async () => {
|
||||
const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-'));
|
||||
try {
|
||||
const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c']));
|
||||
expect(loaded.size).toBe(0);
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
// #1983 parallel serialization: the sync worker writer and the async writer
|
||||
// share one serialization core and MUST produce byte-identical shards (the
|
||||
// loader's deep-equals masks byte drift, so assert raw bytes).
|
||||
it('persistParsedFileShardSync writes byte-identical shards to the async writer', async () => {
|
||||
const asyncDir = await mkdtemp(path.join(tmpdir(), 'pfstore-a-'));
|
||||
const syncDir = await mkdtemp(path.join(tmpdir(), 'pfstore-s-'));
|
||||
try {
|
||||
const files = [makeParsedFile('a.c'), makeParsedFile('b.c')];
|
||||
await persistParsedFileChunk(asyncDir, 'shard', files);
|
||||
persistParsedFileShardSync(syncDir, 'shard', files);
|
||||
const asyncBytes = await readFile(
|
||||
path.join(getParsedFileStoreDir(asyncDir), 'shard.json'),
|
||||
'utf-8',
|
||||
);
|
||||
const syncBytes = await readFile(
|
||||
path.join(getParsedFileStoreDir(syncDir), 'shard.json'),
|
||||
'utf-8',
|
||||
);
|
||||
expect(syncBytes).toBe(asyncBytes);
|
||||
} finally {
|
||||
await rm(asyncDir, { recursive: true, force: true });
|
||||
await rm(syncDir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it('persistParsedFileShardSync round-trips through loadParsedFilesForPaths with Maps intact', async () => {
|
||||
const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-'));
|
||||
try {
|
||||
persistParsedFileShardSync(dir, 'w1-0', [makeParsedFile('a.c')]);
|
||||
const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c']));
|
||||
const scope = loaded.get('a.c')!.scopes[0];
|
||||
expect(scope.bindings).toBeInstanceOf(Map);
|
||||
expect(scope.bindings.get('fn')?.[0]?.defId).toBe('Function:a.c:fn');
|
||||
expect(scope.typeBindings).toBeInstanceOf(Map);
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
// #1983 capture side-channel: a ParsedFile may carry a plain-data
|
||||
// `captureSideChannel` (e.g. C++ ADL / namespace / two-phase marks the worker
|
||||
// computed). It MUST survive the JSON store round-trip so the main thread can
|
||||
// restore those module maps WITHOUT a re-parse. Plain objects/arrays only —
|
||||
// no Maps/Sets — so the interning reviver passes them through unchanged.
|
||||
it('round-trips a ParsedFile.captureSideChannel (plain data) through the store', async () => {
|
||||
const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-'));
|
||||
try {
|
||||
const sideChannel = {
|
||||
adl: {
|
||||
argInfoBySite: [
|
||||
[
|
||||
6,
|
||||
4,
|
||||
[
|
||||
{
|
||||
simpleClassName: 'Event',
|
||||
templateSimpleClassName: '',
|
||||
templateNamespace: '',
|
||||
templateArgClassNames: [],
|
||||
templateArgNamespaces: [],
|
||||
},
|
||||
],
|
||||
],
|
||||
],
|
||||
noAdlSites: [[9, 2]],
|
||||
},
|
||||
inlineNamespaceRanges: ['1:0:3:1'],
|
||||
fileLocal: {
|
||||
fileLocalNames: ['helper'],
|
||||
anonymousNamespaceRanges: ['4:0:6:1'],
|
||||
},
|
||||
twoPhase: {
|
||||
dependentBases: [['Derived', [['Base', ['detail']]]]],
|
||||
dependentPackBaseClasses: ['Mix'],
|
||||
},
|
||||
};
|
||||
const pf = {
|
||||
...(makeParsedFile('app.cpp') as unknown as Record<string, unknown>),
|
||||
captureSideChannel: sideChannel,
|
||||
} as unknown as ParsedFile;
|
||||
|
||||
persistParsedFileShardSync(dir, 'w1-0', [pf]);
|
||||
const loaded = await loadParsedFilesForPaths(dir, new Set(['app.cpp']));
|
||||
const got = loaded.get('app.cpp')!;
|
||||
// Deep-equal: the plain-data snapshot survives byte-for-byte (after JSON).
|
||||
expect((got as { captureSideChannel?: unknown }).captureSideChannel).toEqual(sideChannel);
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
// #1983 (Kotlin): the kotlin provider carries a self-describing companion-
|
||||
// scope side-channel `{ kind: 'kotlin', companionScopes: ScopeId[] }`. It
|
||||
// shares the single generic `captureSideChannel` field with C++, so confirm
|
||||
// the (Set→array) plain-data shape survives the JSON store round-trip too.
|
||||
it('round-trips a Kotlin ParsedFile.captureSideChannel through the store', async () => {
|
||||
const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-'));
|
||||
try {
|
||||
const sideChannel = {
|
||||
kind: 'kotlin',
|
||||
companionScopes: ['scope:Logger.companion', 'scope:Animal.companion'],
|
||||
};
|
||||
const pf = {
|
||||
...(makeParsedFile('App.kt') as unknown as Record<string, unknown>),
|
||||
captureSideChannel: sideChannel,
|
||||
} as unknown as ParsedFile;
|
||||
|
||||
persistParsedFileShardSync(dir, 'w1-0', [pf]);
|
||||
const loaded = await loadParsedFilesForPaths(dir, new Set(['App.kt']));
|
||||
const got = loaded.get('App.kt')!;
|
||||
expect((got as { captureSideChannel?: unknown }).captureSideChannel).toEqual(sideChannel);
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it('persistParsedFileShardSync writes no shard and no directory for empty input', async () => {
|
||||
const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-'));
|
||||
try {
|
||||
persistParsedFileShardSync(dir, 'w1-0', []);
|
||||
let entries: string[] = [];
|
||||
try {
|
||||
entries = await readdir(getParsedFileStoreDir(dir));
|
||||
} catch {
|
||||
entries = []; // store dir not created — the expected parity with the async writer
|
||||
}
|
||||
expect(entries).toHaveLength(0);
|
||||
} finally {
|
||||
await rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
});
|
||||
|
|
@ -50,9 +50,8 @@ describe('processParsing — worker-pool error propagation (U20)', () => {
|
|||
[{ path: 'src/a.ts', content: 'export function a() { return 1; }\n' }],
|
||||
createSymbolTable(),
|
||||
createASTCache(),
|
||||
createASTCache(),
|
||||
() => {},
|
||||
workerPool,
|
||||
() => {},
|
||||
),
|
||||
).rejects.toThrow('replacement worker failed');
|
||||
|
||||
|
|
@ -83,14 +82,16 @@ describe('processParsing — worker-pool error propagation (U20)', () => {
|
|||
],
|
||||
createSymbolTable(),
|
||||
createASTCache(),
|
||||
createASTCache(),
|
||||
() => {},
|
||||
workerPool,
|
||||
() => {},
|
||||
);
|
||||
|
||||
await expect(rejection).rejects.toBeInstanceOf(WorkerPoolDispatchError);
|
||||
const err = await rejection.catch((e) => e as WorkerPoolDispatchError);
|
||||
expect(err.quarantinedPaths).toEqual(['src/poison.ts']);
|
||||
const err = await rejection.then(
|
||||
() => undefined,
|
||||
(e: unknown) => e as WorkerPoolDispatchError,
|
||||
);
|
||||
expect(err?.quarantinedPaths).toEqual(['src/poison.ts']);
|
||||
|
||||
// No sequential fallback ran for either file. The caller (analyze
|
||||
// entry point) is responsible for surfacing this as a hard
|
||||
|
|
@ -129,61 +130,16 @@ describe('processParsing — worker-pool error propagation (U20)', () => {
|
|||
],
|
||||
createSymbolTable(),
|
||||
createASTCache(),
|
||||
createASTCache(),
|
||||
workerPool,
|
||||
(_current, _total, detail) => {
|
||||
progressDetails.push(detail);
|
||||
},
|
||||
workerPool,
|
||||
);
|
||||
|
||||
// Worker path returned successfully (not null — null was the
|
||||
// pre-U20 sentinel for "ran sequential fallback"). The progress
|
||||
// log surfaces the quarantine count for operator visibility.
|
||||
// Worker path returned the merged chunk data (never null — the pre-U20
|
||||
// null sentinel meant "ran sequential fallback", which no longer exists).
|
||||
// The progress log surfaces the quarantine count for operator visibility.
|
||||
expect(result).not.toBeNull();
|
||||
expect(progressDetails).toContain('1 worker-quarantined file(s) skipped');
|
||||
});
|
||||
});
|
||||
|
||||
describe('TypeScript object literal method exports', () => {
|
||||
it('links exported object literal shorthand methods back to the exported object', async () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
|
||||
await processParsing(
|
||||
graph,
|
||||
[
|
||||
{
|
||||
path: 'src/foo.ts',
|
||||
content: `export const fooService = {
|
||||
async getUser(id: string) {
|
||||
return findUser(id);
|
||||
},
|
||||
saveUser(user: User) {
|
||||
return persist(user);
|
||||
},
|
||||
};
|
||||
`,
|
||||
},
|
||||
],
|
||||
createSymbolTable(),
|
||||
createASTCache(),
|
||||
createASTCache(),
|
||||
);
|
||||
|
||||
const service = graph.nodes.find(
|
||||
(node) => node.label === 'Const' && node.properties.name === 'fooService',
|
||||
);
|
||||
expect(service, 'exported object literal should be captured as a Const').toBeDefined();
|
||||
|
||||
const methodNames = new Set(
|
||||
graph.nodes.filter((node) => node.label === 'Method').map((node) => node.properties.name),
|
||||
);
|
||||
expect(methodNames).toEqual(new Set(['getUser', 'saveUser']));
|
||||
|
||||
const linkedMethodNames = graph.relationships
|
||||
.filter((rel) => rel.type === 'HAS_METHOD' && rel.sourceId === service!.id)
|
||||
.map((rel) => graph.getNode(rel.targetId)?.properties.name)
|
||||
.sort();
|
||||
|
||||
expect(linkedMethodNames).toEqual(['getUser', 'saveUser']);
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -1,86 +0,0 @@
|
|||
import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest';
|
||||
|
||||
vi.mock('../../src/core/tree-sitter/parser-loader.js', () => ({
|
||||
loadParser: vi.fn(async () => ({
|
||||
parse: vi.fn(),
|
||||
getLanguage: vi.fn(),
|
||||
})),
|
||||
loadLanguage: vi.fn(async () => undefined),
|
||||
isLanguageAvailable: vi.fn(() => true),
|
||||
}));
|
||||
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import { createASTCache } from '../../src/core/ingestion/ast-cache.js';
|
||||
import { processParsing } from '../../src/core/ingestion/parsing-processor.js';
|
||||
import { createSymbolTable } from '../../src/core/ingestion/model/symbol-table.js';
|
||||
import * as parserLoader from '../../src/core/tree-sitter/parser-loader.js';
|
||||
|
||||
import { _captureLogger } from '../../src/core/logger.js';
|
||||
import type { LoggerCapture } from '../../src/core/logger.js';
|
||||
describe('sequential native parser availability', () => {
|
||||
// Hoisted so a stray live capture from a failed warn test can always be
|
||||
// torn down in afterEach — otherwise a single assertion failure cascades
|
||||
// into `_captureLogger: a previous capture is still active` (logger.ts).
|
||||
let cap: LoggerCapture | undefined;
|
||||
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
cap?.restore();
|
||||
cap = undefined;
|
||||
});
|
||||
|
||||
it('skips Swift files in processParsing when the native parser is unavailable', async () => {
|
||||
vi.mocked(parserLoader.isLanguageAvailable).mockReturnValue(false);
|
||||
|
||||
await expect(
|
||||
processParsing(
|
||||
createKnowledgeGraph(),
|
||||
[{ path: 'App.swift', content: 'class AppViewController: UIViewController {}' }],
|
||||
createSymbolTable(),
|
||||
createASTCache(),
|
||||
),
|
||||
).resolves.toBeNull();
|
||||
|
||||
expect(parserLoader.loadLanguage).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('warns when processParsing skips files in verbose mode', async () => {
|
||||
cap = _captureLogger();
|
||||
const previous = process.env.GITNEXUS_VERBOSE;
|
||||
process.env.GITNEXUS_VERBOSE = '1';
|
||||
try {
|
||||
vi.mocked(parserLoader.isLanguageAvailable).mockReturnValue(false);
|
||||
|
||||
await processParsing(
|
||||
createKnowledgeGraph(),
|
||||
[{ path: 'App.swift', content: 'class AppViewController: UIViewController {}' }],
|
||||
createSymbolTable(),
|
||||
createASTCache(),
|
||||
);
|
||||
|
||||
expect(
|
||||
cap
|
||||
.records()
|
||||
.some(
|
||||
(r) =>
|
||||
r.msg ===
|
||||
'[ingestion] Skipped 1 swift file(s) in parsing processing — swift parser not available.',
|
||||
),
|
||||
).toBe(true);
|
||||
} finally {
|
||||
// Always restore the live capture here (in addition to the afterEach
|
||||
// safety net) so a failing assertion above cannot leak it into the
|
||||
// next test as an "a previous capture is still active" cascade.
|
||||
cap.restore();
|
||||
cap = undefined;
|
||||
if (previous === undefined) {
|
||||
delete process.env.GITNEXUS_VERBOSE;
|
||||
} else {
|
||||
process.env.GITNEXUS_VERBOSE = previous;
|
||||
}
|
||||
}
|
||||
});
|
||||
});
|
||||
Loading…
Add table
Reference in a new issue